tickmarkr 2.5.6 → 2.5.8
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/adapters/prompt.js +21 -1
- package/dist/cli/commands/approve.js +13 -6
- package/dist/cli/commands/compile.js +7 -0
- package/dist/cli/commands/plan.d.ts +5 -0
- package/dist/cli/commands/plan.js +28 -23
- package/dist/cli/commands/resume.js +1 -1
- package/dist/cli/commands/run.js +1 -1
- package/dist/cli/commands/status.js +95 -34
- package/dist/cli/commands/verify.js +6 -4
- package/dist/compile/native.js +68 -7
- package/dist/compile/retired-literals.d.ts +22 -0
- package/dist/compile/retired-literals.js +271 -0
- package/dist/drivers/index.d.ts +4 -2
- package/dist/drivers/index.js +54 -6
- package/dist/gates/baseline.d.ts +8 -3
- package/dist/gates/baseline.js +6 -3
- package/dist/gates/review.js +12 -1
- package/dist/gates/run-gates.d.ts +6 -1
- package/dist/gates/run-gates.js +67 -9
- package/dist/gates/test-manifest.d.ts +12 -0
- package/dist/gates/test-manifest.js +29 -3
- package/dist/gates/test-reporter.js +28 -2
- package/dist/graph/graph.d.ts +2 -0
- package/dist/graph/graph.js +5 -0
- package/dist/graph/schema.d.ts +28 -0
- package/dist/graph/schema.js +13 -1
- package/dist/run/activity.d.ts +28 -0
- package/dist/run/activity.js +194 -0
- package/dist/run/daemon.d.ts +28 -0
- package/dist/run/daemon.js +466 -110
- package/dist/run/git.d.ts +6 -1
- package/dist/run/git.js +63 -11
- package/dist/run/journal.d.ts +57 -5
- package/dist/run/journal.js +103 -9
- package/dist/run/merge.d.ts +2 -0
- package/dist/run/merge.js +1 -0
- package/dist/run/operator-page-summary.d.ts +56 -0
- package/dist/run/operator-page-summary.js +68 -0
- package/dist/run/operator-summary.d.ts +69 -0
- package/dist/run/operator-summary.js +77 -0
- package/dist/run/protocol.d.ts +71 -0
- package/dist/run/protocol.js +32 -0
- package/dist/run/repair-disposition.d.ts +41 -0
- package/dist/run/repair-disposition.js +77 -0
- package/dist/tui/cockpit/board.d.ts +9 -0
- package/dist/tui/cockpit/board.js +10 -0
- package/dist/tui/cockpit/derive.d.ts +35 -0
- package/dist/tui/cockpit/derive.js +152 -10
- package/dist/tui/cockpit/evidence-view.d.ts +2 -0
- package/dist/tui/cockpit/evidence-view.js +42 -12
- package/dist/tui/cockpit/home-view.js +45 -30
- package/dist/tui/cockpit/live-store.d.ts +18 -0
- package/dist/tui/cockpit/run-cockpit.d.ts +8 -1
- package/dist/tui/cockpit/run-cockpit.js +95 -1
- package/dist/tui/cockpit/run-view.d.ts +37 -0
- package/dist/tui/cockpit/run-view.js +189 -2
- package/package.json +1 -1
- package/schema/rungraph.schema.json +54 -0
- package/skills/tickmarkr-overseer/SKILL.md +162 -3
|
@@ -2,7 +2,11 @@ import { jsx as _jsx, jsxs as _jsxs, Fragment as _Fragment } from "react/jsx-run
|
|
|
2
2
|
import { Box } from "ink";
|
|
3
3
|
import { useContext } from "react";
|
|
4
4
|
import { GLYPHS } from "../../brand.js";
|
|
5
|
+
import { formatOwnedName, parseOwnedName } from "../../drivers/types.js";
|
|
5
6
|
import { GATE_NAMES } from "../../graph/schema.js";
|
|
7
|
+
import { projectActivity } from "../../run/activity.js";
|
|
8
|
+
import { projectOperatorSummary } from "../../run/operator-summary.js";
|
|
9
|
+
import { trackJournalRows } from "../../run/protocol.js";
|
|
6
10
|
import { normalizeGateOutcome } from "../../run/outcome.js";
|
|
7
11
|
import { BOARD_KEYS, boardFooter, clipBoard, renderBoardLines } from "./board.js";
|
|
8
12
|
import { BodyText, Panel, ShellTheme } from "./components.js";
|
|
@@ -172,6 +176,185 @@ export function runSummaryLine(s) {
|
|
|
172
176
|
}
|
|
173
177
|
const attemptLabel = (t) => (t.attempt !== undefined ? `attempt ${t.attempt}` : t.dispatches === 0 ? "never dispatched" : "attempt unknown");
|
|
174
178
|
const parkOf = (t) => (t.state === "human" ? `human · ${t.parkKind ?? "unknown kind"}` : t.state);
|
|
179
|
+
export const STALL_MARKER = "⚠ stalled harvest suspected";
|
|
180
|
+
const ordinal = (v) => (typeof v === "number" && Number.isInteger(v) && v >= 0 ? v : undefined);
|
|
181
|
+
const slotNameOf = (data) => {
|
|
182
|
+
const slot = data.slot;
|
|
183
|
+
return str(data.slotName) ?? (typeof slot === "object" && slot !== null ? str(slot.name) : str(slot));
|
|
184
|
+
};
|
|
185
|
+
const NO_ROW = "no journal row";
|
|
186
|
+
const readLabel = (f) => `${f.label} ${f.line === undefined ? `(${NO_ROW})` : `#L${f.line}`}`;
|
|
187
|
+
/**
|
|
188
|
+
* The evidence-only projection status renders (run/activity.ts, run/operator-summary.ts), folded
|
|
189
|
+
* over the journal rows this view was handed, with the physical line each reading was taken from.
|
|
190
|
+
* Nothing is inferred from a title, a provider or a task id: an unrecorded field says so.
|
|
191
|
+
*/
|
|
192
|
+
export function projectRunTasks(snapshot, rows, graph, decisions, runId) {
|
|
193
|
+
const ordered = [...rows].filter((r) => r.event !== undefined).sort((a, b) => a.line - b.line);
|
|
194
|
+
const graphTask = (id) => graph?.tasks.find((t) => t.id === id);
|
|
195
|
+
const activityTasks = snapshot.tasks.map((t) => ({ id: t.id, gates: graphTask(t.id)?.gates ?? [...GATE_NAMES], deps: graphTask(t.id)?.deps ?? [], status: t.state }));
|
|
196
|
+
const tracked = trackJournalRows(runId, ordered.map((r) => ({ raw: r.event, sourceIndex: r.line })));
|
|
197
|
+
const activity = projectActivity(runId, tracked, activityTasks);
|
|
198
|
+
const byTask = (id) => ordered.filter((r) => r.event?.taskId === id);
|
|
199
|
+
const summaries = new Map(projectOperatorSummary(snapshot.tasks.map((t) => {
|
|
200
|
+
const own = byTask(t.id);
|
|
201
|
+
const responsibility = [...own].reverse().find((r) => str(r.event.data.role) !== undefined || str(r.event.data.agent) !== undefined);
|
|
202
|
+
// The summary's own vocabulary (derive.ts reads the graph status for an unmentioned task): a task the
|
|
203
|
+
// operator fold holds as blocked, or a graph task the journal never mentions, is pending on its prerequisites.
|
|
204
|
+
const status = t.state === "blocked" || (t.state === "unknown" && graphTask(t.id) !== undefined) ? "pending" : t.state;
|
|
205
|
+
return {
|
|
206
|
+
id: t.id, status, deps: graphTask(t.id)?.deps ?? [],
|
|
207
|
+
...(own.length === 0 ? {} : { phase: activity.get(t.id)?.state, lastEvidenceAt: own.at(-1).event.ts }),
|
|
208
|
+
...(responsibility ? { responsible: { role: str(responsibility.event.data.role), agent: str(responsibility.event.data.agent) } } : {}),
|
|
209
|
+
};
|
|
210
|
+
}), decisions.map((d) => ({ taskId: d.taskId, park: { kind: d.park.kind, tombstone: d.park.tombstone, ...(d.park.reason === undefined ? {} : { reason: d.park.reason }) }, verbs: d.verbs, ...(d.diagnostic === undefined ? {} : { diagnostic: d.diagnostic }) }))).map((s) => [s.taskId, s]));
|
|
211
|
+
return snapshot.tasks.map((t) => {
|
|
212
|
+
const own = byTask(t.id);
|
|
213
|
+
const act = activity.get(t.id);
|
|
214
|
+
const summary = summaries.get(t.id);
|
|
215
|
+
// identity: the newest row that recorded a role, an agent, an owned slot name or a dispatch assignment
|
|
216
|
+
const identityRow = [...own].reverse().find((r) => { const d = r.event.data; if (act?.attempt !== undefined && ordinal(d.attempt) !== undefined && ordinal(d.attempt) !== act.attempt)
|
|
217
|
+
return false; return str(d.role) !== undefined || str(d.agent) !== undefined || slotNameOf(d) !== undefined || (r.event.event === "task-dispatch" && d.assignment !== undefined); });
|
|
218
|
+
let identity = { label: "identity unrecorded" };
|
|
219
|
+
if (identityRow) {
|
|
220
|
+
const d = identityRow.event.data;
|
|
221
|
+
const name = slotNameOf(d);
|
|
222
|
+
const owned = name === undefined ? null : parseOwnedName(name);
|
|
223
|
+
const role = str(d.role) ?? owned?.role ?? (identityRow.event.event === "worker-launch" ? "worker" : "role unrecorded");
|
|
224
|
+
const a = d.assignment;
|
|
225
|
+
const agent = str(d.agent) ?? name ?? (typeof a?.adapter === "string" && typeof a.model === "string" ? `${a.adapter}:${a.model}` : "agent unrecorded");
|
|
226
|
+
identity = { label: `${role} ${agent}`, line: identityRow.line };
|
|
227
|
+
}
|
|
228
|
+
// Finding 2: the reference is the row whose acceptance CHANGED the projected value — a rejected
|
|
229
|
+
// receipt or a delayed old-attempt result changes nothing and so is never cited. The task's rows
|
|
230
|
+
// and the run-level resets are re-folded prefix by prefix through projectActivity itself.
|
|
231
|
+
// ponytail: O(k²) in the task's own rows; memoize the fold if journals reach thousands of rows per task.
|
|
232
|
+
const changes = acceptedChanges(runId, tracked, t.id, activityTasks.find((a) => a.id === t.id));
|
|
233
|
+
const phase = { label: `phase ${summary?.phase ?? act?.state ?? "unrecorded"}`, ...(own.length > 0 && changes.state !== undefined ? { line: changes.state } : {}) };
|
|
234
|
+
const b = act?.build;
|
|
235
|
+
const receipt = b !== undefined && "receipt" in b ? b.receipt : undefined;
|
|
236
|
+
const build = { label: `build ${b?.state ?? "start-unrecorded"}${receipt ? ` exit ${receipt.exitCode ?? "-"}` : ""}`, ...(changes.build !== undefined ? { line: changes.build } : {}) };
|
|
237
|
+
const decision = decisions.find((d) => d.taskId === t.id);
|
|
238
|
+
const blockerRow = decision ? { line: decision.park.line } : [...own].reverse().find((r) => r.event.event === "task-human" || r.event.event === "task-failed");
|
|
239
|
+
const blk = summary?.blocker;
|
|
240
|
+
const blocker = { label: `blocker ${blk ? `${blk.kind}${blk.diagnostic ? ` · ${blk.diagnostic}` : ""}` : "none"}`, ...(blk && blockerRow ? { line: blockerRow.line } : {}) };
|
|
241
|
+
const nextAction = { label: `next ${blk?.nextAction ?? "none"}`, ...(blk?.nextAction && blockerRow ? { line: blockerRow.line } : {}) };
|
|
242
|
+
// OBS-1048 harvest, chronological: the newest worker-launch opens the attempt; a later worker-result retires it.
|
|
243
|
+
let launched, launchedAttempt, returned = false;
|
|
244
|
+
const nudges = [], pages = [];
|
|
245
|
+
// Finding 1: only rows of the launched attempt count; a row without an attempt belongs to it (derive.ts attemptHarvests).
|
|
246
|
+
const ofLaunched = (d) => launched !== undefined && (ordinal(d.attempt) ?? launchedAttempt) === launchedAttempt;
|
|
247
|
+
for (const r of own) {
|
|
248
|
+
const e = r.event;
|
|
249
|
+
switch (e.event) {
|
|
250
|
+
case "worker-launch":
|
|
251
|
+
launched = r.line;
|
|
252
|
+
launchedAttempt = ordinal(e.data.attempt);
|
|
253
|
+
returned = false;
|
|
254
|
+
nudges.length = 0;
|
|
255
|
+
pages.length = 0;
|
|
256
|
+
break;
|
|
257
|
+
case "worker-nudge-failed":
|
|
258
|
+
if (ofLaunched(e.data))
|
|
259
|
+
nudges.push(r.line);
|
|
260
|
+
break;
|
|
261
|
+
case "operator-page":
|
|
262
|
+
if (ofLaunched(e.data))
|
|
263
|
+
pages.push(r.line);
|
|
264
|
+
break;
|
|
265
|
+
case "worker-result":
|
|
266
|
+
if (ofLaunched(e.data))
|
|
267
|
+
returned = true;
|
|
268
|
+
break;
|
|
269
|
+
}
|
|
270
|
+
}
|
|
271
|
+
const stalled = launched !== undefined && !returned && nudges.length > 0 && pages.length > 0
|
|
272
|
+
? { label: `${STALL_MARKER} · launch #L${launched} · nudge failed ${nudges.map((l) => `#L${l}`).join(",")} · paged ${pages.map((l) => `#L${l}`).join(",")} · no worker-result`, line: launched }
|
|
273
|
+
: undefined;
|
|
274
|
+
return { taskId: t.id, identity, phase, build, blocker, nextAction, ...(stalled ? { stalled } : {}) };
|
|
275
|
+
});
|
|
276
|
+
}
|
|
277
|
+
/** The line at which each projected value last changed, folding the task's rows (and run-level resets) prefix by prefix. */
|
|
278
|
+
function acceptedChanges(runId, tracked, taskId, task) {
|
|
279
|
+
const rows = tracked.filter((r) => { const raw = r.raw; return raw.taskId === taskId || (raw.taskId === undefined && (raw.event === "run-resume" || raw.event === "run-end")); });
|
|
280
|
+
const buildKey = (p) => `${p.build.state}|${"receipt" in p.build ? `${p.build.receipt.attribution.invocation}|${p.build.receipt.exitCode ?? ""}` : ""}`;
|
|
281
|
+
// Seeded from the empty fold: the initial reading is no row's doing, so only a change is cited.
|
|
282
|
+
const initial = projectActivity(runId, [], [task]).get(taskId);
|
|
283
|
+
let state = initial?.state, build = initial === undefined ? undefined : buildKey(initial);
|
|
284
|
+
const out = {};
|
|
285
|
+
for (let i = 0; i < rows.length; i++) {
|
|
286
|
+
const p = projectActivity(runId, rows.slice(0, i + 1), [task]).get(taskId);
|
|
287
|
+
if (!p)
|
|
288
|
+
continue;
|
|
289
|
+
const line = rows[i].sourceIndex;
|
|
290
|
+
if (p.state !== state) {
|
|
291
|
+
state = p.state;
|
|
292
|
+
out.state = line;
|
|
293
|
+
}
|
|
294
|
+
if (buildKey(p) !== build) {
|
|
295
|
+
build = buildKey(p);
|
|
296
|
+
out.build = line;
|
|
297
|
+
}
|
|
298
|
+
}
|
|
299
|
+
return out;
|
|
300
|
+
}
|
|
301
|
+
/** One wrapped line per task: every field with its line, and the stall marker when the harvest is suspect. */
|
|
302
|
+
export const projectionLine = (p) => [p.taskId, readLabel(p.identity), readLabel(p.phase), readLabel(p.build), readLabel(p.blocker), readLabel(p.nextAction), ...(p.stalled ? [p.stalled.label] : [])].join(" · ");
|
|
303
|
+
/**
|
|
304
|
+
* Where the current attempt's pane is recorded — and only that. The reading is exact-name: the
|
|
305
|
+
* launch row's slot name must be the owned name of THIS task, attempt and run. No terminal is ever
|
|
306
|
+
* chosen by title, provider or task id; a missing, foreign, stale or ambiguous record reads
|
|
307
|
+
* unavailable with the rows it was read from, and reading never touches a host.
|
|
308
|
+
*/
|
|
309
|
+
export function paneLocator(task, rows, runId) {
|
|
310
|
+
const ordered = [...rows].filter((r) => r.event !== undefined).sort((a, b) => a.line - b.line);
|
|
311
|
+
// The dispatch-recorded pane field stays reachable beside every reading, with the row it stands on.
|
|
312
|
+
const recorded = task.pane === undefined ? "" : ` · recorded pane ${task.pane}${task.evidence ? ` · evidence #L${task.evidence.line}` : ""}`;
|
|
313
|
+
if (task.attempt === undefined)
|
|
314
|
+
return { status: "unavailable", reason: `no recorded attempt${recorded}` };
|
|
315
|
+
const launches = ordered.filter((r) => r.event.event === "worker-launch" && r.event.taskId === task.id && ordinal(r.event.data.attempt) === task.attempt);
|
|
316
|
+
if (launches.length === 0)
|
|
317
|
+
return { status: "unavailable", reason: `no worker-launch for attempt ${task.attempt}${recorded}` };
|
|
318
|
+
const ids = new Set(launches.map((r) => { const s = r.event.data.slot; return str(s?.id) ?? `#L${r.line}`; }));
|
|
319
|
+
if (ids.size > 1)
|
|
320
|
+
return { status: "unavailable", reason: `ambiguous: ${launches.length} launches ${launches.map((r) => `#L${r.line}`).join(",")}${recorded}` };
|
|
321
|
+
const launch = launches.at(-1);
|
|
322
|
+
const d = launch.event.data;
|
|
323
|
+
const slot = d.slot;
|
|
324
|
+
const expected = formatOwnedName({ role: "worker", taskId: task.id, attempt: task.attempt, runId });
|
|
325
|
+
if (str(d.driver) === undefined || str(slot?.id) === undefined || str(slot?.cwd) === undefined)
|
|
326
|
+
return { status: "unavailable", reason: `launch #L${launch.line} lacks driver or slot identity${recorded}`, line: launch.line };
|
|
327
|
+
if (slot?.name !== expected)
|
|
328
|
+
return { status: "unavailable", reason: `launch #L${launch.line} names ${slotNameOf(d) ?? "no pane"}, not this attempt's owned pane${recorded}`, line: launch.line };
|
|
329
|
+
// Finding 3: only a host whose focus capability VERIFIES the recorded identity can ever confirm it —
|
|
330
|
+
// herdr needs the launch's workspace (HerdrDriver.focus rejects a target without one), orca verifies
|
|
331
|
+
// ownership but cannot focus, subprocess and unknown drivers verify nothing. Reading asks no host.
|
|
332
|
+
const driver = String(d.driver);
|
|
333
|
+
if (driver === "herdr" && str(d.workspace) === undefined)
|
|
334
|
+
return { status: "unavailable", reason: `launch #L${launch.line} records no workspace; herdr cannot verify the pane${recorded}`, line: launch.line };
|
|
335
|
+
if (driver !== "herdr" && driver !== "orca")
|
|
336
|
+
return { status: "unavailable", reason: `launch #L${launch.line} driver ${driver} cannot verify a pane${recorded}`, line: launch.line };
|
|
337
|
+
const stale = ordered.find((r) => {
|
|
338
|
+
if (r.line <= launch.line)
|
|
339
|
+
return false;
|
|
340
|
+
const event = r.event;
|
|
341
|
+
if (event.event === "run-resume" || event.event === "run-end")
|
|
342
|
+
return true;
|
|
343
|
+
if (event.taskId !== task.id)
|
|
344
|
+
return false;
|
|
345
|
+
if (event.event === "worker-result")
|
|
346
|
+
return (ordinal(event.data.attempt) ?? task.attempt) === task.attempt;
|
|
347
|
+
if (event.event === "pane-close")
|
|
348
|
+
return str(event.data.paneId) === str(slot?.id);
|
|
349
|
+
return event.event === "merge";
|
|
350
|
+
});
|
|
351
|
+
if (stale)
|
|
352
|
+
return { status: "unavailable", reason: `launch #L${launch.line} is stale: ${stale.event.event} #L${stale.line} followed it${recorded}`, line: launch.line };
|
|
353
|
+
// A journal launch is evidence of what was opened, not a live-host observation. This pure read
|
|
354
|
+
// deliberately performs no focus/list operation, so it cannot upgrade the locator to recorded:
|
|
355
|
+
// the pane may have been closed outside the journal. The explicit `o` action owns verification.
|
|
356
|
+
return { status: "unavailable", reason: `launch #L${launch.line} records ${driver} ${String(slot.id)}, but live host verification is required${recorded}`, line: launch.line };
|
|
357
|
+
}
|
|
175
358
|
/** The board's footer keys merged with the cockpit's own — every one of them handled by this view or the shell. */
|
|
176
359
|
export const RUN_VIEW_KEYS = [...BOARD_KEYS.slice(0, 2), "←→ verdict gate", "PgUp/PgDn page", "x outcome", "a actions", "o pane", "Tab focus", "? keys", BOARD_KEYS[2]];
|
|
177
360
|
/** The Run body: the approved board (BD-1) over the fold, then the selected task's detail panels. */
|
|
@@ -182,6 +365,10 @@ export function RunView({ snapshot, rows, page, graph, decisions, session, colum
|
|
|
182
365
|
const fit = (text) => fitCells(text, inner);
|
|
183
366
|
const decisionLines = (lines) => lines.flatMap((line) => wrapCells(line, inner));
|
|
184
367
|
const lookup = evidenceLookup(rows, page);
|
|
368
|
+
const wrap = (text) => wrapCells(text, inner);
|
|
369
|
+
// The whole journal, not the retained tail: the projection and the locator read every row.
|
|
370
|
+
const allRows = lookup.later(0);
|
|
371
|
+
const projections = projectRunTasks(snapshot, allRows, graph, decisions, run.runId);
|
|
185
372
|
const tasks = snapshot.tasks;
|
|
186
373
|
const selection = Math.min(session.selection, Math.max(0, tasks.length - 1));
|
|
187
374
|
const task = tasks[selection];
|
|
@@ -195,9 +382,9 @@ export function RunView({ snapshot, rows, page, graph, decisions, session, colum
|
|
|
195
382
|
const { menu, confirming, receipt, notice } = session.decisions;
|
|
196
383
|
const blocked = task === undefined ? [] : graph?.tasks.filter((g) => g.deps.includes(task.id)).map((g) => g.id) ?? decision?.blocks ?? [];
|
|
197
384
|
const board = renderBoardLines({ runId: run.runId, snapshot, graph, now: now(), live: run.live, selection: task?.id, keys: false, colour }, width);
|
|
198
|
-
return (_jsxs(Box, { flexDirection: "column", width: columns, children: [_jsx(BodyText, { emphasis: "dim", children: fitCells(`RUN / ${snapshot.lifecycle} · ${runSummaryLine(snapshot)}`, width) }), board.map((line, i) => _jsx(BodyText, { children: line || " " }, `${i}:${line}`)), tasks.length === 0 && _jsx(BodyText, { children: fitCells("no tasks recorded — no plan", width) }), task !== undefined && (_jsxs(Panel, { title: `SELECTED / ${task.id}${task.title ? ` · ${task.title}` : ""}`, children: [
|
|
385
|
+
return (_jsxs(Box, { flexDirection: "column", width: columns, children: [_jsx(BodyText, { emphasis: "dim", children: fitCells(`RUN / ${snapshot.lifecycle} · ${runSummaryLine(snapshot)}`, width) }), board.map((line, i) => _jsx(BodyText, { children: line || " " }, `${i}:${line}`)), tasks.length === 0 && _jsx(BodyText, { children: fitCells("no tasks recorded — no plan", width) }), task !== undefined && (_jsxs(Panel, { title: `SELECTED / ${task.id}${task.title ? ` · ${task.title}` : ""}`, children: [wrap(`state ${parkOf(task)} · ${attemptLabel(task)} · dispatches ${task.dispatches} · path ${task.path ?? "unknown"} · pane ${task.pane ?? "unknown"} · alarm ${task.alarmMs === undefined ? "unknown" : `${task.alarmMs}ms`}${task.mergeEvidence ? ` · merged #L${task.mergeEvidence.line}` : ""} · locator ${(() => { const l = paneLocator(task, allRows, run.runId); return l.status === "recorded" ? `recorded ${l.reason}` : `unavailable — ${l.reason}`; })()}`).map((line, i) => _jsx(BodyText, { children: line }, `${i}:${line}`)), _jsx(BodyText, { emphasis: "dim", children: fit(`outcome filter ${session.outcomeFilter} (x cycles) · ${shown.length} of ${cells.length} gates shown`) }), shown.map((c) => (_jsx(BodyText, { children: fit(`${c.letter} ${c.gate.padEnd(10)} ${c.labels.join(" · ")}${c.line === undefined ? "" : ` · #L${c.line}`}`) }, c.gate)))] })), projections.length > 0 && (_jsx(Panel, { title: "PROJECTION / every task \u00B7 identity \u00B7 phase \u00B7 build \u00B7 blocker \u00B7 next action, each with its journal line", children: projections.flatMap((p) => wrap(projectionLine(p)).map((line, i) => _jsx(BodyText, { emphasis: p.stalled && i === 0 ? "strong" : "normal", children: line }, `${p.taskId}:${i}`))) })), verdictCell !== undefined && (_jsxs(Panel, { title: `VERDICT / ${verdictCell.gate}${verdictCell.line === undefined ? "" : ` #L${verdictCell.line}`} · ${verdict.length === 0 ? "no verdict text" : `lines ${from + 1}–${from + window.length} of ${verdict.length}`}`, children: [window.map((line, i) => _jsx(BodyText, { children: fit(`${String(from + i + 1).padStart(4)} ${line}`) }, `${from + i}:${line}`)), _jsx(BodyText, { emphasis: "dim", children: fit("←→ gate · PageUp/PageDown page") })] })), task?.state === "human" && (_jsx(Panel, { title: `PARK / ${task.id}`, children: decision === undefined
|
|
199
386
|
? _jsx(BodyText, { children: fit("park evidence unavailable — refresh before deciding") })
|
|
200
|
-
: (_jsxs(_Fragment, { children: [
|
|
387
|
+
: (_jsxs(_Fragment, { children: [wrap(`#L${decision.park.line} ${decision.park.kind ?? "unknown kind"}${decision.park.failedGate ? ` · failed gate ${decision.park.failedGate}` : ""} · ${decision.park.reason ?? "no reason recorded"}`).map((line, i) => _jsx(BodyText, { children: line }, `park:${i}`)), _jsx(BodyText, { children: fit(`blocks ${blocked.length === 0 ? "nothing" : blocked.join(", ")} · attempts ${decision.attempts}`) }), wrap(decision.verbs.length === 0 ? `diagnostic: ${decision.diagnostic ?? "no verb"}` : `decisions: ${decision.verbs.join(" · ")} (a Actions)`).map((line, i) => _jsx(BodyText, { children: line }, `verbs:${i}`)), _jsx(BodyText, { emphasis: "dim", children: fit(run.live ? "matching live daemon enacts a release at its next task boundary" : run.blockingRunId ? `other live run ${run.blockingRunId} holds the lock; resume after it ends` : "no live owner: a release records permission; resume dispatches") })] })) })), menu !== null && (_jsx(Panel, { title: `ACTIONS / ${menu.taskId}`, focused: true, children: menu.verbs.length === 0
|
|
201
388
|
? _jsx(BodyText, { children: fit(`no decision — ${decision?.diagnostic ?? "no verb for this park"}`) })
|
|
202
389
|
: menu.verbs.map((verb, i) => _jsx(BodyText, { emphasis: i === menu.selection ? "strong" : "normal", children: fit(`${i === menu.selection ? `${GLYPHS.pointer} ` : " "}${verb}`) }, verb)) })), confirming !== null && (_jsx(Panel, { title: `CONFIRM ${confirming.command.verb.toUpperCase()}`, focused: true, children: decisionLines(decisionConfirmLines(confirming)).map((line, i) => _jsx(BodyText, { children: line }, `${i}:${line}`)) })), receipt !== null && (_jsx(Panel, { title: receipt.ok ? "RECEIPT · appended" : "RECEIPT · refused", children: decisionLines(decisionReceiptLines(receipt)).map((line, i) => _jsx(BodyText, { children: line }, `${i}:${line}`)) })), notice !== null && _jsx(BodyText, { children: notice }), _jsx(BodyText, { emphasis: "dim", children: clipBoard(boardFooter([...RUN_VIEW_KEYS, decisionKeybar(session.decisions, decision)].filter(Boolean), colour), width) })] }));
|
|
203
390
|
}
|
package/package.json
CHANGED
|
@@ -149,6 +149,10 @@
|
|
|
149
149
|
"text": {
|
|
150
150
|
"type": "string",
|
|
151
151
|
"minLength": 1
|
|
152
|
+
},
|
|
153
|
+
"landing": {
|
|
154
|
+
"type": "string",
|
|
155
|
+
"minLength": 1
|
|
152
156
|
}
|
|
153
157
|
},
|
|
154
158
|
"required": [
|
|
@@ -201,6 +205,56 @@
|
|
|
201
205
|
]
|
|
202
206
|
}
|
|
203
207
|
},
|
|
208
|
+
"pins": {
|
|
209
|
+
"type": "array",
|
|
210
|
+
"items": {
|
|
211
|
+
"oneOf": [
|
|
212
|
+
{
|
|
213
|
+
"type": "object",
|
|
214
|
+
"properties": {
|
|
215
|
+
"kind": {
|
|
216
|
+
"type": "string",
|
|
217
|
+
"const": "literal"
|
|
218
|
+
},
|
|
219
|
+
"text": {
|
|
220
|
+
"type": "string",
|
|
221
|
+
"minLength": 1
|
|
222
|
+
},
|
|
223
|
+
"glob": {
|
|
224
|
+
"type": "string",
|
|
225
|
+
"minLength": 1
|
|
226
|
+
}
|
|
227
|
+
},
|
|
228
|
+
"required": [
|
|
229
|
+
"kind",
|
|
230
|
+
"text",
|
|
231
|
+
"glob"
|
|
232
|
+
]
|
|
233
|
+
},
|
|
234
|
+
{
|
|
235
|
+
"type": "object",
|
|
236
|
+
"properties": {
|
|
237
|
+
"kind": {
|
|
238
|
+
"type": "string",
|
|
239
|
+
"const": "fixture"
|
|
240
|
+
},
|
|
241
|
+
"paths": {
|
|
242
|
+
"minItems": 1,
|
|
243
|
+
"type": "array",
|
|
244
|
+
"items": {
|
|
245
|
+
"type": "string",
|
|
246
|
+
"minLength": 1
|
|
247
|
+
}
|
|
248
|
+
}
|
|
249
|
+
},
|
|
250
|
+
"required": [
|
|
251
|
+
"kind",
|
|
252
|
+
"paths"
|
|
253
|
+
]
|
|
254
|
+
}
|
|
255
|
+
]
|
|
256
|
+
}
|
|
257
|
+
},
|
|
204
258
|
"gates": {
|
|
205
259
|
"default": [
|
|
206
260
|
"build",
|
|
@@ -139,7 +139,7 @@ through brief lineage. **An executor choice nobody made is still an executor cho
|
|
|
139
139
|
fraction (`ORCH · v1.19 4/5`, updated on every task-done); tickmarkr opens ONE TAB PER TASK, labelled
|
|
140
140
|
with the task id and holding that task's worker plus its judge/review/consult panes (tickmarkr
|
|
141
141
|
updates it). Never long context strings or ✓-chains.
|
|
142
|
-
- **On Orca (`TERM_PROGRAM=Orca` and non-empty `ORCA_TERMINAL_HANDLE`)**: Do not load the `herdr` skill and do not map a Herdr workspace. Work from the current Orca terminal context. Keep the overseer in the launching terminal and inspect terminals with `orca terminal list --json`. The daemon self-places the watch board as a horizontal split of the launching terminal (`ORCA_TERMINAL_HANDLE`).
|
|
142
|
+
- **On Orca (`TERM_PROGRAM=Orca` and non-empty `ORCA_TERMINAL_HANDLE`)**: Do not load the `herdr` skill and do not map a Herdr workspace. Work from the current Orca terminal context. Keep the overseer in the launching terminal, name it in the same act — `orca terminal rename --terminal "$ORCA_TERMINAL_HANDLE" --title "OVERSEER · <version>"` (tab title; see the seat-name law under Seat-spawn recipes) — and inspect terminals with `orca terminal list --json`. The daemon self-places the watch board as a horizontal split of the launching terminal (`ORCA_TERMINAL_HANDLE`).
|
|
143
143
|
2. **Orchestrator**: Launch the orchestrator with your agent host.
|
|
144
144
|
- **On herdr (`HERDR_ENV=1`)**: Spawning on current herdr is two-step — the one-shot `agent start --cwd` form was removed in the herdr CLI redesign and now fails with `unknown option` (OBS-138): first create the pane with `herdr tab create --workspace <ws> --cwd <repo> --label "ORCH · <version>"` and parse `result.root_pane.pane_id` from its JSON, then start the agent in it. For Claude Code, use `herdr agent start orchestrator --kind claude --pane <root-pane-id> -- --permission-mode bypassPermissions` (append `--model <m>` after the `--` if the operator has a policy). For Codex, use `herdr agent start orchestrator --kind codex --pane <root-pane-id> -- --dangerously-bypass-approvals-and-sandbox` (add `--model <m>` to specify a model). The unsandboxed flag is REQUIRED: codex's `workspace-write` sandbox keeps `.git` refs read-only, so a sandboxed orchestrator's `tickmarkr run` dies at integration-branch creation — do not downgrade it. Workers you never spawn — tickmarkr spawns its own visible worker panes. Auxiliary agents you do spawn (consultants, reviewers, scouts) follow the same forms: never launch a claude session in plan mode or default permission mode for autonomous work — both stall on per-command approval prompts nobody is watching; claude is always `--permission-mode bypassPermissions --settings '{"promptSuggestionEnabled":false}'`. **For a codex consultant, use `-a never --sandbox workspace-write` — NOT `--sandbox read-only`.** ⚠ **`--sandbox read-only` CONTRADICTS this skill's own completion protocol and will hang the seat.** Every seat you spawn is told to deliver an ARTIFACT ending in a terminal MARKER, because that is the only completion signal the artifact watcher can key on (`done` is turn end). A read-only sandbox cannot write that artifact, so codex blocks on `Would you like to make the following edits?` for its OWN report — and the report exists ONLY in the pending edit, so abandoning the prompt destroys the work rather than merely delaying it. Measured 2026-08-28: a consultant spawned `--sandbox read-only` finished a 14,604-byte verdict, sat blocked on the write, and the operator saw the prompt before the supervising tier did. `read-only` is correct ONLY for a seat that writes nothing at all — which, under the artifact+marker rule, is no seat this skill tells you to spawn. When the prompt does appear, answer **"Yes, and don't ask again for these files"** rather than plain yes: plain yes re-blocks on the next write of the same file. **That `--settings` pair is not cosmetic and it is not optional:** claude-code's AUTOSUGGEST renders context-plausible ghost text into an idle seat's prompt line that is BYTE-IDENTICAL to a typed draft in text-format reads (OBS-482), so a supervising tier cannot tell a seat's own unsent work from a rendering artifact without `agent read --format ansi`. Turning the suggester off at spawn removes the ambiguity at its source instead of paying for the discrimination at every read. Verified against the shipped binary: `claude --settings '{"promptSuggestionEnabled":false}' -p …` exits 0 with a real response, and the key appears in the binary's own settings schema. **For kimi, pass `-y`** (`herdr agent start <name> --kind kimi --pane <id> -- -y`) — the adapter already launches its own workers that way (`src/adapters/kimi.ts:204`), and a kimi seat spawned without it sits on an approval prompt having done nothing. **Herdr cannot see that state**: it reports a kimi pane as `agent_status: working` with `screen_detection_skipped: true` while the prompt is up, so the BLOCKED-STATE watcher below is blind on this vendor and the spawn flag is the ONLY control. Every vendor you spawn needs its auto-approve form named here; a vendor absent from this list is a seat that will hang.
|
|
145
145
|
- **On Orca (`TERM_PROGRAM=Orca` and non-empty `ORCA_TERMINAL_HANDLE`)**: Seats spawn with `orca terminal create` on a path worktree selector with a command:
|
|
@@ -176,6 +176,12 @@ through brief lineage. **An executor choice nobody made is still an executor cho
|
|
|
176
176
|
- **On Orca (`TERM_PROGRAM=Orca` and non-empty `ORCA_TERMINAL_HANDLE`)**: Re-list terminals, re-read the
|
|
177
177
|
recorded terminal handles and evidence files, and re-arm only file/journal watchers. Do not use a Herdr
|
|
178
178
|
pane id, `agent start`, `pane run`, tab canon, or a name-keyed Herdr watcher on an Orca launch.
|
|
179
|
+
**Handles do not survive an Orca restart (OBS-1050, three gates lost to it in v2.5.6).** The restart
|
|
180
|
+
re-issues every terminal handle; a recorded one answers `terminal_handle_stale` and any watcher or gate
|
|
181
|
+
keyed to it is already dead. So: record every seat's handle in `pids/<seat>.handle` beside its title at
|
|
182
|
+
spawn; on `terminal_handle_stale`, re-list terminals and re-resolve the seat by TITLE, then rewrite the
|
|
183
|
+
handle file; and treat a stale handle as a dead seat until the read-back of the new handle proves
|
|
184
|
+
otherwise — a gate that was running in that seat has to be re-launched, never assumed to continue.
|
|
179
185
|
When briefing a seat on either host, take its current address from the handoff file; never hardcode a pane id
|
|
180
186
|
or terminal handle in a brief or command template.
|
|
181
187
|
|
|
@@ -199,7 +205,45 @@ Inventories retain the **full suite log**, not a tail or summary, and no one run
|
|
|
199
205
|
- **On herdr (`HERDR_ENV=1`)**: Every mission to a Claude or Grok seat is delivered only with `herdr pane run <pane> "<message>"` and
|
|
200
206
|
verified by reading the pane back; never use `agent prompt` for mission delivery. Launch a Grok seat with
|
|
201
207
|
`herdr agent start <seat> --kind grok --pane <pane> -- -m grok-4.6`.
|
|
202
|
-
- **On Orca (`TERM_PROGRAM=Orca` and non-empty `ORCA_TERMINAL_HANDLE`)**: Seats spawn with `orca terminal create --worktree path:<repo> --command "<cmd>" --json`.
|
|
208
|
+
- **On Orca (`TERM_PROGRAM=Orca` and non-empty `ORCA_TERMINAL_HANDLE`)**: Seats spawn with `orca terminal create --worktree path:<repo> --command "<cmd>" --json`.
|
|
209
|
+
**`--worktree path:` resolves only an Orca-MANAGED worktree** (`orca worktree list`); on any other checkout —
|
|
210
|
+
a `git worktree add` the overseer made for a spec branch, a throwaway clone — `terminal create` hangs and
|
|
211
|
+
returns "Timed out waiting for terminal handle after creation". The lawful form for such a checkout is a
|
|
212
|
+
terminal in a managed worktree with a cwd change in the command:
|
|
213
|
+
`orca terminal create --worktree path:<managed-repo> --title "ORCH · <version>" --command "cd <checkout> && <agent-cmd>" --json`.
|
|
214
|
+
**An Orca-hosted run's checkout is created by Orca** (`orca worktree create --repo id:<repo> --base-branch <spec-branch>`),
|
|
215
|
+
never by `git worktree add`: the orca driver asks `orca worktree current` from inside every task worktree and refuses
|
|
216
|
+
(`selector_not_found`) on an unmanaged checkout — every task fails pre-launch and the board cannot place (OBS-1061,
|
|
217
|
+
one abandoned run). **The overseer sits in the RUN's workspace**, in its own tab beside the orchestrator's, so the
|
|
218
|
+
run's card shows its supervision; records may live on another branch and are committed there by path.
|
|
219
|
+
**The orchestrator NEVER shares the overseer's tab.** Never spawn it with `orca terminal split` off the
|
|
220
|
+
overseer's own handle, even as a fallback: the daemon self-places its live board as a split of the
|
|
221
|
+
LAUNCHING terminal, so an orchestrator in the overseer's tab drags the board and every worker pane
|
|
222
|
+
into it (measured 2026-09-20: `terminal create` timed out on an unmanaged spec worktree, the overseer
|
|
223
|
+
split instead, and the run had to be re-seated before GO). Before delivering the brief, prove the seat
|
|
224
|
+
is in its own tab: `orca terminal list --json`, the new handle's `tabId` must differ from the
|
|
225
|
+
overseer's; if it does not, close the seat and create it again the lawful way. `split` is for seats
|
|
226
|
+
that BELONG beside another seat (consultant pairs, a reviewer beside its surgeon), never for the
|
|
227
|
+
orchestrator or a gate seat.
|
|
228
|
+
**Seat names on Orca are TAB titles, and a tab title HOLDS; the pane title is the agent's and drifts.**
|
|
229
|
+
Orca keeps two titles per seat. The TAB title is owned: `orca terminal create --title` and
|
|
230
|
+
`orca terminal rename --terminal <handle> --title "<seat>"` set it, and nothing a process writes to
|
|
231
|
+
its pty changes it — measured 2026-09-20 (OBS-1063) on a scratch terminal writing an OSC title every
|
|
232
|
+
2 s: the PANE title flipped to the OSC text within 8 s, the TAB title stayed, and the operator's tab
|
|
233
|
+
strip renders the TAB title. The pane title is whatever the process last wrote (oh-my-zsh, Claude
|
|
234
|
+
Code's turn summary, Codex's last prompt) and is NOT the seat's name. This is vendor-neutral by
|
|
235
|
+
construction — no per-vendor env var, no agent command, no launch-form ritual — which is what makes
|
|
236
|
+
it the same law as Herdr's tab label. So: **the overseer names its OWN seat at Setup 1** with
|
|
237
|
+
`orca terminal rename --terminal "$ORCA_TERMINAL_HANDLE" --title "OVERSEER · <version>"` (the seat
|
|
238
|
+
it was invoked in is the one seat nobody else launches), every seat it spawns is named by
|
|
239
|
+
`create --title`, and a live-label change (progress fraction, hot-state token) is `rename` again.
|
|
240
|
+
**Prove a name by the field that holds:** `tabs[].title` in
|
|
241
|
+
`orca terminal list --include-visual-layouts --json`, never the plain `list`'s `title` — that one is
|
|
242
|
+
the pane title and reads as drift on every agent seat, which is how three sessions "proved" a
|
|
243
|
+
failure that was a misread (`src/drivers/orca.ts:27-42` keys worker identity on the owned tab title
|
|
244
|
+
for the same reason). The earlier three-part launch form (`DISABLE_AUTO_TITLE` + `printf` +
|
|
245
|
+
`CLAUDE_CODE_DISABLE_TERMINAL_TITLE=1`) is withdrawn: it silenced one vendor's pane title and named
|
|
246
|
+
nothing the tab did not already hold. Brief delivery travels as a file announced by `orca terminal send --terminal <handle> --text "<announcement>" --enter --wait-submit 15 --json`. Read `result.send.prompt.stages` and treat the brief as delivered only when a `turn_started` stage is present; `accepted: true` proves input acceptance, not a started turn; never resend on `accepted` alone.
|
|
203
247
|
For Leg-2 (both hosts), a Codex reviewer under `workspace-write` must be briefed with an in-worktree verdict path such as
|
|
204
248
|
`<repo>/.tickmarkr/overseer/verdicts/<task>.md`, and its verdict must be written there before it is read.
|
|
205
249
|
## Supervising tickmarkr as the executor — WHO DOES WHAT
|
|
@@ -244,6 +288,58 @@ journal tail to decide what happens next, or sweeping orphans — you have taken
|
|
|
244
288
|
3. **Gate a mid-run fix at the base, not at a summary (law 47 / OBS-909).** A fix landed on main while a
|
|
245
289
|
run is live is proved with `tickmarkr verify --base <main>`, never with a suite summary copied from a
|
|
246
290
|
different tree. A release proof runs every CI-ordered step — including lint — before its suite.
|
|
291
|
+
4. **Rescue a broken harness from OUTSIDE the run it broke — the daemon is NEVER asked to repair itself
|
|
292
|
+
as a task (v2.5.7 ledger D-59, D-64; OBS-1078).** When the harness is the defect — resume cannot be
|
|
293
|
+
relied on, a gate cannot be relied on, the binary every gate runs on is the thing that is wrong — no
|
|
294
|
+
task inside the run can fix it: the gates that would judge the repair are executed BY the suspect
|
|
295
|
+
binary, and the worktree the repair would run in is created by the same code. Dispatching
|
|
296
|
+
*"fix the daemon"* as a task of the run it broke buys a green gate from the defect and a repair nobody
|
|
297
|
+
can trust. The rescue route has three legs, each with its own record, and there is no fourth:
|
|
298
|
+
- **AN INDEPENDENTLY GATED EXTERNAL FIX LEG.** Its own checkout and branch off the run's base, its own
|
|
299
|
+
brief, its own `files[]`, its own battery — `tickmarkr verify --base <ref> --criteria <file>
|
|
300
|
+
--author <the seat that wrote it>` — plus the cross-vendor review its criterion names, all recorded
|
|
301
|
+
BEFORE the run consumes any of it. The leg's battery is scheduled at a boundary item 2 permits (the
|
|
302
|
+
run parked or ended, or the operator's recorded exception), and its reds are declared to the
|
|
303
|
+
contamination watcher in advance so they are classified instead of assumed.
|
|
304
|
+
- **AN EXPLICITLY RECORDED REBUILT BINARY AND RESTART — the provenance is written down in the same act,
|
|
305
|
+
or the rebuild did not happen.** Record: the fix leg's commit sha; the build command; the install
|
|
306
|
+
form (never `npm i -g .` on the repository directory — npm SYMLINKS a directory install and every
|
|
307
|
+
later build hot-swaps the machine-wide binary with no version change to notice it by); the
|
|
308
|
+
global-versus-repo **inode** comparison that proves a real install, which `tickmarkr version` cannot
|
|
309
|
+
do because it cannot go red when nothing is bumped; the version read-back; and the restart itself —
|
|
310
|
+
`tickmarkr resume <runId>`, the ORCHESTRATOR's command, never yours. An unrecorded rebuild leaves the
|
|
311
|
+
next seat arguing about which binary produced which journal rows, with nothing to read.
|
|
312
|
+
- **A SEPARATELY GATED UNION.** The rebuilt harness and the run's existing work are then gated
|
|
313
|
+
TOGETHER, on their own, at the one commit that carries both. A green fix leg beside a green run is
|
|
314
|
+
no proof of their union — see *a green leg beside a green parent*, under the release criterion
|
|
315
|
+
below — because the union's own conflict resolutions, ordering and collected-count changes are
|
|
316
|
+
exactly what neither green ever saw.
|
|
317
|
+
**NO BRANCH SURGERY: the run keeps its ORIGINAL base and its recorded baseline.** No rebase of the
|
|
318
|
+
integration branch, no re-cut `baseRef`, no re-baselining to make the rescue leg's diff read small.
|
|
319
|
+
Resume REPLAYS the journal's `baseRef`, so a rewritten one is a different run wearing the same id —
|
|
320
|
+
and a baseline re-recorded to suit the rescue launders every red the original baseline was there to
|
|
321
|
+
compare against.
|
|
322
|
+
5. **A rebuilt binary is NOT a rebuilt tree: a claimed fix arrival needs ANCESTRY EVIDENCE naming the
|
|
323
|
+
actual task subject (v2.5.7 ledger D-80; OBS-1078).** Rebuilding and reinstalling `dist` changes the
|
|
324
|
+
daemon the orchestrator launches and nothing else. It puts no test, fixture, schema or skill byte into
|
|
325
|
+
any task worktree or into the integration branch: those trees were checked out from the integration tip
|
|
326
|
+
BEFORE the fix leg landed, and git does not retro-fill a checkout that already exists. So *"the fix has
|
|
327
|
+
reached the task checkouts"* is never accepted on a version read-back, a rebuild log, or the fixer's
|
|
328
|
+
say-so — require ancestry evidence naming the ACTUAL subjects, and read it before the claim, not after:
|
|
329
|
+
- **the TASK subject** — that task worktree's own branch and HEAD sha (`git -C <task-worktree> rev-parse
|
|
330
|
+
--abbrev-ref HEAD` and `git -C <task-worktree> rev-parse HEAD`), with the fix proven an ancestor of
|
|
331
|
+
THAT commit: `git -C <task-worktree> merge-base --is-ancestor <fix-sha> HEAD`, whose exit code is the
|
|
332
|
+
evidence, recorded beside all three shas;
|
|
333
|
+
- **the INTEGRATION subject** — the integration-branch commit that worktree was created from, named by
|
|
334
|
+
sha from the journal's `task-dispatch` row or `tickmarkr status`; never "the run", and never the
|
|
335
|
+
branch name alone, which moves while you are reading it;
|
|
336
|
+
- **and, where the fix is a FILE the tree must hold** — a new test, a fixture, a schema — its presence
|
|
337
|
+
at that commit: `git -C <task-worktree> cat-file -e <task-head>:<path>`. A fix that lives only in the
|
|
338
|
+
daemon's `dist` cannot make a worker's missing oracle appear, and a worker red on that missing file is
|
|
339
|
+
still a plan defect, not a retry.
|
|
340
|
+
A fix present in the daemon and absent from the task trees has been INSTALLED, not ARRIVED: say which
|
|
341
|
+
of the two you mean, and name the commit each half was read against. Evidence rule 13's baseRef trap is
|
|
342
|
+
the same defect wearing a diff instead of a claim.
|
|
247
343
|
|
|
248
344
|
### What the ORCHESTRATOR does, and what you require of it
|
|
249
345
|
|
|
@@ -339,12 +435,67 @@ its own — **not because any instrument detected it.**
|
|
|
339
435
|
> covers — and a re-scope of any named subject voids it AUTOMATICALLY, with no ruling required.**
|
|
340
436
|
> A void condition that needs a ruling to fire is not a void condition; it is a second thing to forget.
|
|
341
437
|
|
|
438
|
+
**⚡ ONE QUALIFICATION, PROSPECTIVE ONLY: A FUTURE SEAL MAY DECLARE THAT AN AUDITED FILES-ONLY AMENDMENT
|
|
439
|
+
DOES NOT VOID IT — AND THAT DECLARATION HAS TO BE IN THE SEAL, WRITTEN AT ISSUE (v2.5.7 ledger D-80, D-83;
|
|
440
|
+
OBS-1078).** The automatic-void rule above is QUALIFIED for seals issued after this clause, never deleted:
|
|
441
|
+
it still fires on its own terms for every seal that does not carry the declaration. A requirements seal may
|
|
442
|
+
declare, prospectively, that exactly one narrow class of change does not void it — an **AUDITED FILES-ONLY
|
|
443
|
+
AMENDMENT**: a recorded amendment to a task's `files[]` and nothing else, which leaves all four of these
|
|
444
|
+
UNTOUCHED.
|
|
445
|
+
|
|
446
|
+
1. **the ACCEPTANCE TEXT** — every criterion's own words, byte for byte, a `test:` leaf title included;
|
|
447
|
+
2. **the TASK IDENTITY** — which task is which: its id, its shape, its deps, the suites it owns;
|
|
448
|
+
3. **the RUN IDENTITY** — the run, its base and its `baseRef`, the integration branch and its tip commit;
|
|
449
|
+
4. **the SAFETY REQUIREMENTS** — the gates, bounds, refusals and contamination rules the run is held to.
|
|
450
|
+
|
|
451
|
+
**Replay it both ways, because the clause is a procedure and not a mood:**
|
|
452
|
+
- *A seal carrying that declaration, then an audited amendment adding one owned path to a task's `files[]`,
|
|
453
|
+
with all four invariants read and found unchanged* → **THE SEAL IS KEPT.** No successor seal and no
|
|
454
|
+
re-grade of what it already sealed: the declaration was the seal's own pre-commitment about this exact
|
|
455
|
+
change, made before the change existed — the only moment such a commitment can honestly be made. Read the
|
|
456
|
+
four invariants BEFORE ruling, and record the read beside the amendment.
|
|
457
|
+
- *The same seal, then an amendment that changes the RUN IDENTITY — a new run id, a re-cut `baseRef`, a
|
|
458
|
+
different integration tip* → **THE SEAL IS VOID, on its own declaration's terms.** Run identity is one of
|
|
459
|
+
the four, so *"files-only"* does not describe that amendment; moving any one of the four puts the change
|
|
460
|
+
outside the declared class, and **any other change of any kind needs a SUCCESSOR SEAL** — a new document,
|
|
461
|
+
sealed before the work it grades, naming the subject as it now stands.
|
|
462
|
+
|
|
463
|
+
**Three limits, and each one has been argued around:**
|
|
464
|
+
- **The declaration is never retroactive and never inferred.** A seal issued BEFORE this change — one whose
|
|
465
|
+
own text does not carry the declaration — **stays governed by the existing void conditions above,
|
|
466
|
+
unmodified: any re-scope, narrowing, widening, split, merge or re-ownership of a named subject voids it
|
|
467
|
+
AUTOMATICALLY, with no ruling required.** So *"the acceptance text never changed, therefore the seal
|
|
468
|
+
still holds"* is **REFUSED** for such a seal once its SUBJECT has moved: unchanged acceptance text alone
|
|
469
|
+
preserves nothing over a changed subject, because a seal grades the subject it named, not the sentences
|
|
470
|
+
it happened to be written in. Retrofitting the declaration onto an already-sealed document is itself an
|
|
471
|
+
amendment to a sealed subject — write the successor seal instead.
|
|
472
|
+
- **A files-only clause confers no amendment authority.** It says nothing about whether the amendment is
|
|
473
|
+
LAWFUL: compile's unit bounds still refuse an over-bound `files[]`, and no seal can widen what compile
|
|
474
|
+
will certify.
|
|
475
|
+
- **AUDITED means recorded, or it did not happen.** The approval row carrying the amended files, the
|
|
476
|
+
before-and-after `files[]`, and the four-invariant read are what let the next seat replay your ruling
|
|
477
|
+
instead of trusting it. An unrecorded *"files-only"* amendment is indistinguishable from a re-scope, and
|
|
478
|
+
is voided as one.
|
|
479
|
+
|
|
342
480
|
**⚡ IDENTIFY THE SUBJECT BY WHAT THE CLAIM IS ABOUT. Half of the failure above was a category error, and
|
|
343
481
|
it is the cheap half to fix:** a graph hash identifies a **PLAN**, and a plan is recompiled, re-cut and
|
|
344
482
|
re-owned as a matter of course. **A criterion about a SHIPPED TREE names the COMMIT** — or the tag, or the
|
|
345
483
|
export tree hash — **never a graph hash, never a run id, never a task list.** Ask what a reader would have
|
|
346
484
|
to hold in their hand to check the clause: if it is bytes, name the bytes.
|
|
347
485
|
|
|
486
|
+
**⚡ A GREEN LEG BESIDE A GREEN PARENT IS NO PROOF OF THEIR UNION: THE RELEASE PROOF BINDS TO THE MERGED
|
|
487
|
+
CANDIDATE'S COMMIT, AND IS VOID THE MOMENT THAT SUBJECT MOVES (v2.5.7 ledger D-80, D-83; OBS-1078).** The
|
|
488
|
+
subject of a release proof is never *"the run"* and never *"the fix leg"* — it is the **MERGED candidate**:
|
|
489
|
+
the one commit that carries the run's work AND every leg merged into it. Name that commit's sha in the
|
|
490
|
+
proof, in the same act as the grade, and the proof is bound to that sha alone. **When the subject moves — a
|
|
491
|
+
later merge, a new integration tip, an amended or re-cut commit, a fresh export, a re-tag — the proof is
|
|
492
|
+
VOID, and the new subject is graded from scratch**, count oracle included, since the oracle is derived from
|
|
493
|
+
the merged tree and not from either parent. Two greens assembled into one verdict are two claims about two
|
|
494
|
+
trees that never contained each other: the union's own conflict resolutions, its ordering and its
|
|
495
|
+
collected-test total are exactly what neither green observed — which is why `RELEASING.md` step 4 matches
|
|
496
|
+
the run's head SHA to the mirror's `HEAD` before a single job log is graded. This is the void-condition duty
|
|
497
|
+
pointed at bytes: **a proof that names no commit cannot notice its subject leaving.**
|
|
498
|
+
|
|
348
499
|
**THE RECIPROCAL DUTY, and it is yours because you write both documents:** when you issue a ruling that
|
|
349
500
|
re-scopes, narrows, splits or re-owns anything, **the ruling must name every sealed document its subject
|
|
350
501
|
appears in** — and say, in the ruling, whether each one is now void. You are the only seat that can do
|
|
@@ -656,6 +807,14 @@ they are left implicit:
|
|
|
656
807
|
fail-closed verdict, no daemon, no retries. A per-lane grep gate re-implements a weaker version of
|
|
657
808
|
this and passes on source text the screen never renders, which is exactly the class the acceptance
|
|
658
809
|
judge exists to reject. One command per lane, named in the lane's own brief.
|
|
810
|
+
Fix-leg laws learned the hard way (v2.5.6 R25/R26/R34):
|
|
811
|
+
- A `test:` criterion for a fix leg is written FROM the landed test's title, verbatim, never before the
|
|
812
|
+
test exists — the acceptance oracle runs the sentence as vitest `-t` and a paraphrase matches zero tests.
|
|
813
|
+
- Every fix-leg verify carries `--author <the seat that wrote it>`; without it the author is HUMAN, every
|
|
814
|
+
vendor stays eligible, and a same-vendor approval is not the cross-vendor review the criterion names.
|
|
815
|
+
- A grader change is proved on a captured REAL job log, never on a synthetic one.
|
|
816
|
+
- A public-CI red is classified in a throttled Linux container first (`docker run --cpus=2 node:22`),
|
|
817
|
+
by measurement: starvation on the base tree is not a defect on the branch.
|
|
659
818
|
|
|
660
819
|
## Pane mechanics that bite
|
|
661
820
|
|
|
@@ -1029,7 +1188,7 @@ orchestrator turn boundary.
|
|
|
1029
1188
|
|
|
1030
1189
|
### On Orca (`TERM_PROGRAM=Orca` and non-empty `ORCA_TERMINAL_HANDLE`) — supervision instruments
|
|
1031
1190
|
|
|
1032
|
-
At seat spawn, arm file/journal watchers for artifact completion, run events, missing progress and context evidence. Record each watcher owner and bounded expiry; renew on every wake and stop on stand-down. Read `orca terminal read --terminal <handle> --screen --json` on each wake to detect blocked or pending input. For a human gate, write a checkpoint evidence file and announce it through verified terminal send. Use Orca notifications only when the installed host advertises a notification capability; the current CLI has no `notification` command, so the file and terminal receipt remain the delivery path. A notification or accepted input alone never proves delivery or completion.
|
|
1191
|
+
At seat spawn, arm file/journal watchers for artifact completion, run events, missing progress and context evidence. Record each watcher owner and bounded expiry; renew on every wake and stop on stand-down. Read `orca terminal read --terminal <handle> --screen --json` on each wake to detect blocked or pending input. For a human gate, write a checkpoint evidence file and announce it through verified terminal send. Use Orca notifications only when the installed host advertises a notification capability; the current CLI has no `notification` command, so the file and terminal receipt remain the delivery path. A notification or accepted input alone never proves delivery or completion. A seat-liveness watcher on the ORCHESTRATOR handle is mandatory, not optional: poll `orca terminal read --terminal <handle> --screen --json` on a bounded interval, treat `terminal_handle_stale` or a missing terminal as seat death, and re-resolve by title before re-arming (OBS-1050 — the orchestrator seat exited silently and the daemon ran unsupervised to a PARTIAL run-end).
|
|
1033
1192
|
|
|
1034
1193
|
## Specialist pipeline rules
|
|
1035
1194
|
|