tickmarkr 2.5.2 → 2.5.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,5 +1,6 @@
1
1
  import { type ClassifiedWorkerResult } from "./prompt.js";
2
2
  import { type WorkerAdapter } from "./types.js";
3
3
  export declare const QWEN_VERSION_IDENTITY: RegExp;
4
+ export declare const QWEN_HARNESS_BANNER_ROWS: readonly ["⚠ SAFE MODE — all customizations disabled (hooks, extensions, skills, MCP servers, QWEN.md). Restart without --safe-mode to resume normal operation.", "Warning: running headless with --yolo / approval-mode=yolo and no sandbox. All tool calls (shell, write, edit) auto-execute at this process's privilege level. Enable a sandbox via --sandbox / QWEN_SANDBOX, or set QWEN_CODE_SUPPRESS_YOLO_WARNING=1 to silence this notice."];
4
5
  export declare function parseQwenResult(raw: string, nonce: string): ClassifiedWorkerResult;
5
6
  export declare const qwen: WorkerAdapter;
@@ -3,6 +3,13 @@ import { parseWorkerResult } from "./prompt.js";
3
3
  import { channelsFromConfig, shq, } from "./types.js";
4
4
  export const QWEN_VERSION_IDENTITY = /^\d+\.\d+\.\d+/;
5
5
  const QWEN_SKIP_UPDATE = "QWEN_CODE_SKIP_UPDATE_CHECK_ONCE=true";
6
+ // Verbatim stderr from tests/fixtures/qwen/safe-mode.stderr (qwen 0.21.15). These bytes are launch
7
+ // harness, not evidence that a reviewer took a turn. Keep the declaration closed and row-shaped so
8
+ // llm.ts can sweep partial terminal paints without a broad SAFE MODE/warning regex.
9
+ export const QWEN_HARNESS_BANNER_ROWS = [
10
+ "⚠ SAFE MODE — all customizations disabled (hooks, extensions, skills, MCP servers, QWEN.md). Restart without --safe-mode to resume normal operation.",
11
+ "Warning: running headless with --yolo / approval-mode=yolo and no sandbox. All tool calls (shell, write, edit) auto-execute at this process's privilege level. Enable a sandbox via --sandbox / QWEN_SANDBOX, or set QWEN_CODE_SUPPRESS_YOLO_WARNING=1 to silence this notice.",
12
+ ];
6
13
  function decodeQwenEvents(events) {
7
14
  const text = [];
8
15
  let failed = false;
@@ -136,6 +143,7 @@ export const qwen = {
136
143
  channels: (cfg) => channelsFromConfig("qwen", cfg),
137
144
  hardcodedFlags: { binary: "qwen", flags: ["--safe-mode", "--approval-mode", "-m", "-o", "-p"] },
138
145
  headlessCommand: (promptFile, model) => `${QWEN_SKIP_UPDATE} qwen --safe-mode --approval-mode yolo -m ${shq(model)} -o json -p '' < ${shq(promptFile)}`,
146
+ harnessBannerRows: QWEN_HARNESS_BANNER_ROWS,
139
147
  // OBS-905: qwen has NO interactive form. The `-i "$(cat prompt)"` TUI launch put the whole prompt in
140
148
  // argv (the OBS-889 leak-and-census shape) and produced a rendered transcript the JSON decoder above
141
149
  // can never read — under the herdr driver every qwen task read "unparseable" and merged only by harvest.
@@ -171,6 +171,7 @@ export interface WorkerAdapter {
171
171
  probe(): Promise<AuthHealth>;
172
172
  channels(cfg: TickmarkrConfig): BillingChannel[];
173
173
  headlessCommand(promptFile: string, model: string): string;
174
+ harnessBannerRows?: readonly string[];
174
175
  interactiveCommand(promptFile: string, model: string): string | null;
175
176
  interactiveSeed?: InteractiveSeed;
176
177
  resumeCommand?(sessionId: string, promptFile: string, model: string): string;
@@ -8,6 +8,12 @@ import { type VitestListResult } from "../../gates/acceptance.js";
8
8
  * concatenation and publishes no index, so the release listing is the only enumerable surface. */
9
9
  export declare const LIVEBENCH_RELEASES_URL = "https://api.github.com/repos/LiveBench/livebench.github.io/contents/public";
10
10
  export declare const LIVEBENCH_TABLE_MAX_AGE_DAYS = 90;
11
+ type ReviewDemotionSummary = {
12
+ reviewer: string;
13
+ count: number;
14
+ causes: Record<string, number>;
15
+ };
16
+ export declare function recentReviewDemotions(cwd: string, lastRuns?: number): ReviewDemotionSummary[];
11
17
  export type DoctorOpts = {
12
18
  banner?: boolean;
13
19
  kimiTurnProbe?: (cwd: string) => Promise<KimiDoctorTurnResult>;
@@ -1,6 +1,6 @@
1
1
  import { writeFileSync } from "node:fs";
2
2
  import { spawnSync } from "node:child_process";
3
- import { existsSync, readFileSync } from "node:fs";
3
+ import { existsSync, readFileSync, readdirSync } from "node:fs";
4
4
  import { join } from "node:path";
5
5
  import { detectPackageManager, turboContinueFindings } from "../../gates/baseline.js";
6
6
  import { version } from "./version.js";
@@ -26,6 +26,46 @@ const initialFetch = globalThis.fetch;
26
26
  const visual = () => process.stdout.isTTY === true && process.env.NO_COLOR === undefined;
27
27
  const alignedStatusRow = (verdict, key, value) => ` ${statusRow(verdict, kvRow(key, value).slice(2))}`;
28
28
  const attentionRow = (text) => ` ${statusRow("warn", text)}`;
29
+ // Use the routing profile's established 50-run recency horizon without importing journal machinery
30
+ // into doctor. Torn/malformed rows are ignored one at a time: diagnostics must survive a killed write.
31
+ export function recentReviewDemotions(cwd, lastRuns = 50) {
32
+ const runs = join(cwd, stateDirName(cwd), "runs");
33
+ if (!existsSync(runs))
34
+ return [];
35
+ const grouped = new Map();
36
+ const ids = readdirSync(runs, { withFileTypes: true })
37
+ .filter((entry) => entry.isDirectory() && entry.name.startsWith("run-")
38
+ && existsSync(join(runs, entry.name, "journal.jsonl")))
39
+ .map((entry) => entry.name).sort().slice(-lastRuns);
40
+ for (const id of ids) {
41
+ const lines = readFileSync(join(runs, id, "journal.jsonl"), "utf8").split("\n");
42
+ for (const line of lines) {
43
+ if (!line.trim())
44
+ continue;
45
+ let row;
46
+ try {
47
+ row = JSON.parse(line);
48
+ }
49
+ catch {
50
+ continue;
51
+ }
52
+ if (!row || typeof row !== "object")
53
+ continue;
54
+ const event = row;
55
+ if (event.event !== "review-pool-demotion" || !event.data || typeof event.data !== "object")
56
+ continue;
57
+ const data = event.data;
58
+ if (typeof data.reviewer !== "string" || !data.reviewer.trim())
59
+ continue;
60
+ const cause = typeof data.cause === "string" && data.cause.trim() ? data.cause : "unknown";
61
+ const summary = grouped.get(data.reviewer) ?? { reviewer: data.reviewer, count: 0, causes: {} };
62
+ summary.count++;
63
+ summary.causes[cause] = (summary.causes[cause] ?? 0) + 1;
64
+ grouped.set(data.reviewer, summary);
65
+ }
66
+ }
67
+ return [...grouped.values()].sort((a, b) => a.reviewer.localeCompare(b.reviewer));
68
+ }
29
69
  const ORCA_HOOK_ADAPTERS = {
30
70
  claude: "claude-code",
31
71
  "claude-code": "claude-code",
@@ -581,6 +621,15 @@ export async function doctor(_argv, cwd = process.cwd(), adapters = allAdapters(
581
621
  const healthy = h.installed && (a.id !== kimi.id || h.authed);
582
622
  return alignedStatusRow(healthy ? "pass" : "fail", a.id, state);
583
623
  });
624
+ const reviewDemotions = recentReviewDemotions(cwd);
625
+ if (reviewDemotions.length) {
626
+ rows.push(legend("recent review-seat demotions:"));
627
+ for (const demotion of reviewDemotions) {
628
+ const causes = Object.entries(demotion.causes)
629
+ .map(([cause, count]) => `${cause}${count === demotion.count ? "" : ` ×${count}`}`).join(", ");
630
+ rows.push(alignedStatusRow("warn", demotion.reviewer, `${demotion.count} review seat${demotion.count === 1 ? "" : "s"} demoted · cause ${causes}`));
631
+ }
632
+ }
584
633
  if (existsSync(graphPath(cwd))) {
585
634
  try {
586
635
  const graph = loadGraph(cwd);
@@ -35,6 +35,7 @@ const RAIL_TONES = {
35
35
  };
36
36
  /** Approval-close lifecycle labels extend the established closed rail vocabulary. */
37
37
  export const APPROVAL_RAIL_ROWS = {
38
+ "end-condition-held": { label: "close held", tone: "attention" },
38
39
  "approval-window-start": { label: "approval window", tone: "attention" },
39
40
  "approval-window-expired": { label: "approval window expired", tone: "attention" },
40
41
  "tip-verify-cancelled": { label: "tip verify cancelled", tone: "attention" },
@@ -124,6 +125,7 @@ export const RAIL_ROWS = {
124
125
  "gate-result": { label: "gate", tone: "pass" },
125
126
  "baseline-wait": { label: "baseline wait", tone: "active" },
126
127
  "suite-budget": { label: "suite budget", tone: "attention" },
128
+ "gate-replayed": { label: "gate replayed", tone: "attention" },
127
129
  "gate-reused": { label: "gate reused", tone: "neutral" },
128
130
  "judge-retry": { label: "judge retry", tone: "attention" },
129
131
  "review-no-verdict": { label: "review unavailable", tone: "attention" },
@@ -825,23 +825,17 @@ const liveness = (events, daemon, now = Date.now()) => {
825
825
  const state = daemon.state === "alive" ? "alive" : ended ? "finished" : "dead";
826
826
  return `last event ${age} ago · daemon pid ${daemon.pid} ${state}${cause ? ` · ${cause}` : ""}`;
827
827
  };
828
- // Resume's shared comparator remains the fail-closed baseline. Status may additionally accept the
829
- // daemon's audited graph-rehash release, but only when its `from` names that baseline identity (or
830
- // null for an explicitly released legacy journal) and its `to` names the graph loaded now.
828
+ // OBS-978: the one comparator resume, plan and the operator fold read decides the join — it already
829
+ // audits the rehash chain. When it binds and the journal holds a graph-rehash, the newest row is the
830
+ // audit that bound the loaded graph; rehashAt marks the facts recorded against a prior graph.
831
831
  const statusEngagement = (events, loadedHash) => {
832
- const baseline = engagementComparable(events, loadedHash);
832
+ if (!engagementComparable(events, loadedHash).comparable)
833
+ return { comparable: false };
833
834
  for (let i = events.length - 1; i >= 0; i--) {
834
- const event = events[i];
835
- if (event.event !== "graph-rehash")
836
- continue;
837
- const auditsBaseline = baseline.comparable || baseline.reason === "mismatch"
838
- ? event.data.from === baseline.recorded
839
- : event.data.from === null;
840
- return event.data.to === loadedHash && auditsBaseline
841
- ? { comparable: true, rehashAt: i }
842
- : { comparable: false };
835
+ if (events[i].event === "graph-rehash")
836
+ return { comparable: true, rehashAt: i };
843
837
  }
844
- return { comparable: baseline.comparable };
838
+ return { comparable: true };
845
839
  };
846
840
  // The journal's own reader rule (src/run/journal.ts readJsonl), applied to bytes already in hand:
847
841
  // skip blanks, drop a line that will not parse (a torn trailing write after a crash), keep the rest.
@@ -13,6 +13,12 @@ export type OwnershipFinding = {
13
13
  taskIds: string[];
14
14
  corroboration?: OwnershipCorroboration;
15
15
  detail: string;
16
+ } | {
17
+ code: "unowned-shape-oracle";
18
+ taskId: string;
19
+ source: string;
20
+ oracle: string;
21
+ detail: string;
16
22
  } | {
17
23
  code: "test-path-outside-allowlist";
18
24
  taskId: string;
@@ -26,6 +32,9 @@ export type OwnershipFinding = {
26
32
  path: string;
27
33
  detail: string;
28
34
  };
35
+ export declare const SHAPE_ORACLES: readonly ["tests/run/narration.test.ts", "tests/cli/brand-surfaces.test.ts", "tests/run/notify-identity.test.ts", "tests/run/outcome-projections.test.ts", "tests/cockpit/setup.test.ts"];
36
+ export declare const SHAPE_ORACLE_SOURCES: readonly ["src/run/daemon.ts", "src/cli/commands/run.ts"];
37
+ export declare const SHAPE_ORACLE_MAP: Record<string, readonly string[]>;
29
38
  /**
30
39
  * Cross-task ownership evidence. Findings are data: this checker never throws or changes the graph;
31
40
  * the compile seam promotes only a corroborated unowned-test finding and reports every other shape.
@@ -2,6 +2,38 @@ import { readFileSync, readdirSync } from "node:fs";
2
2
  import { basename, extname, join, posix } from "node:path";
3
3
  import { filesGlob } from "../graph/files-glob.js";
4
4
  import { collateralHits } from "./collateral.js";
5
+ export const SHAPE_ORACLES = [
6
+ "tests/run/narration.test.ts",
7
+ "tests/cli/brand-surfaces.test.ts",
8
+ "tests/run/notify-identity.test.ts",
9
+ "tests/run/outcome-projections.test.ts",
10
+ "tests/cockpit/setup.test.ts",
11
+ ];
12
+ export const SHAPE_ORACLE_SOURCES = [
13
+ "src/run/daemon.ts",
14
+ "src/cli/commands/run.ts",
15
+ ];
16
+ export const SHAPE_ORACLE_MAP = {
17
+ "src/run/daemon.ts": SHAPE_ORACLES,
18
+ "src/cli/commands/run.ts": SHAPE_ORACLES,
19
+ };
20
+ // Anchored-glob only: a files[] entry touches a mapped source when it names the source (or its
21
+ // extensionless stem) exactly, or when its glob's literal head — everything before the first
22
+ // wildcard — is the stem plus a literal dot, i.e. the wildcard only ever spans the extension
23
+ // ("src/run/daemon.*"). A broad multi-file glob like "src/**" that merely happens to cover the
24
+ // source is an unrelated task casting a wide net, not one touching daemon narration — that
25
+ // distinction is what broke every fixture using tests/fixtures/sample.prd.md's files: src/** task.
26
+ function touchesSource(files, source) {
27
+ const stem = source.replace(/\.(?:[cm]?[jt]sx?)$/, "");
28
+ return files.some((entry) => {
29
+ if (entry === source || entry === stem)
30
+ return true;
31
+ const special = entry.search(/[*?{[]/);
32
+ // Leg-2 v2.5.3: the anchored head is necessary, not sufficient — the pattern must also MATCH the
33
+ // source under the shared matcher, or "src/run/daemon.{js,jsx}" (never daemon.ts) would count as a touch.
34
+ return special !== -1 && entry.slice(0, special) === `${stem}.` && filesGlob([entry])(source);
35
+ });
36
+ }
5
37
  const normalize = (path) => path.replace(/^\.\//, "").split("\\").join("/");
6
38
  function testSources(repoRoot) {
7
39
  const root = join(repoRoot, "tests");
@@ -155,6 +187,7 @@ export function ownershipFindings(tasks, repoRoot) {
155
187
  const context = task.context.map(normalize);
156
188
  return {
157
189
  task,
190
+ files,
158
191
  owns: files.length === 0 ? () => false : filesGlob(files),
159
192
  allows: files.length === 0 ? () => true : filesGlob([...files, ...context]),
160
193
  };
@@ -204,6 +237,23 @@ export function ownershipFindings(tasks, repoRoot) {
204
237
  });
205
238
  }
206
239
  }
240
+ for (const entry of indexed) {
241
+ for (const [source, oracles] of Object.entries(SHAPE_ORACLE_MAP)) {
242
+ if (!touchesSource(entry.files, source))
243
+ continue;
244
+ for (const oracle of oracles) {
245
+ if (!entry.owns(oracle)) {
246
+ findings.push({
247
+ code: "unowned-shape-oracle",
248
+ taskId: entry.task.id,
249
+ source,
250
+ oracle,
251
+ detail: `${entry.task.id} touches ${source} without owning shape oracle ${oracle}`,
252
+ });
253
+ }
254
+ }
255
+ }
256
+ }
207
257
  for (const source of sources) {
208
258
  for (const owner of owners(source.path)) {
209
259
  for (const path of repositoryPaths(source.text)) {
@@ -234,11 +284,24 @@ export function ownershipFindings(tasks, repoRoot) {
234
284
  }
235
285
  }
236
286
  }
237
- return findings.sort((a, b) => `${a.code}:${"test" in a ? a.test : a.path}:${"taskId" in a ? a.taskId : ""}`.localeCompare(`${b.code}:${"test" in b ? b.test : b.path}:${"taskId" in b ? b.taskId : ""}`));
287
+ const target = (f) => {
288
+ switch (f.code) {
289
+ case "unowned-test": return f.test;
290
+ case "unowned-shape-oracle": return f.oracle;
291
+ case "test-path-outside-allowlist": return f.test;
292
+ case "unordered-context-write": return f.path;
293
+ }
294
+ };
295
+ return findings.sort((a, b) => {
296
+ const taskA = "taskId" in a ? a.taskId : "";
297
+ const taskB = "taskId" in b ? b.taskId : "";
298
+ return `${a.code}:${target(a)}:${taskA}`.localeCompare(`${b.code}:${target(b)}:${taskB}`);
299
+ });
238
300
  }
239
301
  export function renderOwnershipFinding(finding) {
240
302
  return `tickmarkr: ownership-lint[${finding.code}]: ${finding.detail}`;
241
303
  }
242
304
  export function blocksCompile(finding) {
243
- return finding.code === "unowned-test" && finding.corroboration !== undefined;
305
+ return (finding.code === "unowned-test" && finding.corroboration !== undefined)
306
+ || finding.code === "unowned-shape-oracle";
244
307
  }
@@ -161,6 +161,7 @@ export declare function detectVacuousOracles(cwd: string, tasks: ReadonlyArray<{
161
161
  export declare function compareToBaseline(cwd: string, commands: Record<string, string>, baseline: Baseline, enabled: string[], opts?: {
162
162
  rerunOf?: HostStarvedRerun;
163
163
  infraRerun?: HostStarvedRerun;
164
+ selected?: readonly string[];
164
165
  }): Promise<GateResult[]>;
165
166
  export interface HostStarvedRerun {
166
167
  durationMs: number;
@@ -184,7 +185,10 @@ export declare function resetCalmWindowForTests(): void;
184
185
  export declare function waitForCalmWindow(): Promise<number>;
185
186
  /** Summary totals survive digit-normalized fingerprints and exclude test-owned echoed output. */
186
187
  export declare function runnerFileCount(raw: string): number | null;
187
- export declare function fileCountDeficit(entry: BaselineCommand | undefined, raw: string): string | undefined;
188
+ export declare function fileCountDeficit(entry: BaselineCommand | undefined, raw: string, opts?: {
189
+ name?: string;
190
+ selected?: readonly string[];
191
+ }): string | undefined;
188
192
  /** Both readers classify fresh evidence first, retaining the whole-output guard against infra forgiveness. */
189
193
  export declare function classifyFreshRunnerOutput(entry: BaselineCommand | undefined, raw: string, code: number): FailureClassification | undefined;
190
194
  export {};
@@ -627,7 +627,16 @@ export async function compareToBaseline(cwd, commands, baseline, enabled, opts =
627
627
  details: `${g.meta?.classification === "infra" ? "infra; " : ""}runner-infra rerun after waiting ${opts.infraRerun.waitedMs}ms for a calm load window: ${withRerun.details.replace(/^infra; /, "")}`,
628
628
  meta: { ...withRerun.meta, runnerInfraRerun: opts.infraRerun },
629
629
  } : withRerun;
630
- results.push(r.capacity ? { ...final, capacity: r.capacity } : final);
630
+ const withSelected = name === "test" && opts.selected
631
+ ? {
632
+ ...final,
633
+ meta: {
634
+ ...final.meta,
635
+ ...(Array.isArray(opts.selected) ? { selectedTests: [...opts.selected] } : {}),
636
+ },
637
+ }
638
+ : final;
639
+ results.push(r.capacity ? { ...withSelected, capacity: r.capacity } : withSelected);
631
640
  };
632
641
  // …and whether the entry that would forgive this command was measured in the same world. A
633
642
  // baseline captured under a different fork cap forgives nothing: its fingerprints describe a
@@ -657,7 +666,7 @@ export async function compareToBaseline(cwd, commands, baseline, enabled, opts =
657
666
  continue;
658
667
  }
659
668
  const raw = (r.stdout + "\n" + r.stderr).split(cwd).join("");
660
- const deficit = fileCountDeficit(entry, raw);
669
+ const deficit = fileCountDeficit(entry, raw, { name, selected: opts.selected });
661
670
  if (deficit) {
662
671
  record({ gate: name, pass: false, details: deficit, meta: { classification: "infra", infra: true } });
663
672
  continue;
@@ -689,7 +698,7 @@ export async function compareToBaseline(cwd, commands, baseline, enabled, opts =
689
698
  if (name === "test" && classification === "infra" && !rerunOf && !opts.infraRerun) {
690
699
  const waitedMs = await waitForCalmWindow();
691
700
  const provenance = { durationMs: r.durationMs ?? 0, referenceMs: entry?.durationMs ?? 0, waitedMs };
692
- results.push(...await compareToBaseline(cwd, { [name]: cmd }, baseline, [name], { infraRerun: provenance }));
701
+ results.push(...await compareToBaseline(cwd, { [name]: cmd }, baseline, [name], { ...opts, infraRerun: provenance }));
693
702
  continue;
694
703
  }
695
704
  // OBS-896: every fresh failure must be timeout-class, and the suite must take at least twice its
@@ -698,7 +707,7 @@ export async function compareToBaseline(cwd, commands, baseline, enabled, opts =
698
707
  && hostStarved(failing.join("\n"), r.durationMs ?? 0, entry?.durationMs)) {
699
708
  const waitedMs = await waitForCalmWindow();
700
709
  const provenance = { durationMs: r.durationMs ?? 0, referenceMs: entry?.durationMs ?? 0, waitedMs };
701
- results.push(...await compareToBaseline(cwd, { [name]: cmd }, baseline, [name], { rerunOf: provenance }));
710
+ results.push(...await compareToBaseline(cwd, { [name]: cmd }, baseline, [name], { ...opts, rerunOf: provenance }));
702
711
  continue;
703
712
  }
704
713
  if (classification === "infra") {
@@ -833,7 +842,11 @@ export function runnerFileCount(raw) {
833
842
  });
834
843
  return counts.length ? counts.reduce((sum, count) => sum + count, 0) : null;
835
844
  }
836
- export function fileCountDeficit(entry, raw) {
845
+ export function fileCountDeficit(entry, raw, opts) {
846
+ // OBS-985: only a real, named selected-test run of the TEST gate is exempt — a truthy flag or an
847
+ // empty list named no selection and let a full-suite call opt itself out of the deficit guard.
848
+ if (opts?.name === "test" && opts.selected !== undefined && opts.selected.length > 0)
849
+ return undefined;
837
850
  const actual = runnerFileCount(raw);
838
851
  return entry?.fileCount != null && actual !== null && actual < entry.fileCount
839
852
  ? `infra; runner reported ${actual} test files, below baseline ${entry.fileCount} — suite incomplete`
@@ -65,7 +65,7 @@ export interface LlmRunResult {
65
65
  seatAuthoredBytes?: number;
66
66
  }
67
67
  export declare const PROMPT_GLYPHS: readonly ["➜", "❯", "$", "%", ">>", ">"];
68
- export declare function reviewSeatOutput(raw: string, nonce: string): string;
68
+ export declare function reviewSeatOutput(raw: string, nonce: string, adapterBannerRows?: readonly string[]): string;
69
69
  export declare const REVIEW_FIRST_LIVENESS_MS = 30000;
70
70
  export declare function runHeadless(adapter: WorkerAdapter, model: string, prompt: string, cwd: string, timeoutMs?: number): Promise<string>;
71
71
  export declare function runViaDriver(adapter: WorkerAdapter, model: string, prompt: string, cwd: string, via: LlmVia, timeoutMs?: number): Promise<string>;
package/dist/gates/llm.js CHANGED
@@ -204,12 +204,14 @@ const IDENTITY_OPENERS = ["review ·", "tickmarkr"];
204
204
  // Stages of the preamble walk. Each complete harness row is accepted only at or after its stage.
205
205
  const ECHO = 0, START = 1, BANNER = 2, IDENTITY = 3, SEAT = 4;
206
206
  // The char offset where the seat's own text begins, or -1 when the capture ends inside the preamble.
207
- function seatStart(output) {
207
+ function seatStart(output, adapterBannerRows) {
208
208
  const rows = output.split("\n");
209
209
  if (rows.length > 1 && rows[rows.length - 1] === "")
210
210
  rows.pop(); // the read's own line terminator
211
211
  let stage = ECHO;
212
212
  let bannerAt; // next banner row expected once the banner has begun
213
+ let adapterBannerAt;
214
+ let identitySeen = false;
213
215
  let offset = 0;
214
216
  for (let i = 0; i < rows.length; i++) {
215
217
  const row = rows[i].replace(/[ \t]+$/, "");
@@ -217,7 +219,7 @@ function seatStart(output) {
217
219
  const last = i === rows.length - 1;
218
220
  // Complete harness rows: equality against the shape the preamble allows at this stage.
219
221
  let accepted = false;
220
- if (stage < SEAT && t.length === 0)
222
+ if (!identitySeen && stage < SEAT && t.length === 0)
221
223
  accepted = true; // blank rows between preamble rows
222
224
  else if (stage <= ECHO && completeEchoRow(row))
223
225
  accepted = true;
@@ -225,17 +227,27 @@ function seatStart(output) {
225
227
  stage = BANNER;
226
228
  accepted = true;
227
229
  }
228
- else if (stage <= BANNER && bannerAt === undefined && BANNER_ROWS.includes(row)) {
230
+ else if (!identitySeen && stage <= BANNER && bannerAt === undefined && BANNER_ROWS.includes(row)) {
229
231
  stage = BANNER;
230
232
  bannerAt = BANNER_ROWS.indexOf(row) + 1;
231
233
  accepted = true;
232
234
  }
233
- else if (stage <= BANNER && bannerAt !== undefined && row === BANNER_ROWS[bannerAt]) {
235
+ else if (!identitySeen && stage <= BANNER && bannerAt !== undefined && row === BANNER_ROWS[bannerAt]) {
234
236
  bannerAt++;
235
237
  accepted = true;
236
238
  }
237
- else if (stage <= IDENTITY && IDENTITY_LINE.test(t)) {
238
- stage = SEAT;
239
+ else if (stage <= IDENTITY && adapterBannerAt === undefined && adapterBannerRows.includes(row)) {
240
+ stage = BANNER;
241
+ adapterBannerAt = adapterBannerRows.indexOf(row) + 1;
242
+ accepted = true;
243
+ }
244
+ else if (stage <= IDENTITY && adapterBannerAt !== undefined && row === adapterBannerRows[adapterBannerAt]) {
245
+ adapterBannerAt++;
246
+ accepted = true;
247
+ }
248
+ else if (!identitySeen && stage <= IDENTITY && IDENTITY_LINE.test(t)) {
249
+ stage = IDENTITY;
250
+ identitySeen = true;
239
251
  accepted = true;
240
252
  }
241
253
  if (accepted) {
@@ -253,9 +265,16 @@ function seatStart(output) {
253
265
  ? BANNER_ROWS.some((b) => b.startsWith(row))
254
266
  : BANNER_ROWS[bannerAt]?.startsWith(row) === true))
255
267
  return -1;
268
+ if (stage <= IDENTITY && (adapterBannerAt === undefined
269
+ ? adapterBannerRows.some((b) => b.startsWith(row))
270
+ : adapterBannerRows[adapterBannerAt]?.startsWith(row) === true))
271
+ return -1;
256
272
  // The identity row is painted right after the banner, so its prefix is a partial paint only there;
257
273
  // with no banner in the capture, "review" or "tick" alone is the seat's own first row.
258
- if (stage <= IDENTITY && bannerAt === BANNER_ROWS.length && IDENTITY_OPENERS.some((o) => o.startsWith(t)))
274
+ if (!identitySeen && stage <= IDENTITY
275
+ && (bannerAt === BANNER_ROWS.length
276
+ || (adapterBannerRows.length > 0 && adapterBannerAt === adapterBannerRows.length))
277
+ && IDENTITY_OPENERS.some((o) => o.startsWith(t)))
259
278
  return -1;
260
279
  return offset;
261
280
  }
@@ -266,9 +285,9 @@ function seatStart(output) {
266
285
  // Every capture taken before the preamble finishes therefore measures ZERO seat-authored bytes,
267
286
  // which is what makes the caller's running Math.max safe: a partial banner counted once would be
268
287
  // retained for the whole call and buy a silent seat its full ceiling.
269
- export function reviewSeatOutput(raw, nonce) {
288
+ export function reviewSeatOutput(raw, nonce, adapterBannerRows = []) {
270
289
  const output = stripVTControlCharacters(raw).replace(/\r\n?/g, "\n");
271
- const start = seatStart(output);
290
+ const start = seatStart(output, adapterBannerRows);
272
291
  if (start < 0)
273
292
  return "";
274
293
  const seat = output.slice(start);
@@ -287,7 +306,10 @@ async function runHeadlessDetailed(adapter, model, prompt, cwd, timeoutMs = 3000
287
306
  const pf = join(dir, "prompt.md");
288
307
  writeFileSync(pf, prompt);
289
308
  const r = await sh(adapter.headlessCommand(pf, model), cwd, timeoutMs);
290
- return { output: r.stdout + "\n" + r.stderr, exitCode: r.code, timedOut: r.timedOut === true, seatAuthoredBytes: Buffer.byteLength((r.stdout + r.stderr).trim()) };
309
+ const output = r.stdout + "\n" + r.stderr;
310
+ const nonce = extractPromptNonce(prompt) ?? "";
311
+ return { output, exitCode: r.code, timedOut: r.timedOut === true,
312
+ seatAuthoredBytes: Buffer.byteLength(reviewSeatOutput(output, nonce, adapter.harnessBannerRows).trim()) };
291
313
  }
292
314
  finally {
293
315
  rmSync(dir, { recursive: true, force: true });
@@ -349,7 +371,7 @@ async function runViaDriverDetailed(adapter, model, prompt, cwd, via, timeoutMs
349
371
  const startedAt = Date.now();
350
372
  out = await via.driver.read(slot, 400);
351
373
  const reviewing = prompt.startsWith("TICKMARKR-REVIEW");
352
- seatAuthoredBytes = Buffer.byteLength(reviewSeatOutput(out, nonce));
374
+ seatAuthoredBytes = Buffer.byteLength(reviewSeatOutput(out, nonce, adapter.harnessBannerRows));
353
375
  let firstLivenessObserved = false;
354
376
  let priorSnapshot = normalizeStallSnapshot(out);
355
377
  const anchoredAt = Date.now();
@@ -365,7 +387,7 @@ async function runViaDriverDetailed(adapter, model, prompt, cwd, via, timeoutMs
365
387
  const matched = await via.driver.waitOutput(slot, exitPattern, sliceMs, { regex: true });
366
388
  const raw = await via.driver.read(slot, 400);
367
389
  out = raw;
368
- seatAuthoredBytes = Math.max(seatAuthoredBytes, Buffer.byteLength(reviewSeatOutput(raw, nonce)));
390
+ seatAuthoredBytes = Math.max(seatAuthoredBytes, Buffer.byteLength(reviewSeatOutput(raw, nonce, adapter.harnessBannerRows)));
369
391
  // waitOutput is the driver's authoritative marker match. The raw check covers drivers whose
370
392
  // wait timed out at the same boundary the marker landed; either way a trailer completes
371
393
  // normally and is never mistaken for inactivity.
@@ -53,6 +53,17 @@ export declare function isDiffCapPark(result: GateResult): boolean;
53
53
  export declare function diffCapParkReason(results: GateResult[]): string | null;
54
54
  export declare function modelId(model: string): string;
55
55
  export { modelProvider };
56
+ /**
57
+ * One function decides whether a reviewer's `resolved` or `reraised` id names a carried fingerprint,
58
+ * comparing both sides with every whitespace run removed (`s.replace(/\s+/g, "")`).
59
+ */
60
+ export declare function matchClosureId(candidate: unknown, fingerprint: string): boolean;
61
+ export declare function matchClosureId(candidate: unknown, fingerprints: Iterable<string>): string | undefined;
62
+ /**
63
+ * Validates closure ids in a review verdict: membership, duplication, and coverage of every prior id
64
+ * all route through matchClosureId.
65
+ */
66
+ export declare function isReviewClosureInvalid(v: Pick<ReviewVerdict, "resolved" | "reraised"> | null | undefined, priorIds: ReadonlySet<string> | readonly string[]): boolean;
56
67
  export declare function pickReviewer(author: Assignment, channels: BillingChannel[], exclude?: string[], // v1.1 failover: reviewer channels that already produced garbage for this task
57
68
  prefer?: string[], // v1.53 T2: review.prefer — reorders eligible channels, never changes eligibility
58
69
  floor?: Tier, // task-declared only; config floors govern workers and must not silently move review seats
@@ -1,4 +1,4 @@
1
- import { writeFileSync } from "node:fs";
1
+ import { existsSync, writeFileSync } from "node:fs";
2
2
  import { join } from "node:path";
3
3
  import { channelKey, shq } from "../adapters/types.js";
4
4
  import { criticalPathHits, DEFAULT_DIFF_CAP, DEFAULT_REVIEW_CRITICAL_PATHS, declaredReviewPolicy, isReviewLeafPath, raiseReviewPolicy, REVIEW_VERSION_MIRRORS, TIER_RANK, } from "../config/config.js";
@@ -168,6 +168,31 @@ export function modelId(model) {
168
168
  return model.slice(model.lastIndexOf("/") + 1);
169
169
  }
170
170
  export { modelProvider };
171
+ export function matchClosureId(candidate, target) {
172
+ if (typeof candidate !== "string")
173
+ return typeof target === "string" ? false : undefined;
174
+ const normCandidate = candidate.replace(/\s+/g, "");
175
+ if (typeof target === "string") {
176
+ return normCandidate === target.replace(/\s+/g, "");
177
+ }
178
+ for (const fp of target) {
179
+ if (typeof fp === "string" && normCandidate === fp.replace(/\s+/g, ""))
180
+ return fp;
181
+ }
182
+ return undefined;
183
+ }
184
+ /**
185
+ * Validates closure ids in a review verdict: membership, duplication, and coverage of every prior id
186
+ * all route through matchClosureId.
187
+ */
188
+ export function isReviewClosureInvalid(v, priorIds) {
189
+ const priors = priorIds instanceof Set ? priorIds : new Set(priorIds);
190
+ const closureLists = [v?.resolved, v?.reraised];
191
+ const allCandidateIds = [...(v?.resolved ?? []), ...(v?.reraised ?? [])];
192
+ return !!v && (priors.size > 0 || closureLists.some((list) => list !== undefined)) && (closureLists.some((list) => !Array.isArray(list) || list.some((id) => !matchClosureId(id, priors)))
193
+ || new Set(allCandidateIds.map((id) => matchClosureId(id, priors) ?? id)).size !== allCandidateIds.length
194
+ || [...priors].some((id) => !allCandidateIds.some((candidate) => matchClosureId(candidate, id))));
195
+ }
171
196
  // v1.53 T2: same entry grammar as routing.map.prefer (router.ts preferIndex — router is out of this
172
197
  // module's dependency direction for a private fn, so the 3 lines live here too): `adapter` matches
173
198
  // every channel of that adapter, `adapter:model` exactly one; unmatched channels sort after all entries.
@@ -357,7 +382,19 @@ The top-level comments array is optional. Use it only for actionable line-anchor
357
382
  // make two otherwise-identical runs diverge in their journal bytes. The reviewer channel already
358
383
  // disambiguates every call that matters: a retry always excludes the flaked channel (run-gates.ts),
359
384
  // so it can never collide with the attempt it replaces.
360
- const artifactId = `${task.id}-${channelKey(reviewer).replace(/[^a-zA-Z0-9_.-]/g, "-")}`;
385
+ const baseArtifactId = `${task.id}-${channelKey(reviewer).replace(/[^a-zA-Z0-9_.-]/g, "-")}`;
386
+ let artifactId = baseArtifactId;
387
+ if (artifactDir) {
388
+ if (existsSync(join(artifactDir, `review-brief-${baseArtifactId}.md`)) ||
389
+ existsSync(join(artifactDir, `review-raw-${baseArtifactId}.txt`))) {
390
+ let counter = 2;
391
+ while (existsSync(join(artifactDir, `review-brief-${baseArtifactId}-${counter}.md`)) ||
392
+ existsSync(join(artifactDir, `review-raw-${baseArtifactId}-${counter}.txt`))) {
393
+ counter++;
394
+ }
395
+ artifactId = `${baseArtifactId}-${counter}`;
396
+ }
397
+ }
361
398
  const briefPath = artifactDir ? join(artifactDir, `review-brief-${artifactId}.md`) : undefined;
362
399
  // Persistence is evidence, not a gate input: a full disk or a removed run dir never fails the gate.
363
400
  let savedBrief;
@@ -397,10 +434,7 @@ The top-level comments array is optional. Use it only for actionable line-anchor
397
434
  const v = extractVerdictJson(raw, nonce);
398
435
  const findings = v && Array.isArray(v.findings) ? v.findings : null;
399
436
  const priorIds = new Set(priorMaterials.map((finding) => finding.fingerprint));
400
- const closureLists = [v?.resolved, v?.reraised];
401
- const closureInvalid = !!v && (priorIds.size > 0 || closureLists.some((list) => list !== undefined)) && (closureLists.some((list) => !Array.isArray(list) || list.some((id) => typeof id !== "string" || !priorIds.has(id)))
402
- || new Set([...(v?.resolved ?? []), ...(v?.reraised ?? [])]).size !== (v?.resolved?.length ?? 0) + (v?.reraised?.length ?? 0)
403
- || [...priorIds].some((id) => !v?.resolved?.includes(id) && !v?.reraised?.includes(id)));
437
+ const closureInvalid = isReviewClosureInvalid(v, priorIds);
404
438
  // findings decides the verdict on its own; the legacy path still needs approve + issues to parse.
405
439
  if (!v || closureInvalid || (findings === null && (typeof v.approve !== "boolean" || !Array.isArray(v.issues)))) {
406
440
  // OBS-196: name the cause and persist the raw bytes — a ruled-on "unparseable" without its
@@ -434,7 +468,7 @@ The top-level comments array is optional. Use it only for actionable line-anchor
434
468
  const decided = findings !== null
435
469
  ? classifyReviewFindings(findings)
436
470
  : classifyReviewIssues(v.approve, v.issues);
437
- const reraised = priorMaterials.filter((finding) => v.reraised?.includes(finding.fingerprint));
471
+ const reraised = priorMaterials.filter((finding) => v.reraised?.some((id) => matchClosureId(id, finding.fingerprint)));
438
472
  if (reraised.length) {
439
473
  if (decided.pass)
440
474
  decided.headline = "requested changes";
@@ -455,7 +489,15 @@ The top-level comments array is optional. Use it only for actionable line-anchor
455
489
  details,
456
490
  meta: {
457
491
  ...policyMeta, ...rotationMeta, reviewer: channelKey(reviewer), vendor: reviewer.vendor, provider,
458
- ...(priorMaterials.length ? { resolved: v.resolved, reraised: v.reraised } : {}),
492
+ ...(priorMaterials.length ? {
493
+ resolved: v.resolved,
494
+ reraised: v.reraised,
495
+ normalisedMatches: (v.resolved ?? []).map((id) => matchClosureId(id, priorIds)).filter((id) => id !== undefined),
496
+ resolvedMatches: (v.resolved ?? []).map((id) => matchClosureId(id, priorIds)).filter((id) => id !== undefined),
497
+ reraisedMatches: (v.reraised ?? []).map((id) => matchClosureId(id, priorIds)).filter((id) => id !== undefined),
498
+ normalisedResolved: (v.resolved ?? []).map((id) => matchClosureId(id, priorIds)).filter((id) => id !== undefined),
499
+ normalisedReraised: (v.reraised ?? []).map((id) => matchClosureId(id, priorIds)).filter((id) => id !== undefined),
500
+ } : {}),
459
501
  ...(reraised.length ? { findings: [
460
502
  ...structuredFindings("review", details).filter((finding) => !reraised.some((prior) => prior.note === finding.note)),
461
503
  ...reraised,
@@ -366,7 +366,7 @@ export async function runGates(task, ctx) {
366
366
  // ponytail: legacy runs adjacent tools in ONE compareToBaseline call, so there is one interval
367
367
  // to measure and each of its gates carries it. Split it only if this branch ever stops batching.
368
368
  const finish = startMeasurement();
369
- const toolResults = await compareToBaseline(ctx.worktree, commands, ctx.baseline, [...gates]);
369
+ const toolResults = await compareToBaseline(ctx.worktree, commands, ctx.baseline, [...gates], selected ? { selected } : {});
370
370
  const batch = finish();
371
371
  for (const g of gates)
372
372
  addMeasurement(g, batch);
@@ -386,7 +386,7 @@ export async function runGates(task, ctx) {
386
386
  // any later tool before anyone reads its verdict.
387
387
  for (const g of gates) {
388
388
  await emitStart(g);
389
- const [r] = await measure(g, () => compareToBaseline(ctx.worktree, commands, ctx.baseline, [g]));
389
+ const [r] = await measure(g, () => compareToBaseline(ctx.worktree, commands, ctx.baseline, [g], g === "test" && selected ? { selected } : {}));
390
390
  // the screen's interval IS the test gate's first interval, so the split needs no second clock
391
391
  if (g === "test" && selected)
392
392
  selectedDurationMs = spans.get("test").durationMs;
@@ -109,7 +109,9 @@ export declare const SUITE_WAIT_CEILING_MS = 600000;
109
109
  export declare const setSuiteWaitCeilingForTests: (ms: number) => void;
110
110
  export declare const resetSuiteWaitCeilingForTests: () => void;
111
111
  export declare const APPROVAL_POLL_MS = 250;
112
- export declare const APPROVAL_WINDOW_MS = 1000;
112
+ export declare const APPROVAL_WINDOW_MS = 120000;
113
+ export declare const setApprovalWindowForTests: (ms: number) => void;
114
+ export declare const resetApprovalWindowForTests: () => void;
113
115
  export declare const EARLY_LAUNCH_LIVENESS_MS = 60000;
114
116
  /** Test seam — lowers the empty-pane liveness window without sleeping 60s per case. */
115
117
  export declare function setEarlyLaunchLivenessMsForTests(ms: number): void;
@@ -307,7 +307,13 @@ let suiteWaitCeilingMs = SUITE_WAIT_CEILING_MS;
307
307
  export const setSuiteWaitCeilingForTests = (ms) => { suiteWaitCeilingMs = ms; };
308
308
  export const resetSuiteWaitCeilingForTests = () => { suiteWaitCeilingMs = SUITE_WAIT_CEILING_MS; };
309
309
  export const APPROVAL_POLL_MS = 250;
310
- export const APPROVAL_WINDOW_MS = 1_000;
310
+ export const APPROVAL_WINDOW_MS = 120_000;
311
+ // Keep ordinary park tests off the operator's production wait. Explicit timing tests
312
+ // use the same setter/reset pattern as the suite wait ceiling above.
313
+ const DEFAULT_APPROVAL_WINDOW_MS = process.env.VITEST ? 1 : APPROVAL_WINDOW_MS;
314
+ let approvalWindowMs = DEFAULT_APPROVAL_WINDOW_MS;
315
+ export const setApprovalWindowForTests = (ms) => { approvalWindowMs = ms; };
316
+ export const resetApprovalWindowForTests = () => { approvalWindowMs = DEFAULT_APPROVAL_WINDOW_MS; };
311
317
  const PROVIDER_DEATH_REQUEUE_CAP = 2; // v1.46 T1: requeue same assignment twice, then fall through to the normal ladder
312
318
  const PROVIDER_DEATH_BACKOFF_MS = 500; // short backoff before provider-death requeue
313
319
  const NO_TRAILER_DEMOTION_STREAK = 2; // OBS-57: consecutive no-trailer windows demote a channel for the rest of the run
@@ -1768,6 +1774,17 @@ export async function runDaemon(repoRoot, opts = {}) {
1768
1774
  return `${identity} — release with ${commands.map((command) => `\`${command}\``).join(" or ")}`;
1769
1775
  };
1770
1776
  const park = async (t, reason, kind, assignment, attempts, startMs, gateFails = 0, consults = 0, tokens, metered = 0, retryMode = "fresh", details = {}) => {
1777
+ // OBS-979: a worker refusal can identify the missing authoring scope even when gates
1778
+ // subsequently supply the park's disposition. Keep that actionable path on the park itself.
1779
+ const worker = journal.read().reverse().find((e) => e.taskId === t.id && e.event === "worker-result");
1780
+ if (t.files.length > 0 && worker?.data.ok === false && typeof worker.data.summary === "string") {
1781
+ const allowed = filesGlob(t.files);
1782
+ const paths = [...worker.data.summary.matchAll(/(?:^|[\s`'"(])((?:[A-Za-z0-9_@.()[\]-]+\/)+[A-Za-z0-9_@.[\]-]+|[A-Za-z0-9_@-]+(?:\.[A-Za-z0-9_-]+)+)(?=$|[\s`'"),:;.!?])/g)]
1783
+ .map((match) => match[1].replace(/^\.\//, "").replace(/\.$/, ""))
1784
+ .filter((path) => !path.split("/").includes("..") && !allowed(path));
1785
+ if (paths.length)
1786
+ reason += ` — files[] repair hint: ${[...new Set(paths)].join(", ")}`;
1787
+ }
1771
1788
  graph = setStatus(graph, t.id, "human");
1772
1789
  saveGraph(repoRoot, graph);
1773
1790
  journal.append("task-human", t.id, { ...details, reason, kind });
@@ -2026,6 +2043,7 @@ export async function runDaemon(repoRoot, opts = {}) {
2026
2043
  gate: g.gate, ...(unverdicted ? {} : { pass: g.pass }), details: g.details,
2027
2044
  ...(gateSubject ? { commit: gateSubject.commit, attempt: gateSubject.attempt } : {}),
2028
2045
  ...(gateSubject?.replayMeasurement ? { replayMeasurement: true } : {}),
2046
+ ...(gateSubject?.replayedFromAttempt !== undefined ? { replayedFromAttempt: gateSubject.replayedFromAttempt } : {}),
2029
2047
  ...(g.meta?.skipped === true || noVerdictReview ? { skipped: true } : {}),
2030
2048
  // T9: an infra-only exit is journaled AS one. The operator reading a red `test` row has to
2031
2049
  // be able to tell "the suite found a defect" from "the runner never ran", and the merge
@@ -4136,33 +4154,64 @@ export async function runDaemon(repoRoot, opts = {}) {
4136
4154
  const gated = await gitHead(wt);
4137
4155
  gateSubject = { commit: await gateCommitSubject(taskBase, gated, wt), attempt };
4138
4156
  await trackedDriver.project?.(t.id, "in-review");
4157
+ // Only the immediately preceding attempt can lend a red. Re-journal its results
4158
+ // as this attempt's verdicts so all existing disposition and fingerprint accounting
4159
+ // sees the replay, without buying another command or reviewer invocation.
4160
+ const taskEvents = journal.read().filter((e) => e.taskId === t.id);
4161
+ const previousRound = taskEvents.map((e) => e.event === "phase-start" && e.data.phase === "gates").lastIndexOf(true);
4162
+ const previousRows = retryMode === "repair"
4163
+ ? taskEvents.slice(previousRound + 1).filter((e) => e.event === "gate-result"
4164
+ && e.data.attempt === attempt - 1)
4165
+ : [];
4139
4166
  journal.phaseStart(t.id, "gates");
4140
- ({ results, commits } = await withSuiteWindow(t.id, t.gates.includes("test") && commands.test !== undefined, () => runReviewRecovery(t, {
4141
- carriedFindings: outstandingFindings,
4142
- worktree: wt, baseRef: taskBase, result, author: assignment,
4143
- commands, baseline, channels: pools.review, judgeChannels: pools.judge, adapters, cfg, artifactDir: journal.dir,
4144
- collateral: collateral.get(t.id) ?? [],
4145
- pipeline: "v185", selectTests: !testGateFailed,
4146
- via: cfg.visibility.llm === "pane"
4147
- ? {
4148
- driver: trackedDriver,
4149
- // D-07: judge/review panes self-clean when their verdict is read (keepLlm) — only "forever" keeps them.
4150
- keep: keepLlm,
4151
- onSlot: keepLlm ? (s) => keptSlots.push(s) : undefined,
4152
- // T2 ownership contract: canonical names (tickmarkr:<role>:<task>:0:<runId>) so reconcile
4153
- // owns judge/review panes; run-gates' -r1 retry suffix becomes attempt 1 in llm.ts.
4154
- // Same-name reuse across worker attempts is safe: panes self-clean when read (keepLlm),
4155
- // and herdr's DEFECT-01 reclaim covers a kept holdover under keepPanes:forever.
4156
- nameFor: (role) => formatOwnedName({ role, taskId: t.id, attempt: 0, runId }),
4157
- // role-tab label (SUP-01): role-first + task id, unique per concurrent instance within a run.
4158
- // Duplicate labels from a resumed run or operator-made tabs are accepted (per-process state).
4159
- labelFor: (role) => `${role.toUpperCase()} ${t.id}`,
4160
- }
4161
- : undefined,
4162
- excludeReviewers: badReviewers,
4163
- reviewHistory, demotedReviewers,
4164
- onGate,
4165
- })));
4167
+ const replay = previousRows.length > 0
4168
+ && previousRows.every((e) => e.data.commit === gateSubject.commit)
4169
+ && previousRows.some((e) => e.data.pass === false && e.data.skipped !== true);
4170
+ if (replay) {
4171
+ gateSubject.replayedFromAttempt = attempt - 1;
4172
+ results = previousRows.map(({ data }) => ({
4173
+ gate: String(data.gate), pass: data.pass === true, details: String(data.details),
4174
+ // The verdict is reused; its old timing is not a measurement of this attempt.
4175
+ meta: Object.fromEntries(Object.entries(data).filter(([key]) => !GATE_TELEMETRY_KEYS.includes(key) && key !== "capacity")),
4176
+ }));
4177
+ commits = await commitsAheadOf(taskBase, wt);
4178
+ for (const g of results) {
4179
+ journal.append("gate-replayed", t.id, {
4180
+ attempt, priorAttempt: attempt - 1, gate: g.gate, commit: gateSubject.commit,
4181
+ ...(g.meta?.skipped === true ? { skipped: true } : { pass: g.pass }),
4182
+ details: g.details,
4183
+ });
4184
+ journalGateResult(g);
4185
+ }
4186
+ }
4187
+ else {
4188
+ ({ results, commits } = await withSuiteWindow(t.id, t.gates.includes("test") && commands.test !== undefined, () => runReviewRecovery(t, {
4189
+ carriedFindings: outstandingFindings,
4190
+ worktree: wt, baseRef: taskBase, result, author: assignment,
4191
+ commands, baseline, channels: pools.review, judgeChannels: pools.judge, adapters, cfg, artifactDir: journal.dir,
4192
+ collateral: collateral.get(t.id) ?? [],
4193
+ pipeline: "v185", selectTests: !testGateFailed,
4194
+ via: cfg.visibility.llm === "pane"
4195
+ ? {
4196
+ driver: trackedDriver,
4197
+ // D-07: judge/review panes self-clean when their verdict is read (keepLlm) — only "forever" keeps them.
4198
+ keep: keepLlm,
4199
+ onSlot: keepLlm ? (s) => keptSlots.push(s) : undefined,
4200
+ // T2 ownership contract: canonical names (tickmarkr:<role>:<task>:0:<runId>) so reconcile
4201
+ // owns judge/review panes; run-gates' -r1 retry suffix becomes attempt 1 in llm.ts.
4202
+ // Same-name reuse across worker attempts is safe: panes self-clean when read (keepLlm),
4203
+ // and herdr's DEFECT-01 reclaim covers a kept holdover under keepPanes:forever.
4204
+ nameFor: (role) => formatOwnedName({ role, taskId: t.id, attempt: 0, runId }),
4205
+ // role-tab label (SUP-01): role-first + task id, unique per concurrent instance within a run.
4206
+ // Duplicate labels from a resumed run or operator-made tabs are accepted (per-process state).
4207
+ labelFor: (role) => `${role.toUpperCase()} ${t.id}`,
4208
+ }
4209
+ : undefined,
4210
+ excludeReviewers: badReviewers,
4211
+ reviewHistory, demotedReviewers,
4212
+ onGate,
4213
+ })));
4214
+ }
4166
4215
  results.forEach(classifySignalOnlyTest);
4167
4216
  graph = addEvidence(graph, t.id, { commits, gateResults: results, artifacts: [promptFile] });
4168
4217
  saveGraph(repoRoot, graph);
@@ -4377,6 +4426,19 @@ export async function runDaemon(repoRoot, opts = {}) {
4377
4426
  };
4378
4427
  taskLoopStarted = true;
4379
4428
  const inflight = new Map();
4429
+ // A settled worker can release a dependency or an approval in the same poll tick.
4430
+ // Audit the current graph at every empty-flight boundary, never a prior ready snapshot.
4431
+ const holdEndCondition = () => {
4432
+ sweepLiveApprovals();
4433
+ const freeSlots = Math.max(0, concurrency - inflight.size);
4434
+ const dispatchable = readyTasks(graph).filter((t) => !inflight.has(t.id));
4435
+ if (freeSlots === 0 || dispatchable.length === 0)
4436
+ return false;
4437
+ for (const task of dispatchable) {
4438
+ journal.append("end-condition-held", task.id, { deps: task.deps, freeSlots });
4439
+ }
4440
+ return true;
4441
+ };
4380
4442
  closeLoop: while (true) {
4381
4443
  let approvalDeadline;
4382
4444
  while (true) {
@@ -4412,7 +4474,7 @@ export async function runDaemon(repoRoot, opts = {}) {
4412
4474
  .filter((t) => t.status === "pending").every(behindPark);
4413
4475
  if (onlyParks) {
4414
4476
  if (approvalDeadline === undefined) {
4415
- const windowMs = opts.approvalWindowMs ?? APPROVAL_WINDOW_MS;
4477
+ const windowMs = opts.approvalWindowMs ?? approvalWindowMs;
4416
4478
  approvalDeadline = Date.now() + windowMs;
4417
4479
  journal.append("approval-window-start", undefined, { windowMs, parked: [...parked] });
4418
4480
  // The narrator may itself append a decision at this boundary.
@@ -4428,6 +4490,10 @@ export async function runDaemon(repoRoot, opts = {}) {
4428
4490
  }
4429
4491
  journal.append("approval-window-expired", undefined, { parked: [...parked] });
4430
4492
  }
4493
+ if (holdEndCondition()) {
4494
+ approvalDeadline = undefined;
4495
+ continue;
4496
+ }
4431
4497
  break;
4432
4498
  }
4433
4499
  approvalDeadline = undefined;
@@ -4438,6 +4504,8 @@ export async function runDaemon(repoRoot, opts = {}) {
4438
4504
  waiters.push(new Promise((wake) => setTimeout(wake, APPROVAL_POLL_MS)));
4439
4505
  }
4440
4506
  await Promise.race(waiters); // aborted rejects on termination — unwinds the run
4507
+ if (inflight.size === 0)
4508
+ holdEndCondition();
4441
4509
  }
4442
4510
  // D-07: the sweep now closes only what's LEFT in keptSlots — done-closed worker slots were removed
4443
4511
  // (no double-close) and self-cleaned LLM/consult panes were never added under keepLlm:false. This
@@ -4515,12 +4583,11 @@ export async function runDaemon(repoRoot, opts = {}) {
4515
4583
  const approvalSerialization = await acquireApprovalSerialization(repoRoot, runId);
4516
4584
  releaseApprovalSerialization = approvalSerialization.release;
4517
4585
  // Close the last poll-to-run-end race while holding the same serializer as approve.
4518
- sweepLiveApprovals();
4519
- if (readyTasks(graph).length) {
4586
+ if (holdEndCondition()) {
4520
4587
  releaseApprovalSerialization();
4521
4588
  releaseApprovalSerialization = undefined;
4522
- if (summary.tipVerify)
4523
- journal.append("tip-verify-cancelled", undefined, { reason: "approval", lastMergedTask });
4589
+ // Verification has settled: retain its verdict. The cache key will require a
4590
+ // new battery if dispatch actually moves the tip or changes its commands.
4524
4591
  continue closeLoop;
4525
4592
  }
4526
4593
  const outstanding = outstandingApprovals(journal.read());
@@ -949,18 +949,27 @@ export function gateResultJournalData(gate, pass, details, meta = {}) {
949
949
  const signalBasis = deriveSignalBasis(gate, pass, details, meta);
950
950
  return { gate, pass, details, ...meta, signalBasis, signalQuality: signalQualityFromBasis(signalBasis) };
951
951
  }
952
- // T3 (Sol #2 / Fable F2): one canonical engagement identity, shared by status AND resume. The run-start
953
- // event records graphDefinitionHash (over compiled task definitions only — see graph.graphDefinitionHash);
954
- // this is the single field both consumers read, and the single comparator below is the single place the
955
- // journal↔graph join is decided. unbound (no recorded definition hash, e.g. a pre-v1.44 journal) and
952
+ // T3 (Sol #2 / Fable F2) + OBS-978: one canonical engagement identity, shared by status, plan, the operator
953
+ // fold AND resume. The run-start event records graphDefinitionHash (over compiled task definitions only — see
954
+ // graph.graphDefinitionHash); each audited graph-rehash row (resume --graph-changed) then moves the identity to
955
+ // its `to`, so the recorded hash is the last audited rehash, else run-start. A row is audited when its `from`
956
+ // names the identity it replaced, or the run-start one (all pre-OBS-978 daemons wrote); a row auditing neither
957
+ // binds nothing — the journal is unbound until a release from null. unbound (also a pre-v1.44 journal) and
956
958
  // mismatch are both not-comparable — status renders the notice either way; resume refuses either way and
957
959
  // distinguishes the reason only for its message and the --graph-changed release event.
958
960
  export function recordedGraphDefinitionHash(events) {
961
+ const start = events.find((e) => e.event === "run-start");
962
+ if (!start)
963
+ return undefined;
964
+ const origin = typeof start.data.graphDefinitionHash === "string" ? start.data.graphDefinitionHash : null;
965
+ let recorded = origin;
959
966
  for (const e of events) {
960
- if (e.event === "run-start" && typeof e.data.graphDefinitionHash === "string")
961
- return e.data.graphDefinitionHash;
967
+ if (e.event !== "graph-rehash")
968
+ continue;
969
+ const audited = e.data.from === recorded || e.data.from === origin;
970
+ recorded = audited && typeof e.data.to === "string" ? e.data.to : null;
962
971
  }
963
- return undefined;
972
+ return recorded ?? undefined;
964
973
  }
965
974
  // THE shared comparator (criterion: status and resume decide through one comparator). status reads
966
975
  // .comparable; resume reads .comparable plus .reason/.recorded for its refusal message and the release.
@@ -57,7 +57,7 @@ export declare class OperatorStateFold {
57
57
  private tasks;
58
58
  private start?;
59
59
  private startEvent?;
60
- private latestGraphRehash?;
60
+ private rehashes;
61
61
  private end?;
62
62
  private active;
63
63
  private approved;
@@ -10,7 +10,7 @@ export class OperatorStateFold {
10
10
  tasks = new Map();
11
11
  start;
12
12
  startEvent;
13
- latestGraphRehash;
13
+ rehashes = [];
14
14
  end;
15
15
  active = false;
16
16
  approved = false;
@@ -34,7 +34,7 @@ export class OperatorStateFold {
34
34
  this.tipFailed = false;
35
35
  }
36
36
  if (e.event === "graph-rehash")
37
- this.latestGraphRehash = { ...e, data: { from: e.data.from, to: e.data.to } };
37
+ this.rehashes = [...this.rehashes, { ...e, data: { from: e.data.from, to: e.data.to } }];
38
38
  if (e.event === "tip-verify-failed" || (e.event === "tip-verify" && e.data.pass === false))
39
39
  this.tipFailed = true;
40
40
  if (e.event === "run-end") {
@@ -146,13 +146,9 @@ export class OperatorStateFold {
146
146
  comparableTo(hash) {
147
147
  if (!hash)
148
148
  return false;
149
- const events = [this.startEvent, this.latestGraphRehash].filter((e) => e !== undefined);
150
- const baseline = engagementComparable(events, hash);
151
- if (this.latestGraphRehash) {
152
- const from = baseline.comparable ? baseline.recorded : baseline.reason === "mismatch" ? baseline.recorded : null;
153
- return this.latestGraphRehash.data.to === hash && this.latestGraphRehash.data.from === from;
154
- }
155
- return baseline.comparable;
149
+ // The shared comparator audits the rehash chain; the fold keeps only the rows it reads.
150
+ const events = [this.startEvent, ...this.rehashes].filter((e) => e !== undefined);
151
+ return engagementComparable(events, hash).comparable;
156
152
  }
157
153
  }
158
154
  /** C1/C6 share this pure reader; callers supply the same observation and journal snapshot. */
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "tickmarkr",
3
- "version": "2.5.2",
3
+ "version": "2.5.3",
4
4
  "description": "Spec in, verified work out.",
5
5
  "type": "module",
6
6
  "license": "MIT",
@@ -92,7 +92,7 @@ After sending, **confirm delivery** by reading the target pane and verifying the
92
92
  1. **Prepare** — confirm the target list. Run the [binary preflight](#binary-preflight-before-compile-or-run). Check `git status`, confirm no tickmarkr run is active, and work from a non-main branch.
93
93
  2. **Compile** — run `tickmarkr compile <spec-or-directory>`. Fix source-spec defects instead of editing the generated graph.
94
94
  3. **Plan** — run `tickmarkr plan`. Review routes, capability-floor warnings, and human gates before execution.
95
- 4. **Run** — run `tickmarkr run`. Watch the run journal for its terminal events rather than polling agents, using the shipped watcher — `.claude/skills/tickmarkr-overseer/scripts/watch-journal.sh <state-dir>/runs 20 28800` — which takes a line baseline at arm time, then wakes ONCE on `run-end`, `task-human`, `task-failed` or `consult-verdict` and grades the run-end summary against every green clause for you. Re-arm after every wake. ⛔ Never `tail -F | grep -m1` (run-end is the journal's last line, so tail never notices the broken pipe and the watcher hangs forever) and never a pane-level done wait (it fires on every agent turn end, not mission end). ⚠ A bare whole-file `grep -q '"event":"run-end"'` is the trap the watcher exists to avoid: on a resume it matches the PREVIOUS run's run-end and returns instantly, so a re-armed watcher reads as coverage that does not exist. Resolve blocked interactions in the relevant agent session.
95
+ 4. **Run** — run `tickmarkr run`. A watch ending the seat's turn is no watch: keep a **blocking journal consumer** alive for the run's terminal events — the shipped watcher below, or a foreground `until grep` on the run's terminal events — and ensure it is re-armed at most every twenty minutes. Never rely on a `Monitor`-only wake. Watch the run journal rather than polling agents, using the shipped watcher — `.claude/skills/tickmarkr-overseer/scripts/watch-journal.sh <state-dir>/runs 20 28800` — which takes a line baseline at arm time, then wakes ONCE on `run-end`, `task-human`, `task-failed` or `consult-verdict` and grades the run-end summary against every green clause for you. Re-arm after every wake. ⛔ Never `tail -F | grep -m1` (run-end is the journal's last line, so tail never notices the broken pipe and the watcher hangs forever) and never a pane-level done wait (it fires on every agent turn end, not mission end). ⚠ A bare whole-file `grep -q '"event":"run-end"'` is the trap the watcher exists to avoid: on a resume it matches the PREVIOUS run's run-end and returns instantly, so a re-armed watcher reads as coverage that does not exist. Resolve blocked interactions in the relevant agent session.
96
96
  5. **Verify and consolidate** — continue only after a green run. A run is green when the run-end event exists in the journal, the tip verify is not "failed", and the summary's `failed`, `human`, `blocked` and `pending` buckets are all empty — a run with a parked task is partial, not green. Tickmarkr consolidates accepted work on `tickmarkr/<runId>` and never signs off to the main branch. A human controls any later release merge.
97
97
  6. **Record** — `tickmarkr report <runId> --md` prints Markdown to stdout; redirect explicitly beside the spec (for example `tickmarkr report <runId> --md > feature.record.md`) and commit the execution record when the repository tracks those records.
98
98
  7. **Continue** — move to the next requested target. If a target fails or is parked, stop with the journal evidence rather than silently skipping it.
@@ -88,7 +88,7 @@ When spawning consultants (agents gathering synthesis input for decisions like S
88
88
  1. **Prepare** — start from the requested spec. Run the [binary preflight](#binary-preflight-before-compile-or-run). Check `git status`, confirm no tickmarkr run is active, and work from a non-main branch.
89
89
  2. **Compile** — run `tickmarkr compile <spec>`. Correct compilation errors in the spec, never in the generated graph.
90
90
  3. **Plan** — run `tickmarkr plan`. Review the routing table, capability-floor warnings, and every human gate, including work that each gate blocks.
91
- 4. **Run** — run `tickmarkr run`. Watch the run journal for its terminal events rather than polling agents, using the shipped watcher — `.claude/skills/tickmarkr-overseer/scripts/watch-journal.sh <state-dir>/runs 20 28800` — which takes a line baseline at arm time, then wakes ONCE on `run-end`, `task-human`, `task-failed` or `consult-verdict` and grades the run-end summary against every green clause for you. Re-arm after every wake. ⛔ Never `tail -F | grep -m1` (run-end is the journal's last line, so tail never notices the broken pipe and the watcher hangs forever) and never a pane-level done wait (it fires on every agent turn end, not mission end). ⚠ A bare whole-file `grep -q '"event":"run-end"'` is the trap the watcher exists to avoid: on a resume it matches the PREVIOUS run's run-end and returns instantly, so a re-armed watcher reads as coverage that does not exist. Resolve blocked interactions in the agent session; do not turn them into proxy questions.
91
+ 4. **Run** — run `tickmarkr run`. A watch ending the seat's turn is no watch: keep a **blocking journal consumer** alive for the run's terminal events — the shipped watcher below, or a foreground `until grep` on the run's terminal events — and ensure it is re-armed at most every twenty minutes. Never rely on a `Monitor`-only wake. Watch the run journal rather than polling agents, using the shipped watcher — `.claude/skills/tickmarkr-overseer/scripts/watch-journal.sh <state-dir>/runs 20 28800` — which takes a line baseline at arm time, then wakes ONCE on `run-end`, `task-human`, `task-failed` or `consult-verdict` and grades the run-end summary against every green clause for you. Re-arm after every wake. ⛔ Never `tail -F | grep -m1` (run-end is the journal's last line, so tail never notices the broken pipe and the watcher hangs forever) and never a pane-level done wait (it fires on every agent turn end, not mission end). ⚠ A bare whole-file `grep -q '"event":"run-end"'` is the trap the watcher exists to avoid: on a resume it matches the PREVIOUS run's run-end and returns instantly, so a re-armed watcher reads as coverage that does not exist. Resolve blocked interactions in the agent session; do not turn them into proxy questions.
92
92
  5. **Verify and consolidate** — accept only a green run. A run is green when the run-end event exists in the journal, the tip verify is not "failed", and the summary's `failed`, `human`, `blocked` and `pending` buckets are all empty — a run with a parked task is partial, not green. Tickmarkr consolidates accepted task work on `tickmarkr/<runId>`; it never signs off to the main branch. A human may later merge that integration branch through the repository's normal release process.
93
93
  6. **Record** — `tickmarkr report <runId> --md` prints Markdown to stdout. Redirect it explicitly beside the source spec (for example `tickmarkr report <runId> --md > feature.record.md`) and commit the execution record when the repository tracks those records. Then [stand down](#stand-down-mission-end-and-retirement).
94
94
 
@@ -149,6 +149,14 @@ through brief lineage. **An executor choice nobody made is still an executor cho
149
149
  guidance belongs in the memory file or the shipped docs.
150
150
  4. Arm the watcher and your own supervision beat (Supervision). Report the hierarchy map (pane ids + names) to the user.
151
151
 
152
+ ### Seat-spawn and Leg-2 recipes
153
+
154
+ Every mission to a Claude or Grok seat is delivered only with `herdr pane run <pane> "<message>"` and
155
+ verified by reading the pane back; never use `agent prompt` for mission delivery. Launch a Grok seat with
156
+ `herdr agent start <seat> --kind grok --pane <pane> -- -m grok-4.6`. For Leg-2, a Codex reviewer under
157
+ `workspace-write` must be briefed with an in-worktree verdict path such as
158
+ `<repo>/.tickmarkr/overseer/verdicts/<task>.md`, and its verdict must be written there before it is read.
159
+
152
160
  ## Supervising tickmarkr as the executor — WHO DOES WHAT
153
161
 
154
162
  When the mission runs `/tickmarkr-auto` (tickmarkr dispatches the workers), supervision changes shape —
@@ -204,6 +212,9 @@ journal tail to decide what happens next, or sweeping orphans — you have taken
204
212
  - **The journal is the source of truth**, not panes. Watchers go on `run-end` / `task-human` /
205
213
  `task-failed` / `consult-verdict`; never sleep-poll inside an agent turn. **Never key a watcher on an
206
214
  agent's `done`** — that is turn end and fires the moment a seat finishes acknowledging you.
215
+ A watch ending the seat's turn is no watch: keep a **blocking journal consumer** alive for those terminal
216
+ events — the shipped watcher below, or a foreground `until grep` on the run's terminal events — and
217
+ ensure it is re-armed at most every twenty minutes. Never rely on a `Monitor`-only wake.
207
218
  **All four are covered by one shipped instrument** — `scripts/watch-journal.sh <runs-dir> [poll] [cap]
208
219
  [events-csv]` — which arms on a line baseline, wakes once, and grades a `run-end` against every green
209
220
  clause. `scripts/watch-parks.sh` stays the park-specific wake for THIS seat (it counts parks and speaks
@@ -476,6 +487,52 @@ number — an unmeasured budget is not a small budget.
476
487
  .claude/skills/tickmarkr-overseer/scripts/watch-context.sh overseer <overseer-agent-or-pane> 50 50 <handoff-file>
477
488
  ```
478
489
 
490
+ ### A GO has a deadline — arm `watch-launch.sh` in the same act as the GO
491
+
492
+ A GO that produces no run is a silent failure until someone notices; on 2026-09-11 an orchestrator's codex
493
+ sandbox was rooted at the main repo, the spec worktree was outside its writable roots, it stopped at the
494
+ denial without reporting, the overseer's 10-minute wake expired un-re-armed, and three hours passed.
495
+ Two rules close that hole:
496
+
497
+ - **Every orchestrator seat is sandbox-rooted at the worktree it will run in** (`cd <worktree>` before
498
+ `herdr agent start … --sandbox workspace-write`), and its brief says: *a denied path or refused command
499
+ is reported to the overseer pane within 60 s — never a silent stop.*
500
+ - **The overseer arms the launch watcher in the SAME act as the GO**, with the lock path the run will
501
+ create, and treats `LAUNCH_OVERDUE` as a first-class event (read the orchestrator pane, fix the seat,
502
+ re-issue the GO):
503
+
504
+ ```bash
505
+ .claude/skills/tickmarkr-overseer/scripts/watch-launch.sh <worktree>/.tickmarkr/graph.lock 900 <overseer-pane> &
506
+ ```
507
+
508
+ It prints `LAUNCH_OK` with the lock's contents when the run starts (exit 0) and, past the deadline, delivers
509
+ `LAUNCH OVERDUE …` to the overseer pane AND as an OS notification (exit 3). Any wake you arm yourself with
510
+ a cap (a background `until` loop) must be RE-ARMED on every expiry; an expired wake is not a watch.
511
+
512
+
513
+ ### A GO has a deadline — arm `watch-launch.sh` in the same act as the GO
514
+
515
+ A GO that produces no run is a silent failure until someone notices; on 2026-09-11 an orchestrator's codex
516
+ sandbox was rooted at the main repo, the spec worktree was outside its writable roots, it stopped at the
517
+ denial without reporting, the overseer's 10-minute wake expired un-re-armed, and three hours passed.
518
+ Two rules close that hole:
519
+
520
+ - **Every orchestrator seat is sandbox-rooted at the worktree it will run in** (`cd <worktree>` before
521
+ `herdr agent start … --sandbox workspace-write`), and its brief says: *a denied path or refused command
522
+ is reported to the overseer pane within 60 s — never a silent stop.*
523
+ - **The overseer arms the launch watcher in the SAME act as the GO**, with the lock path the run will
524
+ create, and treats `LAUNCH_OVERDUE` as a first-class event (read the orchestrator pane, fix the seat,
525
+ re-issue the GO):
526
+
527
+ ```bash
528
+ .claude/skills/tickmarkr-overseer/scripts/watch-launch.sh <worktree>/.tickmarkr/graph.lock 900 <overseer-pane> &
529
+ ```
530
+
531
+ It prints `LAUNCH_OK` with the lock's contents when the run starts (exit 0) and, past the deadline, delivers
532
+ `LAUNCH OVERDUE …` to the overseer pane AND as an OS notification (exit 3). Any wake you arm yourself with
533
+ a cap (a background `until` loop) must be RE-ARMED on every expiry; an expired wake is not a watch.
534
+
535
+
479
536
  The first argument chooses the closed per-seat tier (`orchestrator-context` or `overseer-context`),
480
537
  and every beat names the second argument as that tier's seat. The watcher beats only after reading a
481
538
  rendered percentage, keeps beating on the supervision cadence even when its requested poll is slower,
@@ -580,9 +637,9 @@ they are left implicit:
580
637
  Send only when the seat is idle and the ANSI prompt line is empty or dim-only (the Esc/SGR discriminator
581
638
  separates an autosuggest ghost from typed input), then read back activity or an ACK; presence is not
582
639
  delivery. If a stale draft must be replaced, supersede it explicitly with
583
- `agent prompt " <-- disregard … ACTUAL: …"` instead of stacking another instruction behind it.
640
+ `herdr pane run <pane> "<-- disregard … ACTUAL: …"` instead of stacking another instruction behind it.
584
641
  - **A MESSAGE TO A WORKING SEAT IS A QUEUED MESSAGE, AND THE QUEUE DRAINS ONLY AT TURN BOUNDARIES.**
585
- Delivery is not arrival: `agent prompt` to a `working` claude seat lands in its queue (`Press up to
642
+ Delivery is not arrival: a message sent to a `working` claude seat lands in its queue (`Press up to
586
643
  edit queued messages` on the seat's prompt line is the tell) and is READ only when the current turn
587
644
  ends — and with in-process teammates a turn runs 20–40 minutes, so steering latency equals subagent
588
645
  runtime. Measured 2026-08-17/18 (P98 leg 1): a FREEZE HOLD and a checker-release directive stacked
@@ -622,7 +679,7 @@ they are left implicit:
622
679
  - **AGENT NAMES ARE GLOBAL ACROSS WORKSPACES — verify a seat you spawned by PANE ID, never by name.**
623
680
  Names must be unique among live agents *everywhere*, not within your workspace, so another workspace can
624
681
  already hold `opus`, `sol`, `reviewer` or `orch`. When it does, your `agent start` **fails**, your pane
625
- is left a bare shell, and `agent list` / `agent read` / `agent prompt` for that name then resolve to the
682
+ is left a bare shell, and `agent list` / `agent read` for that name then resolve to the
626
683
  **stranger's seat**. Measured 2026-08-06 (OBS-392): a spawn of `fable` collided with a live seat in
627
684
  another workspace; `agent list` reported `fable -> blocked` and it was read as *this* seat coming up
628
685
  blocked. It was an operator research session sitting on a *"Resume full session?"* prompt. One more
@@ -644,7 +701,7 @@ they are left implicit:
644
701
  Re-arm name-keyed watchers in the same act as the rename; file-keyed artifact watchers are
645
702
  unaffected (one more reason to prefer them).
646
703
  - Stale typed input is unclearable via CLI — supersede it:
647
- `pane run "<-- disregard everything before this arrow (stale draft). ACTUAL: <message>"`.
704
+ `herdr pane run <pane> "<-- disregard everything before this arrow (stale draft). ACTUAL: <message>"`.
648
705
  **But DISCRIMINATE before you supersede or file it: text on an idle seat's prompt line has FOUR
649
706
  authors** — the seat's own draft, an operator, another agent's `agent send` (writes WITHOUT Enter),
650
707
  and claude-code's AUTOSUGGEST, which renders context-plausible ghost text BYTE-IDENTICAL to a typed
@@ -0,0 +1,38 @@
1
+ #!/usr/bin/env bash
2
+ # watch-launch.sh — a GO that produced no run is a silent failure until someone notices. This watcher
3
+ # notices. Arm it in the SAME act as the GO (orchestrator briefed to compile → plan → run) and it waits
4
+ # for the run's lock; when the lock has not appeared by the deadline it delivers LAUNCH OVERDUE to the
5
+ # overseer's pane AND as an OS notification, so the wake reaches a seat instead of a log nobody reads.
6
+ #
7
+ # Why it exists (2026-09-11): an orchestrator's codex sandbox was rooted at the main repo, the spec
8
+ # worktree was outside its writable roots, it stopped at the denial without reporting, and the overseer's
9
+ # own 10-minute wake expired un-re-armed. Three hours passed before anyone looked. A launch has a
10
+ # deadline; silence past it is the event.
11
+ #
12
+ # usage: watch-launch.sh <lock-path> <deadline-s> <overseer-pane> [poll-s]
13
+ # <lock-path> the run's .tickmarkr/graph.lock in the worktree the run will be launched in
14
+ # <deadline-s> seconds from now by which the lock must exist (a compile+plan+launch takes minutes,
15
+ # never hours; 900 is a generous default for a 7-task spec)
16
+ # <overseer-pane> the pane that must hear about it (herdr pane id), e.g. wZ:p18S
17
+ # [poll-s] poll interval, default 15
18
+ # exit 0 LAUNCH_OK (lock seen; prints its contents) · exit 3 LAUNCH_OVERDUE (delivered) · exit 64 usage
19
+ set -u
20
+ LOCK="${1:-}"; DEADLINE="${2:-}"; PANE="${3:-}"; POLL="${4:-15}"
21
+ [ -n "$LOCK" ] && [ -n "$DEADLINE" ] && [ -n "$PANE" ] || { echo "usage: watch-launch.sh <lock-path> <deadline-s> <overseer-pane> [poll-s]" >&2; exit 64; }
22
+ start=$(date +%s)
23
+ while :; do
24
+ if [ -f "$LOCK" ]; then
25
+ printf 'LAUNCH_OK %s %s\n' "$(date -u +%H:%M:%SZ)" "$(cat "$LOCK" 2>/dev/null | tr -d '\n')"
26
+ exit 0
27
+ fi
28
+ now=$(date +%s)
29
+ if [ $((now - start)) -ge "$DEADLINE" ]; then
30
+ msg="LAUNCH OVERDUE $(date -u +%H:%M:%SZ): no lock at $LOCK after ${DEADLINE}s — read the orchestrator pane NOW (sandbox denial? preflight refusal? unsubmitted GO?)"
31
+ echo "LAUNCH_OVERDUE $msg"
32
+ # Both deliveries, always: a pane the overseer reads AND a notification the operator sees.
33
+ herdr pane run "$PANE" "$msg" >/dev/null 2>&1 || echo " (pane delivery failed — the notification is the only path)"
34
+ herdr notification show "$msg" >/dev/null 2>&1 || true
35
+ exit 3
36
+ fi
37
+ sleep "$POLL"
38
+ done