tickmarkr 2.6.2 → 2.6.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (63) hide show
  1. package/README.md +2 -0
  2. package/dist/cli/commands/approve.d.ts +6 -3
  3. package/dist/cli/commands/approve.js +16 -4
  4. package/dist/cli/commands/doctor.d.ts +6 -2
  5. package/dist/cli/commands/fleet.js +46 -8
  6. package/dist/cli/commands/plan.js +13 -8
  7. package/dist/cli/commands/report.d.ts +2 -1
  8. package/dist/cli/commands/report.js +56 -6
  9. package/dist/cli/commands/status.js +19 -1
  10. package/dist/compile/native.js +7 -0
  11. package/dist/config/config.d.ts +17 -4
  12. package/dist/config/config.js +43 -6
  13. package/dist/config/fleet-overlay.d.ts +13 -3
  14. package/dist/config/fleet-overlay.js +12 -8
  15. package/dist/drivers/herdr.d.ts +12 -0
  16. package/dist/drivers/herdr.js +51 -0
  17. package/dist/drivers/orca.d.ts +9 -1
  18. package/dist/drivers/orca.js +29 -7
  19. package/dist/drivers/types.d.ts +2 -0
  20. package/dist/drivers/types.js +2 -2
  21. package/dist/gates/acceptance.d.ts +7 -0
  22. package/dist/gates/acceptance.js +27 -5
  23. package/dist/gates/baseline.d.ts +20 -1
  24. package/dist/gates/baseline.js +100 -20
  25. package/dist/gates/cache.d.ts +8 -0
  26. package/dist/gates/cache.js +12 -2
  27. package/dist/gates/llm.d.ts +6 -0
  28. package/dist/gates/llm.js +27 -8
  29. package/dist/gates/review.d.ts +6 -1
  30. package/dist/gates/review.js +122 -32
  31. package/dist/gates/run-gates.d.ts +54 -3
  32. package/dist/gates/run-gates.js +331 -45
  33. package/dist/gates/test-manifest.d.ts +42 -0
  34. package/dist/gates/test-manifest.js +69 -10
  35. package/dist/route/router.d.ts +12 -1
  36. package/dist/route/router.js +26 -9
  37. package/dist/run/consult.d.ts +3 -1
  38. package/dist/run/consult.js +4 -2
  39. package/dist/run/daemon.d.ts +2 -1
  40. package/dist/run/daemon.js +342 -77
  41. package/dist/run/interactive-seed.d.ts +4 -0
  42. package/dist/run/interactive-seed.js +35 -9
  43. package/dist/run/journal.d.ts +26 -0
  44. package/dist/run/journal.js +143 -15
  45. package/dist/run/lease.d.ts +13 -0
  46. package/dist/run/lease.js +45 -0
  47. package/dist/run/protocol.d.ts +15 -0
  48. package/dist/run/protocol.js +11 -1
  49. package/dist/run/receipt-resolver.d.ts +22 -0
  50. package/dist/run/receipt-resolver.js +40 -1
  51. package/dist/run/repair-selection.d.ts +11 -1
  52. package/dist/run/repair-selection.js +17 -9
  53. package/dist/run/wall-budget.d.ts +48 -0
  54. package/dist/run/wall-budget.js +280 -0
  55. package/dist/tui/cockpit/run-cockpit.js +2 -2
  56. package/dist/tui/cockpit/run-view.d.ts +2 -1
  57. package/dist/tui/cockpit/run-view.js +13 -9
  58. package/dist/tui/cockpit/setup-cockpit.d.ts +2 -0
  59. package/dist/tui/cockpit/setup-cockpit.js +4 -0
  60. package/package.json +2 -1
  61. package/schema/config.schema.json +8 -1
  62. package/skills/tickmarkr-loop/SKILL.md +7 -1
  63. package/skills/tickmarkr-overseer/scripts/classify-vitest-log.sh +4 -1
@@ -1,25 +1,81 @@
1
1
  import { createHash, randomUUID } from "node:crypto";
2
2
  import { execFileSync } from "node:child_process";
3
- import { existsSync, readFileSync, mkdirSync, readdirSync, statSync, unlinkSync, writeFileSync, rmdirSync } from "node:fs";
3
+ import { EVICTION_TOMBSTONE_LIMIT, EVICTION_TOMBSTONES_FILE, parseEvictionTombstones } from "../run/receipt-resolver.js";
4
+ import { existsSync, readFileSync, mkdirSync, readdirSync, renameSync, statSync, unlinkSync, writeFileSync, rmdirSync } from "node:fs";
4
5
  import { availableParallelism, loadavg, tmpdir } from "node:os";
5
- import { join } from "node:path";
6
+ import { basename, join } from "node:path";
7
+ import { DEFAULT_EVIDENCE_QUOTA_BYTES } from "../config/config.js";
6
8
  import { dependencyLinkRefusal, DEFAULT_SHELL_TIMEOUT_MS, describeCapacity, sameCapacity, sh } from "../run/git.js";
7
9
  import { executionSignal } from "../run/execution-budget.js";
8
10
  import { isVitestTestCommand, manifestFileCount } from "./test-manifest.js";
9
11
  export const EVIDENCE_TAIL_BYTES = 16 * 1024;
10
- export const EVIDENCE_RUN_QUOTA_BYTES = 8 * 1024 * 1024;
11
- export function redactGateOutput(text, env) {
12
- // One pass over the original bytes: replacing an environment value must not break a token
13
- // recognizer, and a short environment value must not rewrite the redaction marker itself.
12
+ /** The default per-run quota; `gates.evidenceQuotaBytes` in config overrides it (OBS-1140). */
13
+ export const EVIDENCE_RUN_QUOTA_BYTES = DEFAULT_EVIDENCE_QUOTA_BYTES;
14
+ /** Benign environment locations substituted by name; every other redaction is material. */
15
+ const BENIGN_ENV_KEYS = ["HOME", "TMPDIR"];
16
+ const REDACTION_CATEGORIES = ["token", "assignment", "secretEnv", "benignEnv"];
17
+ /**
18
+ * OBS-1139: every category scans the ORIGINAL text on its own; overlapping spans then merge into one
19
+ * span counted once under its highest-priority member — token, assignment, secret environment, benign
20
+ * HOME/TMPDIR — so precedence never depends on which match starts first. Counts only: never a value.
21
+ */
22
+ export function classifyGateOutput(text, env) {
23
+ // HOME/TMPDIR substitute by name at any length; "/" alone would rewrite every path separator.
24
+ const benign = BENIGN_ENV_KEYS.flatMap(key => { const value = env[key]; return value && value.length > 1 ? [{ value, label: `$${key}` }] : []; });
14
25
  // Short operational values (0, true, vi, test) occur throughout ordinary diagnostics.
15
- // Only secret-named keys justify redacting short values.
16
- const values = [...new Set(Object.entries(env)
17
- .filter(([key, value]) => !!value && (value.length >= 8 || /KEY|TOKEN|SECRET|PASSWORD|CREDENTIAL|AUTH/i.test(key)))
18
- .map(([, value]) => value))]
19
- .sort((a, b) => b.length - a.length).map(v => v.replace(/[.*+?^${}()|[\]\\]/g, "\\$&"));
26
+ // Only secret-named keys justify redacting short values. A secret-named value that happens to equal
27
+ // HOME/TMPDIR stays a secret candidate: precedence decides the span, never candidate pruning.
28
+ const secrets = [...new Set(Object.entries(env)
29
+ .filter(([key, value]) => !!value && !BENIGN_ENV_KEYS.includes(key)
30
+ && (value.length >= 8 || /KEY|TOKEN|SECRET|PASSWORD|CREDENTIAL|AUTH/i.test(key)))
31
+ .map(([, value]) => value))];
32
+ const literal = (v) => v.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
20
33
  const tokens = String.raw `\b(?:sk-[A-Za-z0-9_-]{12,}|gh[pousr]_[A-Za-z0-9_]{16,}|github_pat_[A-Za-z0-9_]{16,}|AKIA[A-Z0-9]{16}|eyJ[A-Za-z0-9_-]+\.[A-Za-z0-9_-]+\.[A-Za-z0-9_-]+)\b`;
21
34
  const assignments = String.raw `(?:authorization\s*:\s*bearer|(?:api[_-]?key|token|password|secret)\s*[:=])\s*[^\s"']+`;
22
- return text.replace(new RegExp([tokens, assignments, ...values].join("|"), "gi"), "[REDACTED]");
35
+ // Every value scans the original text on its own: an alternation would consume the first match and
36
+ // skip a longer overlapping value's tail, so precedence is resolved on the collected spans instead.
37
+ const scan = (source, rank, label) => [...text.matchAll(new RegExp(source, "gi"))].map(m => ({ start: m.index, end: m.index + m[0].length, rank, label }));
38
+ const spans = [
39
+ ...scan(tokens, 0), ...scan(assignments, 1),
40
+ ...secrets.flatMap(v => scan(literal(v), 2)),
41
+ ...benign.flatMap(b => scan(literal(b.value), 3, b.label)),
42
+ ].sort((a, b) => a.start - b.start || b.end - a.end);
43
+ const merged = [];
44
+ for (const span of spans) {
45
+ const last = merged.at(-1);
46
+ if (!last || span.start >= last.end) {
47
+ merged.push({ ...span });
48
+ continue;
49
+ }
50
+ // C-11 (D-673): a benign value nested in (or identical to) another is one location, named by the containing
51
+ // span, which sorts first. Only a PARTIAL overlap of two different benign values is ambiguous: withhold it.
52
+ if (last.rank === 3 && span.rank === 3 && last.label !== span.label && span.end > last.end)
53
+ last.rank = 2;
54
+ else
55
+ last.rank = Math.min(last.rank, span.rank);
56
+ last.end = Math.max(last.end, span.end);
57
+ }
58
+ const counts = { token: 0, assignment: 0, secretEnv: 0, benignEnv: 0 };
59
+ let out = "", at = 0;
60
+ for (const { start, end, rank, label } of merged) {
61
+ counts[REDACTION_CATEGORIES[rank]]++;
62
+ out += text.slice(at, start) + (rank === 3 ? label : "[REDACTED]");
63
+ at = end;
64
+ }
65
+ return { text: out + text.slice(at), counts };
66
+ }
67
+ export function redactGateOutput(text, env) {
68
+ return classifyGateOutput(text, env).text;
69
+ }
70
+ /** Material means bytes were withheld; a benign location substitution withholds nothing. */
71
+ export const redactionMaterial = (c) => c.token + c.assignment + c.secretEnv > 0;
72
+ function readEvictionTombstones(root) {
73
+ try {
74
+ return parseEvictionTombstones(readFileSync(join(root, EVICTION_TOMBSTONES_FILE)));
75
+ }
76
+ catch {
77
+ return [];
78
+ }
23
79
  }
24
80
  /** Resolve a snapshot reference against actual bytes: eviction never fabricates a live artifact. */
25
81
  export function resolveEvidenceArtifact(root, ref) {
@@ -75,7 +131,12 @@ export function beginGateEvidence(cwd, gate, command, opts = {}, nonce) {
75
131
  exitCode: observed?.exitCode ?? null, signal: observed?.signal ?? null,
76
132
  timedOut: observed ? observed.outcome === "timed-out" : null,
77
133
  };
78
- const clean = [redactGateOutput(stdout, env), redactGateOutput(stderr, env)];
134
+ const classified = [classifyGateOutput(stdout, env), classifyGateOutput(stderr, env)];
135
+ const clean = classified.map(c => c.text);
136
+ const counts = { token: 0, assignment: 0, secretEnv: 0, benignEnv: 0 };
137
+ for (const c of classified)
138
+ for (const k of Object.keys(counts))
139
+ counts[k] += c.counts[k];
79
140
  const refs = clean.map((text, i) => {
80
141
  const bytes = Buffer.from(text);
81
142
  const tail = bytes.subarray(-EVIDENCE_TAIL_BYTES);
@@ -94,19 +155,38 @@ export function beginGateEvidence(cwd, gate, command, opts = {}, nonce) {
94
155
  lock = join(dir, ".write-lock");
95
156
  const quota = opts.quotaBytes !== undefined && Number.isFinite(opts.quotaBytes)
96
157
  ? Math.max(0, Math.floor(opts.quotaBytes)) : EVIDENCE_RUN_QUOTA_BYTES;
158
+ // An empty artifact costs no quota, so it is never evicted: every eviction frees bytes.
97
159
  const files = readdirSync(dir).filter(f => /^[a-zA-Z0-9-]+-(stdout|stderr)\.log$/.test(f)).map(f => ({ path: join(dir, f), stat: statSync(join(dir, f)) }))
98
- .sort((a, b) => a.stat.mtimeMs - b.stat.mtimeMs || a.path.localeCompare(b.path));
160
+ .filter(f => f.stat.size > 0).sort((a, b) => a.stat.mtimeMs - b.stat.mtimeMs || a.path.localeCompare(b.path));
99
161
  let total = files.reduce((sum, f) => sum + f.stat.size, 0);
100
162
  const incoming = refs.reduce((sum, ref) => sum + ref.retainedBytes, 0);
101
- if (incoming > quota)
102
- throw new Error("evidence quota cannot retain this invocation");
103
- while (files.length && total + incoming > quota) {
163
+ // OBS-1140: the quota binds what is already retained too, so a resumed run at zero quota keeps
164
+ // zero bytes. An invocation larger than the whole quota is expired at birth; older artifacts
165
+ // are evicted only as far as the quota itself demands, never for bytes that could not fit.
166
+ const fits = incoming <= quota;
167
+ const evicted = [];
168
+ while (files.length && total + (fits ? incoming : 0) > quota) {
104
169
  const oldest = files.shift();
170
+ const bytes = readFileSync(oldest.path);
105
171
  unlinkSync(oldest.path);
106
172
  total -= oldest.stat.size;
173
+ evicted.push({ path: `gate-evidence/${basename(oldest.path)}`, sha256: createHash("sha256").update(bytes).digest("hex"), retainedBytes: bytes.length });
174
+ }
175
+ if (evicted.length) {
176
+ // Identity-bound (path + hash + length) and bounded: the newest 256 survive a restart.
177
+ const tombstones = join(root, EVICTION_TOMBSTONES_FILE);
178
+ writeFileSync(`${tombstones}.tmp`, JSON.stringify([...readEvictionTombstones(root), ...evicted].slice(-EVICTION_TOMBSTONE_LIMIT)));
179
+ renameSync(`${tombstones}.tmp`, tombstones);
107
180
  }
108
- for (const [i, ref] of refs.entries()) {
109
- (opts.write ?? writeFileSync)(join(root, ref.path), Buffer.from(clean[i]).subarray(-EVIDENCE_TAIL_BYTES));
181
+ if (!fits) {
182
+ availability = "expired";
183
+ for (const ref of refs)
184
+ ref.availability = "expired";
185
+ }
186
+ else {
187
+ for (const [i, ref] of refs.entries()) {
188
+ (opts.write ?? writeFileSync)(join(root, ref.path), Buffer.from(clean[i]).subarray(-EVIDENCE_TAIL_BYTES));
189
+ }
110
190
  }
111
191
  }
112
192
  catch {
@@ -131,7 +211,7 @@ export function beginGateEvidence(cwd, gate, command, opts = {}, nonce) {
131
211
  }
132
212
  return { invocationId, ...(nonce ? { nonce } : {}),
133
213
  subject: { runId: opts.runId ?? "standalone", taskId: opts.taskId ?? null, attempt: opts.attempt ?? null, gate, subjectCommit },
134
- termination, availability, redaction: { material: clean[0] !== stdout || clean[1] !== stderr }, stdout: refs[0], stderr: refs[1] };
214
+ termination, availability, redaction: { material: redactionMaterial(counts), counts }, stdout: refs[0], stderr: refs[1] };
135
215
  },
136
216
  };
137
217
  }
@@ -9,8 +9,15 @@ export declare function lockfileHash(worktree: string): string;
9
9
  export declare function canonicalJson(obj: unknown): string;
10
10
  export declare function baselineIdentity(baseline?: Baseline): string;
11
11
  export declare function getWorktreeTree(worktree: string): Promise<string>;
12
+ /** OBS-635: environment inputs a runner child reads that capacity and lifecycle do not already bind.
13
+ * ponytail: a named list, not the whole env (pids and terminal vars would defeat every reuse); extend
14
+ * it when another variable is shown to change what a runner executes. */
15
+ export declare const RUNNER_ENV_KEYS: readonly ["PATH", "NODE_OPTIONS", "NODE_PATH", "NODE_ENV", "TZ"];
16
+ export declare function runnerInputsHash(env?: NodeJS.ProcessEnv): string;
12
17
  export interface GateEnvironmentInput {
13
18
  nodeRuntime?: string;
19
+ /** the runner inputs hash; defaults to this process's (runnerInputsHash) */
20
+ runner?: string;
14
21
  lockfile?: string;
15
22
  worktree?: string;
16
23
  capacity?: RunCapacity;
@@ -21,6 +28,7 @@ export interface GateEnvironmentInput {
21
28
  }
22
29
  export interface EnvironmentParts {
23
30
  nodeRuntime: string;
31
+ runner: string;
24
32
  lockfile: string;
25
33
  capacity: RunCapacity;
26
34
  selectedSet?: readonly string[];
@@ -77,8 +77,17 @@ export async function getWorktreeTree(worktree) {
77
77
  rmSync(scratch, { recursive: true, force: true });
78
78
  }
79
79
  }
80
+ /** OBS-635: environment inputs a runner child reads that capacity and lifecycle do not already bind.
81
+ * ponytail: a named list, not the whole env (pids and terminal vars would defeat every reuse); extend
82
+ * it when another variable is shown to change what a runner executes. */
83
+ export const RUNNER_ENV_KEYS = ["PATH", "NODE_OPTIONS", "NODE_PATH", "NODE_ENV", "TZ"];
84
+ export function runnerInputsHash(env = process.env) {
85
+ return createHash("sha256").update(canonicalJson(Object.fromEntries(RUNNER_ENV_KEYS.map((k) => [k, env[k] ?? null]))))
86
+ .digest("hex").slice(0, 16);
87
+ }
80
88
  export function environmentFingerprint(env) {
81
89
  const nodeRuntime = env.nodeRuntime ?? process.version;
90
+ const runner = env.runner ?? runnerInputsHash();
82
91
  const lockfile = env.lockfile ?? (env.worktree ? lockfileHash(env.worktree) : "no-lockfile");
83
92
  const cap = env.capacity ?? resolvedCapacity();
84
93
  const capacity = { forkCap: cap.forkCap, cores: cap.cores };
@@ -97,6 +106,7 @@ export function environmentFingerprint(env) {
97
106
  // explicit `false` and an npmrc `false` are the same policy for the child that ran.
98
107
  const payload = canonicalJson({
99
108
  nodeRuntime,
109
+ runner,
100
110
  lockfile,
101
111
  capacity,
102
112
  resolution,
@@ -107,7 +117,7 @@ export function environmentFingerprint(env) {
107
117
  const fingerprint = createHash("sha256").update(payload).digest("hex").slice(0, 16);
108
118
  return {
109
119
  fingerprint,
110
- parts: { nodeRuntime, lockfile, capacity, selectedSet, verification },
120
+ parts: { nodeRuntime, runner, lockfile, capacity, selectedSet, verification },
111
121
  };
112
122
  }
113
123
  export async function computeVerificationIdentity(params) {
@@ -151,7 +161,7 @@ export function verificationIdentityKey(id) {
151
161
  export function formatReusedDetails(originalDetails, id) {
152
162
  const unadorned = originalDetails.replace(/^reused verdict \(identity: [^)]+\):\s*/, "");
153
163
  const envDesc = id.envParts
154
- ? ` [node=${id.envParts.nodeRuntime}, lockfile=${id.envParts.lockfile}, capacity=${describeCapacity(id.envParts.capacity)}${id.envParts.selectedSet ? `, selected=${id.envParts.selectedSet.join(",")}` : ""}, protocol=${id.envParts.verification.protocol}, lifecycle=${id.envParts.verification.lifecycle} (${id.envParts.verification.source})]`
164
+ ? ` [node=${id.envParts.nodeRuntime}, runner=${id.envParts.runner}, lockfile=${id.envParts.lockfile}, capacity=${describeCapacity(id.envParts.capacity)}${id.envParts.selectedSet ? `, selected=${id.envParts.selectedSet.join(",")}` : ""}, protocol=${id.envParts.verification.protocol}, lifecycle=${id.envParts.verification.lifecycle} (${id.envParts.verification.source})]`
155
165
  : "";
156
166
  const gate = id.gate ?? "gate";
157
167
  const prefix = `reused ${id.scope === "tip" ? "tip " : ""}verdict (identity: gate=${gate} tree=${id.tree}${id.worktree ? ` worktree=${id.worktree}` : ""} command=${id.command} baseline=${id.baseline} env=${id.environment}${envDesc})`;
@@ -69,6 +69,12 @@ export interface LlmRunResult {
69
69
  }
70
70
  export declare const PROMPT_GLYPHS: readonly ["➜", "❯", "$", "%", ">>", ">"];
71
71
  export declare function reviewSeatOutput(raw: string, nonce: string, adapterBannerRows?: readonly string[]): string;
72
+ /** OBS-1168(b): a judge/review seat whose pane could not be created or launched. Typed so the gates can
73
+ * contain it as seat recovery (another seat) or an infra park — never as a worker failure. */
74
+ export declare class SeatLaunchError extends Error {
75
+ readonly seat: string;
76
+ constructor(seat: string, cause: unknown);
77
+ }
72
78
  export declare const REVIEW_FIRST_LIVENESS_MS = 30000;
73
79
  export declare const REVIEW_SILENT_BYTE_FLOOR = 64;
74
80
  export declare function runHeadless(adapter: WorkerAdapter, model: string, prompt: string, cwd: string, timeoutMs?: number, effort?: Effort): Promise<string>;
package/dist/gates/llm.js CHANGED
@@ -81,19 +81,22 @@ export function gatePaneName(role, taskId, suffix = "") {
81
81
  // T2 ownership contract: a canonical owned fallback (the daemon's nameFor now emits one) passes
82
82
  // through untouched; run-gates' "-r1" judge-retry suffix becomes attempt+1 so the retry pane's name
83
83
  // stays contract-parseable (tickmarkr:judge:<task>:1:<runId>) instead of a corrupted-runId shape.
84
+ // C-3 (D-669): the hop suffix is -r<N> — the judge-flake retry is -r1 and the adjudicator -r2, so the two
85
+ // never share an owned name (a retained retry pane must not shadow the adjudicator's slot).
84
86
  export function rolePaneNameFromPrompt(prompt, fallback) {
85
- const retry = fallback.endsWith("-r1");
86
- const base = retry ? fallback.slice(0, -3) : fallback;
87
+ const hop = /-r([1-9])$/.exec(fallback);
88
+ const suffix = hop ? hop[0] : "";
89
+ const base = hop ? fallback.slice(0, -suffix.length) : fallback;
87
90
  const owned = parseOwnedName(base);
88
91
  if (owned)
89
- return retry ? formatOwnedName({ ...owned, attempt: owned.attempt + 1 }) : base;
92
+ return hop ? formatOwnedName({ ...owned, attempt: owned.attempt + Number(hop[1]) }) : base;
90
93
  const id = prompt.match(/## Task ([^\n:]+):/)?.[1];
91
94
  if (!id)
92
95
  return fallback;
93
96
  if (prompt.startsWith("TICKMARKR-JUDGE"))
94
- return gatePaneName("judge", id, retry ? "-r1" : "");
97
+ return gatePaneName("judge", id, suffix);
95
98
  if (prompt.startsWith("TICKMARKR-REVIEW"))
96
- return gatePaneName("review", id, retry ? "-r1" : "");
99
+ return gatePaneName("review", id, suffix);
97
100
  return fallback;
98
101
  }
99
102
  const llmOutputCapture = new AsyncLocalStorage();
@@ -299,6 +302,16 @@ export function reviewSeatOutput(raw, nonce, adapterBannerRows = []) {
299
302
  // often the seat's first byte than the harness's exit marker, and the next read completes either.
300
303
  return trailer ? seat.slice(0, trailer.index) : seat;
301
304
  }
305
+ /** OBS-1168(b): a judge/review seat whose pane could not be created or launched. Typed so the gates can
306
+ * contain it as seat recovery (another seat) or an infra park — never as a worker failure. */
307
+ export class SeatLaunchError extends Error {
308
+ seat;
309
+ constructor(seat, cause) {
310
+ super(`seat ${seat} failed to launch: ${cause instanceof Error ? cause.message : String(cause)}`);
311
+ this.seat = seat;
312
+ this.name = "SeatLaunchError";
313
+ }
314
+ }
302
315
  export const REVIEW_FIRST_LIVENESS_MS = 30_000;
303
316
  // OBS-1039: a seat that wrote ten bytes and went quiet escaped the zero-byte beat and sat to the
304
317
  // ceiling. Below this many seat-authored bytes at the first beat the seat is `silent` — demoted and
@@ -341,9 +354,15 @@ async function runViaDriverDetailed(adapter, model, prompt, cwd, via, timeoutMs
341
354
  adapter.headlessCommand(pf, model, effort),
342
355
  gateExitTrailer(nonce),
343
356
  ].join("\n"));
344
- slot = await via.driver.slot(cwd, rolePaneNameFromPrompt(prompt, via.name), via.label ? { label: via.label } : undefined);
345
- via.onSlot?.(slot);
346
- await via.driver.run(slot, paneDispatchCommand(scriptPath));
357
+ try {
358
+ slot = await via.driver.slot(cwd, rolePaneNameFromPrompt(prompt, via.name), via.label ? { label: via.label } : undefined);
359
+ via.onSlot?.(slot);
360
+ await via.driver.run(slot, paneDispatchCommand(scriptPath));
361
+ }
362
+ catch (error) {
363
+ forceClose = true; // a half-launched pane is closed, never kept
364
+ throw new SeatLaunchError(`${adapter.id}:${model}`, error);
365
+ }
347
366
  if (via.driver.sendKey) {
348
367
  try {
349
368
  if (matchesTrustDialog(await via.driver.read(slot, 400), adapter.trustDialog)) {
@@ -51,6 +51,11 @@ export declare function fetchTaskDiff(worktree: string, baseRef: string, files?:
51
51
  export declare function checkDiffCap(gate: string, measured: number, cap: number, prefix?: string): GateResult | null;
52
52
  /** Apply the strict reviewable-logic cap and the finite, larger capture cap independently. */
53
53
  export declare function checkTaskDiffCaps(gate: string, measured: Pick<TaskDiffMeasurement, "logicBytes" | "captureBytes">, logicCap: number, prefix?: string): GateResult | null;
54
+ export declare function isGarbageReview(result: GateResult): result is GateResult & {
55
+ meta: {
56
+ reviewer: string;
57
+ };
58
+ };
54
59
  export declare function isDiffCapPark(result: GateResult): boolean;
55
60
  export declare function diffCapParkReason(results: GateResult[]): string | null;
56
61
  export declare function modelId(model: string): string;
@@ -111,7 +116,7 @@ prefer?: string[], // v1.53 T2: review.prefer — reorders eligible channels, ne
111
116
  floor?: Tier, // task/config/prior floor from the caller; the author's own tier is ALWAYS applied here (RF-1)
112
117
  history?: string[], // run-scoped picks, oldest to newest; empty preserves the established ranking
113
118
  onSeat?: (seat: number, count: number) => void, demoted?: ReadonlySet<string>, excludeVendors?: ReadonlySet<string>, authors?: readonly string[]): BillingChannel | null;
114
- export type ReviewUnparseableCause = VerdictUnparseableCause | "launch-never-started" | "truncated" | "silent" | "closure-mismatch";
119
+ export type ReviewUnparseableCause = VerdictUnparseableCause | "launch-never-started" | "truncated" | "silent" | "closure-mismatch" | "seat-launch-failed";
115
120
  /**
116
121
  * This shows the reviewer what the task DECLARED, never what the diff may actually reach. The diff
117
122
  * remains a separate stated input, so whether its touched paths fit the declaration stays a reviewer
@@ -6,12 +6,12 @@ import { filesGlob } from "../graph/files-glob.js";
6
6
  import { renderAcceptanceItem, TIERS } from "../graph/schema.js";
7
7
  import { getAdapter } from "../adapters/registry.js";
8
8
  import { shOk } from "../run/git.js";
9
- import { carryReviewFindings, observedReviewFingerprints, reviewFingerprintMatches, structuredFindings, UNIDENTIFIED } from "../run/journal.js";
9
+ import { carryReviewFindings, isDeferredFinding, observedReviewFingerprints, reviewFingerprintMatches, structuredFindings, UNIDENTIFIED } from "../run/journal.js";
10
10
  import { redactSecrets } from "../run/redact.js";
11
11
  import { rankPreferredChannels, reviewPreferenceTieBreak } from "../route/role-pick.js";
12
12
  import { modelProvider } from "../route/preference.js";
13
13
  import { resolveStateDir } from "./cache.js";
14
- import { appendAnchoredReview, COMPLETION_FAKING_CHECKLIST, dewrapPaneVerdict, extractVerdictJson, generateVerdictNonce, parseAnchoredComments, runLlmDetailed, verdictNonceLine } from "./llm.js";
14
+ import { appendAnchoredReview, COMPLETION_FAKING_CHECKLIST, dewrapPaneVerdict, extractVerdictJson, generateVerdictNonce, parseAnchoredComments, runLlmDetailed, SeatLaunchError, verdictNonceLine } from "./llm.js";
15
15
  import { classifyVerdictCause } from "./verdict-cause.js";
16
16
  import { captureDiffCapFor, measureArtifactDiff, reviewableLogicDiff, } from "./artifact-manifest.js";
17
17
  export { isProtectedEvidence, PROTECTED_EVIDENCE_PREFIXES, REGENERABLE_CAPTURE_PATHS, setAsideReceiptPath, setAsideRegenerableCaptures, } from "./artifact-manifest.js";
@@ -164,6 +164,12 @@ export function checkTaskDiffCaps(gate, measured, logicCap, prefix = "") {
164
164
  },
165
165
  };
166
166
  }
167
+ // C-12: a reviewer is excluded for garbage only on the typed malformed-verdict flag. A material finding whose
168
+ // prose mentions the word "unparseable" is a delivered verdict; matching the prose dropped T7's only reviewer.
169
+ export function isGarbageReview(result) {
170
+ return result.gate === "review" && !result.pass && result.meta?.unparseable === true
171
+ && typeof result.meta?.reviewer === "string";
172
+ }
167
173
  export function isDiffCapPark(result) {
168
174
  return result.pass === false
169
175
  && result.meta?.parkKind === "diff-cap"
@@ -400,6 +406,23 @@ function withoutExampleEcho(raw, nonce) {
400
406
  const chars = [...reviewResponseExample(nonce).replace(/\s+/g, "")];
401
407
  return raw.replace(new RegExp(chars.map((c) => c.replace(/[.*+?^${}()|[\]\\]/g, "\\$&")).join("[\\s│|]*"), "g"), "");
402
408
  }
409
+ /**
410
+ * OBS-1195: the parent an anchored comment names in its own optional `finding` field — a 1-based
411
+ * index into this verdict's findings, or a carried prior's id. A deferred entry or a resolved prior
412
+ * settles the anchor; any other valid parent binds it open; anything else leaves it unbound.
413
+ */
414
+ function anchorParent(binding, findings, priors, resolved) {
415
+ if (typeof binding === "number") {
416
+ const entry = Array.isArray(findings) && Number.isInteger(binding) && binding >= 1 ? findings[binding - 1] : undefined;
417
+ if (!entry || typeof entry !== "object" || typeof entry.note !== "string")
418
+ return {};
419
+ return entry.defer === true ? { settled: "deferred" } : { entry: entry.note };
420
+ }
421
+ const prior = matchClosureId(binding, priors);
422
+ if (prior === undefined)
423
+ return {};
424
+ return resolved.includes(prior) ? { settled: "resolved", parent: prior } : { parent: prior };
425
+ }
403
426
  export async function reviewGate(task, worktree, baseRef, author, channels, adapters, cfg, via, excludeReviewers,
404
427
  // OBS-196: run dir for raw-output persistence on an unparseable verdict; absent (older callers,
405
428
  // direct tests) skips persistence and changes nothing else.
@@ -550,13 +573,21 @@ block the merge) or "minor" (style, naming, or preference that should not block)
550
573
  block approval. For a minor concern you have decided not to block on, set "defer": true and give a
551
574
  one-line "rationale" — it is recorded in the review, never dropped.
552
575
  A fix you prescribe that would break suites outside the task's declared write scope (files[]) is a scope finding, never a material one.
576
+ Each material finding names the class of defect it belongs to and binds that class to the goal clause or
577
+ acceptance criterion it violates (or to the regression this diff introduces), states its input → consequence,
578
+ and marks its evidence executed (you ran the reproducer), static (you traced it by reading) or blocked (it
579
+ could not run here). Blocked evidence never turns a finding into a pass, and a worker's own case table or
580
+ enumeration never resolves a finding: judge the diff itself.
553
581
 
554
582
  Respond with ONLY this JSON:
555
- {"nonce": "${nonce}", "approve": true|false, "resolved": [], "reraised": [], "findings": [{"note": "...", "severity": "material"|"minor", "defer": false, "rationale": ""}], "comments": [{"path": "path/to/file", "line": 42, "body": "actionable feedback"}]}
583
+ {"nonce": "${nonce}", "approve": true|false, "resolved": [], "reraised": [], "findings": [{"note": "...", "severity": "material"|"minor", "defer": false, "rationale": ""}], "comments": [{"path": "path/to/file", "line": 42, "body": "actionable feedback", "finding": 1}]}
556
584
  For every prior material, put its fingerprint in exactly one of resolved (verified fixed) or reraised
557
585
  (still a blocking defect). Use only the listed fingerprints; never omit one or put it in both lists.
558
586
  Approve iff no material finding remains and every prior material is resolved.
559
587
  The top-level comments array is optional. Use it only for actionable line-anchored feedback.
588
+ A comment's optional "finding" names its parent: the 1-based index of its entry in findings, or a prior
589
+ fingerprint copied from above. A comment anchored to a deferred entry or a resolved prior does not block;
590
+ a comment naming no parent stays open.
560
591
 
561
592
  ${responseRequirement}
562
593
  `;
@@ -590,18 +621,33 @@ ${responseRequirement}
590
621
  savedBrief = undefined;
591
622
  }
592
623
  }
593
- const llm = await runLlmDetailed(getAdapter(reviewer.adapter, adapters), reviewer.model, prompt, worktree, via ? {
594
- driver: via.driver,
595
- keep: via.keep,
596
- onSlot: via.onSlot,
597
- name: via.nameFor("review", reviewer.adapter),
598
- label: via.labelFor("review"),
599
- } : undefined,
600
- // frontier reviewers routinely need >5min on a configured-cap-sized diff, and `claude -p` buffers all
601
- // output until completion — runLlm's 300s default killed reviews mid-flight, returning empty
602
- // stdout that read as "unparseable" and escalated to re-implementation of green code
603
- // (run-20260709-104447 P87-09). The configured ceiling defaults to that measured 15 minutes.
604
- cfg.review.timeoutMs, reviewer.effort);
624
+ const provider = modelProvider(reviewer.model, reviewer.vendor);
625
+ let llm;
626
+ try {
627
+ llm = await runLlmDetailed(getAdapter(reviewer.adapter, adapters), reviewer.model, prompt, worktree, via ? {
628
+ driver: via.driver,
629
+ keep: via.keep,
630
+ onSlot: via.onSlot,
631
+ name: via.nameFor("review", reviewer.adapter),
632
+ label: via.labelFor("review"),
633
+ } : undefined,
634
+ // frontier reviewers routinely need >5min on a configured-cap-sized diff, and `claude -p` buffers all
635
+ // output until completion — runLlm's 300s default killed reviews mid-flight, returning empty
636
+ // stdout that read as "unparseable" and escalated to re-implementation of green code
637
+ // (run-20260709-104447 P87-09). The configured ceiling defaults to that measured 15 minutes.
638
+ cfg.review.timeoutMs, reviewer.effort);
639
+ }
640
+ catch (error) {
641
+ if (!(error instanceof SeatLaunchError))
642
+ throw error;
643
+ // OBS-1168(b): the seat never launched, so there is no verdict and nothing about the WORK. A typed
644
+ // no-verdict re-routes to another seat in run-gates; an exhausted pool is an infra park.
645
+ return { gate: "review", pass: false,
646
+ details: `review dispatch failed — ${error.message} (reviewer ${reviewer.adapter}:${reviewer.model}; vendor: ${reviewer.vendor}; provider: ${provider}; cause: seat-launch-failed) — failing closed`,
647
+ meta: { ...policyMeta, ...rotationMeta, ...floorMeta, reviewer: channelKey(reviewer), reviewerTier: reviewer.tier,
648
+ vendor: reviewer.vendor, provider, noVerdict: true, classification: "infra", infra: true,
649
+ cause: "seat-launch-failed", ...(savedBrief ? { briefPath: savedBrief } : {}) } };
650
+ }
605
651
  const raw = llm.output;
606
652
  let saved;
607
653
  if (artifactDir) {
@@ -613,7 +659,6 @@ ${responseRequirement}
613
659
  saved = undefined; // persistence is evidence, not a gate input — never fail the gate on it
614
660
  }
615
661
  }
616
- const provider = modelProvider(reviewer.model, reviewer.vendor);
617
662
  // A pane's own dewrap stops at the first parseable nonce-bound object; once the example's echo is gone
618
663
  // a genuinely wrapped verdict behind it is reconstructed here, exactly as llm.ts would have.
619
664
  const echoFree = withoutExampleEcho(raw, nonce);
@@ -653,6 +698,8 @@ ${responseRequirement}
653
698
  ...(cause === "malformed-verdict" ? { unparseable: true } : { noVerdict: true, classification: "infra", infra: true }),
654
699
  cause,
655
700
  ...(closureMismatch ? { resolved: v?.resolved, reraised: v?.reraised, carriedFingerprints: priorIds.flatMap(observedReviewFingerprints) } : {}),
701
+ // OBS-1196: a whole verdict that breaks the closure protocol was DELIVERED; only undelivered bytes are re-asked.
702
+ ...(closureInvalid ? { closureInvalid: true } : {}),
656
703
  bytes, seatAuthoredBytes: bytes,
657
704
  ...(saved ? { rawPath: saved } : {}),
658
705
  ...(savedBrief ? { briefPath: savedBrief } : {}),
@@ -669,25 +716,24 @@ ${responseRequirement}
669
716
  if (decided.pass)
670
717
  decided.headline = "requested changes";
671
718
  decided.pass = false;
672
- // A reviewer may also restate a re-raised material in findings. Preserve the original
673
- // prose once so an unchanged defect keeps the same failure brief across repair rounds.
674
- for (const finding of reraised) {
675
- const line = `- [material] ${finding.note}`;
676
- if (!decided.lines.includes(line))
677
- decided.lines.push(line);
678
- }
679
719
  }
680
- const prose = `reviewer ${reviewer.adapter}:${reviewer.model} (vendor: ${reviewer.vendor}; provider: ${provider}): ${decided.headline}${decided.lines.length ? "\n" + decided.lines.join("\n") : ""}`;
681
- const details = appendAnchoredReview(prose, v);
720
+ // OBS-1195: an anchor leaves the blocking set only through an explicit, unambiguous parent
721
+ // disposition: its own `finding` names a deferred entry of this verdict or a prior it resolved.
722
+ // Path coincidence binds nothing; an unbound or ambiguous anchor stays open exactly as before.
723
+ const comments = parseAnchoredComments(v);
724
+ const rawComments = (comments.length ? v.comments : []);
725
+ const resolvedIds = (v.resolved ?? []).map((id) => matchClosureId(id, priorIds));
726
+ const anchors = comments.map((comment, i) => ({ ...comment, ...anchorParent(rawComments[i]?.finding, v.findings, priorIds, resolvedIds) }));
727
+ const openAnchors = { comments: anchors.filter((a) => !a.settled) };
728
+ const settledAnchors = anchors.flatMap((a, i) => a.settled
729
+ ? [{ path: a.path, line: a.line, body: a.body, disposition: a.settled, finding: rawComments[i].finding }] : []);
682
730
  // Only the verdict's anchors may supply missing evidence, and only when unambiguous.
683
731
  // Reuse the journal's path normalization without changing legacy details-only parsing.
684
- const anchoredPaths = new Set(parseAnchoredComments(v).map((comment) => {
685
- const anchor = structuredFindings("review", `- ${comment.path}:${comment.line} — anchor`)
686
- .find((finding) => finding.class === "review:anchored");
687
- return anchor?.path ?? comment.path;
688
- }));
732
+ const anchorRow = (comment) => structuredFindings("review", `- ${comment.path}:${comment.line} — ${comment.body}`)
733
+ .find((finding) => finding.class === "review:anchored");
734
+ const anchoredPaths = new Set(comments.map((comment) => anchorRow({ ...comment, body: "anchor" })?.path ?? comment.path));
689
735
  const anchoredPath = anchoredPaths.size === 1 ? [...anchoredPaths][0] : undefined;
690
- const ownDetails = appendAnchoredReview(ownLines.join("\n"), v);
736
+ const ownDetails = appendAnchoredReview(ownLines.join("\n"), openAnchors);
691
737
  const currentRows = (ownDetails.trim() ? structuredFindings("review", ownDetails) : [])
692
738
  .map((finding) => finding.path === UNIDENTIFIED && anchoredPath
693
739
  ? { ...finding, path: anchoredPath, fingerprint: `${finding.class}|${anchoredPath}|${finding.symbol}` }
@@ -708,7 +754,50 @@ ${responseRequirement}
708
754
  const unambiguousRows = linkedRows.map((finding) => finding.reraisedFrom
709
755
  && linkedRows.filter((row) => row.reraisedFrom === finding.reraisedFrom).length > 1
710
756
  ? { ...finding, reraisedFrom: undefined } : finding);
711
- const carriedRows = carryReviewFindings(reraised, unambiguousRows);
757
+ // OBS-1195: a deferred entry echoing a prior's id names that prior as the concern it defers. The
758
+ // link rides the row for the journal's anchor fold only; a deferral never re-seats a material chain.
759
+ // A prior any material row restates — even ambiguously, its link cleared above — is claimed, not
760
+ // deferred: the deferral gets no link, so its bound anchors stay open (fail closed).
761
+ const claimed = new Set(linkedRows.map((row) => row.reraisedFrom).filter((id) => id !== undefined));
762
+ const rows = unambiguousRows.map((finding) => {
763
+ if (!isDeferredFinding(finding))
764
+ return finding;
765
+ const entry = v.findings?.find((entry) => entry && typeof entry === "object" && entry.note === finding.note);
766
+ const id = matchClosureId(entry?.reraised, reraised);
767
+ return id && !claimed.has(id) ? { ...finding, reraisedFrom: id } : finding;
768
+ });
769
+ // OBS-1195: current prose is rendered once. A prior's original wording is echoed only when no
770
+ // current finding positively binds it; a bound chain keeps that spelling as lineage instead.
771
+ for (const finding of reraised) {
772
+ const line = `- [material] ${finding.note}`;
773
+ const bound = unambiguousRows.some((row) => row.reraisedFrom === finding.fingerprint);
774
+ if (!bound && !decided.lines.includes(line))
775
+ decided.lines.push(line);
776
+ }
777
+ const prose = `reviewer ${reviewer.adapter}:${reviewer.model} (vendor: ${reviewer.vendor}; provider: ${provider}): ${decided.headline}${decided.lines.length ? "\n" + decided.lines.join("\n") : ""}`;
778
+ const details = appendAnchoredReview(prose, openAnchors);
779
+ // OBS-1195: a prior's `reraisedFrom` is the link an EARLIER verdict drew. Carried forward unrestated,
780
+ // it would read as this verdict's own material claim on that parent and block the anchor fold from
781
+ // honouring an explicit deferral of it. Lineage lives in observedFingerprints; drop the stale link.
782
+ const carried = carryReviewFindings(reraised.map(({ reraisedFrom: _stale, ...prior }) => prior), rows);
783
+ // An open anchor bound to a current entry names that entry's carried chain, only when unique.
784
+ const parentOf = (anchor) => {
785
+ if (anchor.entry === undefined)
786
+ return anchor.parent;
787
+ const owners = carried.filter((f) => f.class !== "review:anchored" && f.note === anchor.entry);
788
+ return owners.length === 1 ? owners[0].fingerprint : undefined;
789
+ };
790
+ const carriedRows = carried.map((finding) => {
791
+ if (finding.class !== "review:anchored")
792
+ return finding;
793
+ // Every comment spelling this row must name the same parent, else the anchor stays unbound.
794
+ const parents = new Set(anchors.filter((a) => {
795
+ const row = a.settled ? undefined : anchorRow(a);
796
+ return row?.path === finding.path && row.note === finding.note;
797
+ }).map(parentOf));
798
+ const [parent] = parents;
799
+ return parents.size === 1 && parent ? { ...finding, boundTo: parent } : finding;
800
+ });
712
801
  return {
713
802
  gate: "review",
714
803
  pass: decided.pass,
@@ -723,6 +812,7 @@ ${responseRequirement}
723
812
  reraisedMatches: (v.reraised ?? []).map((id) => matchClosureId(id, priorIds)).filter((id) => id !== undefined),
724
813
  } : {}),
725
814
  ...(!decided.pass ? { findings: carriedRows } : {}),
815
+ ...(settledAnchors.length ? { settledAnchors } : {}),
726
816
  ...(saved ? { rawPath: saved } : {}),
727
817
  ...(savedBrief ? { briefPath: savedBrief } : {}),
728
818
  },