tickmarkr 2.1.5 → 2.1.6

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,21 @@
1
+ export declare const STATS_COLUMNS: readonly ["author", "reviewer", "dispatches", "deliveries", "delivery rate", "attempts to green", "real reds", "infra reds", "within-task rescues"];
2
+ export interface ChannelStats {
3
+ author: string;
4
+ reviewers: string[];
5
+ dispatches: number;
6
+ deliveries: number;
7
+ deliveryRate: number;
8
+ /** Mean journal-recorded attempts among this channel's delivered tasks; null when it delivered none. */
9
+ attemptsToGreen: number | null;
10
+ realReds: number;
11
+ infraReds: number;
12
+ rescues: string[];
13
+ }
14
+ export interface StatsReport {
15
+ runs: number;
16
+ channels: ChannelStats[];
17
+ }
18
+ /** Reduce every readable run journal under the repository's state directory. */
19
+ export declare function collectChannelStats(cwd?: string): StatsReport;
20
+ export declare function renderStats(report: StatsReport): string;
21
+ export declare function stats(argv: string[], cwd?: string): Promise<string>;
@@ -0,0 +1,210 @@
1
+ import { existsSync, readdirSync } from "node:fs";
2
+ import { join } from "node:path";
3
+ import { classifyFailureOutput } from "../../gates/baseline.js";
4
+ import { stateDirName } from "../../graph/graph.js";
5
+ import { Journal } from "../../run/journal.js";
6
+ // The operator-facing schema is deliberately closed. Adding a fact to the table therefore requires
7
+ // adding it here as a conscious compatibility change, rather than letting incidental journal fields
8
+ // leak into an analytics surface.
9
+ export const STATS_COLUMNS = [
10
+ "author",
11
+ "reviewer",
12
+ "dispatches",
13
+ "deliveries",
14
+ "delivery rate",
15
+ "attempts to green",
16
+ "real reds",
17
+ "infra reds",
18
+ "within-task rescues",
19
+ ];
20
+ const assignmentAuthor = (data) => {
21
+ const assignment = data.assignment;
22
+ if (!assignment || typeof assignment !== "object")
23
+ return undefined;
24
+ const { adapter, model } = assignment;
25
+ return typeof adapter === "string" && typeof model === "string"
26
+ ? `${adapter}:${model}`
27
+ : undefined;
28
+ };
29
+ // Current journals write the reviewer into review prose; the structured field is also accepted so
30
+ // journals produced through gateResultJournalData retain their stronger identity representation.
31
+ const reviewerFrom = (data) => {
32
+ if (typeof data.reviewer === "string" && data.reviewer.trim())
33
+ return data.reviewer.trim();
34
+ if (typeof data.details !== "string")
35
+ return undefined;
36
+ return /\breviewer(?:\s+|:\s*)([\w@./+-]+:[\w@./+-]+)/iu.exec(data.details)?.[1];
37
+ };
38
+ const redEvidence = (data) => {
39
+ const fingerprints = Array.isArray(data.fingerprints)
40
+ ? data.fingerprints.filter((value) => typeof value === "string")
41
+ : [];
42
+ const prose = [data.details, data.error, data.reason]
43
+ .filter((value) => typeof value === "string");
44
+ return [...fingerprints, ...prose].join("\n");
45
+ };
46
+ const historyFor = (tasks, taskId) => {
47
+ let history = tasks.get(taskId);
48
+ if (!history) {
49
+ history = { dispatches: [], deliveries: [] };
50
+ tasks.set(taskId, history);
51
+ }
52
+ return history;
53
+ };
54
+ const dispatchedAuthorFor = (history, data) => {
55
+ const attempt = typeof data.attempt === "number" && Number.isInteger(data.attempt)
56
+ ? data.attempt
57
+ : undefined;
58
+ if (attempt !== undefined) {
59
+ const matched = [...history.dispatches].reverse().find((dispatch) => dispatch.attempt === attempt);
60
+ if (matched)
61
+ return matched.author;
62
+ }
63
+ return history.dispatches.at(-1)?.author;
64
+ };
65
+ const runIdsWithJournals = (cwd) => {
66
+ const runsDir = join(cwd, stateDirName(cwd), "runs");
67
+ if (!existsSync(runsDir))
68
+ return [];
69
+ return readdirSync(runsDir, { withFileTypes: true })
70
+ .filter((entry) => entry.isDirectory() && entry.name.startsWith("run-")
71
+ && existsSync(join(runsDir, entry.name, "journal.jsonl")))
72
+ .map((entry) => entry.name)
73
+ .sort();
74
+ };
75
+ /** Reduce every readable run journal under the repository's state directory. */
76
+ export function collectChannelStats(cwd = process.cwd()) {
77
+ const runIds = runIdsWithJournals(cwd);
78
+ const channels = new Map();
79
+ const channelFor = (author) => {
80
+ let channel = channels.get(author);
81
+ if (!channel) {
82
+ channel = {
83
+ author, reviewers: new Set(), dispatches: 0, deliveries: 0,
84
+ attemptsToGreen: 0, realReds: 0, infraReds: 0, rescues: new Set(),
85
+ };
86
+ channels.set(author, channel);
87
+ }
88
+ return channel;
89
+ };
90
+ for (const runId of runIds) {
91
+ const events = Journal.open(cwd, runId).read();
92
+ const tasks = new Map();
93
+ for (const [eventIndex, event] of events.entries()) {
94
+ if (!event.taskId)
95
+ continue;
96
+ const history = historyFor(tasks, event.taskId);
97
+ if (event.event === "task-dispatch") {
98
+ const author = assignmentAuthor(event.data);
99
+ if (!author)
100
+ continue;
101
+ channelFor(author).dispatches += 1;
102
+ history.dispatches.push({
103
+ author,
104
+ ...(typeof event.data.attempt === "number" && Number.isInteger(event.data.attempt)
105
+ ? { attempt: event.data.attempt }
106
+ : {}),
107
+ eventIndex,
108
+ });
109
+ continue;
110
+ }
111
+ if (event.event === "task-done") {
112
+ const author = assignmentAuthor(event.data) ?? history.dispatches.at(-1)?.author;
113
+ if (!author)
114
+ continue;
115
+ const channel = channelFor(author);
116
+ channel.deliveries += 1;
117
+ const recordedAttempts = typeof event.data.attempts === "number"
118
+ && Number.isInteger(event.data.attempts) && event.data.attempts >= 0
119
+ ? event.data.attempts
120
+ : history.dispatches.filter((dispatch) => dispatch.eventIndex < eventIndex).length;
121
+ channel.attemptsToGreen += recordedAttempts;
122
+ history.deliveries.push({ author, eventIndex });
123
+ continue;
124
+ }
125
+ if (event.event === "review-retry") {
126
+ const author = dispatchedAuthorFor(history, event.data);
127
+ if (!author)
128
+ continue;
129
+ for (const reviewer of [event.data.flaked, event.data.retried]) {
130
+ if (typeof reviewer === "string" && reviewer.trim())
131
+ channelFor(author).reviewers.add(reviewer.trim());
132
+ }
133
+ continue;
134
+ }
135
+ if (event.event !== "gate-result")
136
+ continue;
137
+ const author = dispatchedAuthorFor(history, event.data);
138
+ if (!author)
139
+ continue;
140
+ const channel = channelFor(author);
141
+ if (event.data.gate === "review") {
142
+ const reviewer = reviewerFrom(event.data);
143
+ if (reviewer)
144
+ channel.reviewers.add(reviewer);
145
+ }
146
+ if (event.data.pass !== false)
147
+ continue;
148
+ const infra = event.data.infra === true || classifyFailureOutput(redEvidence(event.data)) === "infra";
149
+ if (infra)
150
+ channel.infraReds += 1;
151
+ else
152
+ channel.realReds += 1;
153
+ }
154
+ // A rescue is task-matched, not attempt-matched: one edge says the failed author and delivering
155
+ // author faced the same task in the same run. Repeated attempts on the failed author do not make
156
+ // the task easier and therefore do not manufacture extra controlled comparisons.
157
+ for (const [taskId, history] of tasks) {
158
+ for (const delivery of history.deliveries) {
159
+ const failedAuthors = new Set(history.dispatches
160
+ .filter((dispatch) => dispatch.eventIndex < delivery.eventIndex && dispatch.author !== delivery.author)
161
+ .map((dispatch) => dispatch.author));
162
+ for (const failedAuthor of failedAuthors) {
163
+ channelFor(delivery.author).rescues.add(`${failedAuthor} → ${delivery.author} (${runId}/${taskId})`);
164
+ }
165
+ }
166
+ }
167
+ }
168
+ return {
169
+ runs: runIds.length,
170
+ channels: [...channels.values()]
171
+ .sort((a, b) => a.author.localeCompare(b.author, "en"))
172
+ .map((channel) => ({
173
+ author: channel.author,
174
+ reviewers: [...channel.reviewers].sort((a, b) => a.localeCompare(b, "en")),
175
+ dispatches: channel.dispatches,
176
+ deliveries: channel.deliveries,
177
+ deliveryRate: channel.dispatches === 0 ? 0 : channel.deliveries / channel.dispatches,
178
+ attemptsToGreen: channel.deliveries === 0 ? null : channel.attemptsToGreen / channel.deliveries,
179
+ realReds: channel.realReds,
180
+ infraReds: channel.infraReds,
181
+ rescues: [...channel.rescues].sort((a, b) => a.localeCompare(b, "en")),
182
+ })),
183
+ };
184
+ }
185
+ const percent = (rate) => `${Number((rate * 100).toFixed(1))}%`;
186
+ export function renderStats(report) {
187
+ const lines = [`tickmarkr stats — ${report.runs} run${report.runs === 1 ? "" : "s"}`];
188
+ if (report.channels.length === 0)
189
+ return [...lines, "no channels"].join("\n");
190
+ lines.push(STATS_COLUMNS.join(" | "));
191
+ for (const channel of report.channels) {
192
+ lines.push([
193
+ channel.author,
194
+ channel.reviewers.join(", ") || "—",
195
+ channel.dispatches,
196
+ channel.deliveries,
197
+ percent(channel.deliveryRate),
198
+ channel.attemptsToGreen === null ? "—" : Number(channel.attemptsToGreen.toFixed(2)),
199
+ channel.realReds,
200
+ channel.infraReds,
201
+ channel.rescues.join("; ") || "—",
202
+ ].join(" | "));
203
+ }
204
+ return lines.join("\n");
205
+ }
206
+ export async function stats(argv, cwd = process.cwd()) {
207
+ if (argv.length > 0)
208
+ throw new Error("stats takes no run id — it reads every run in the state directory");
209
+ return renderStats(collectChannelStats(cwd));
210
+ }
@@ -7,7 +7,7 @@ import { formatOwnedName, parseOwnedName, } from "../../drivers/types.js";
7
7
  import { blockedTasks, graphDefinitionHash, loadGraph, stateDirName } from "../../graph/graph.js";
8
8
  import { GATE_NAMES } from "../../graph/schema.js";
9
9
  import { foldActivity } from "../../run/activity.js";
10
- import { Journal, engagementComparable, isQualityFailureParkKind, recordedTaskFailureKind, runHasEnded, } from "../../run/journal.js";
10
+ import { Journal, engagementComparable, isQualityFailureParkKind, preservedRefsByTask, recordedTaskFailureKind, runHasEnded, upheldFeedbackByTask, } from "../../run/journal.js";
11
11
  import { isPidLive } from "../../run/lock.js";
12
12
  import { normalizeGateOutcome } from "../../run/outcome.js";
13
13
  import { desiredPanes } from "../../run/reconcile.js";
@@ -979,6 +979,21 @@ binaryVersion, now = Date.now(), animationFrame = 0, workerLiveness = new Map(),
979
979
  if (!taskIds.has(taskId))
980
980
  phases.delete(taskId);
981
981
  const hotPhase = [...phases.values()].sort((a, b) => a.order - b.order).at(-1);
982
+ // OBS-738: recovery facts stay on the journal's two established reducers. The prefix fold keeps an
983
+ // older journal equally readable by asking upheldFeedbackByTask what was active at that exact
984
+ // restore boundary; current resume-restore rows additionally carry the same task for narration.
985
+ const preservedRefs = preservedRefsByTask(events);
986
+ const restoredUpheldFeedback = new Set(events.flatMap((event, index) => event.event === "resume-restore" && event.taskId
987
+ && upheldFeedbackByTask(events.slice(0, index)).has(event.taskId)
988
+ ? [event.taskId]
989
+ : []));
990
+ const recoveryLinesForTask = (taskId) => [
991
+ ...(preservedRefs.get(taskId) ?? []).flatMap(({ ref, diffCommand }) => [
992
+ `preserved worktree — ${taskId} — ${ref}`,
993
+ ` ${diffCommand}`,
994
+ ]),
995
+ ...(restoredUpheldFeedback.has(taskId) ? [`upheld feedback restored for ${taskId}`] : []),
996
+ ].map(sanitizeTaskText);
982
997
  const cells = renderedTasks.map((t) => {
983
998
  const folded = journalRowsOnly && comparable ? taskRows.get(t.id) : undefined;
984
999
  const st = folded?.state ?? replayed?.get(t.id) ?? t.status;
@@ -1011,7 +1026,7 @@ binaryVersion, now = Date.now(), animationFrame = 0, workerLiveness = new Map(),
1011
1026
  if (!unicode) {
1012
1027
  // machine/CI surface — journals without phase-start stay byte-identical; new phase-aware frames
1013
1028
  // use an ASCII spinner so pipes never receive terminal-only braille/ANSI.
1014
- const rows = cells.map(({ t, st, merged, label, assignCol, livePhase, states, priorGraph, pane }) => {
1029
+ const rows = cells.flatMap(({ t, st, merged, label, assignCol, livePhase, states, priorGraph, pane }) => {
1015
1030
  const chain = gateChain(states, false);
1016
1031
  const prefix = livePhase ? ` ${ASCII_SPINNER[animationFrame % ASCII_SPINNER.length]} ${t.id} ` : ` ${surfaceTaskBox(st, merged)} ${t.id} `;
1017
1032
  const suffix = ` ${chain}${priorGraph ? ` ${PRIOR_GRAPH_MARKER}` : ""} ${livePhase ? "running" : surfaceStatusWord(st)}${label} ${assignCol}${pane ? ` pane ${pane}` : ""}`;
@@ -1020,7 +1035,10 @@ binaryVersion, now = Date.now(), animationFrame = 0, workerLiveness = new Map(),
1020
1035
  // can exceed the terminal width, and the pre-floor math paid for it out of the TITLE — 20 of
1021
1036
  // 41 rows in a live 41-task run named no task at all. These rows already overflow `width`
1022
1037
  // (a pane name is 60 columns on its own), so the floor costs wrapping, never the graph.
1023
- return `${prefix}${shortGoal(t.title, Math.max(MACHINE_TITLE_FLOOR, width - prefix.length - suffix.length))}${suffix}`;
1038
+ return [
1039
+ `${prefix}${shortGoal(t.title, Math.max(MACHINE_TITLE_FLOOR, width - prefix.length - suffix.length))}${suffix}`,
1040
+ ...recoveryLinesForTask(t.id).map((line) => ` ${line}`),
1041
+ ];
1024
1042
  });
1025
1043
  const zone = journalRowsOnly ? `${divider}zone ${localZoneLabel(zoneReference)}` : "";
1026
1044
  const header = runId
@@ -1210,6 +1228,12 @@ binaryVersion, now = Date.now(), animationFrame = 0, workerLiveness = new Map(),
1210
1228
  return [legend(headerRow), dim(ruleRow), ...rows];
1211
1229
  };
1212
1230
  const gatesLegend = legend(` gates ${GATE_NAMES.map((gate) => `${gate.slice(0, 2)} ${gate}`).join(" ")}`);
1231
+ const recoveryLines = cells.flatMap((cell) => recoveryLinesForTask(cell.t.id));
1232
+ const recoverySection = recoveryLines.length === 0 ? [] : [
1233
+ "",
1234
+ ` ${ok("▌")} ${title("RECOVERY")}`,
1235
+ ...recoveryLines.map((line) => legend(` ${line}`)),
1236
+ ];
1213
1237
  const effort = effortPanel(g.tasks, record && !comparable ? undefined : events, boardColumns);
1214
1238
  return {
1215
1239
  content: boardRows([
@@ -1220,6 +1244,7 @@ binaryVersion, now = Date.now(), animationFrame = 0, workerLiveness = new Map(),
1220
1244
  taskSection,
1221
1245
  "",
1222
1246
  ...tableRows(),
1247
+ ...recoverySection,
1223
1248
  gatesLegend,
1224
1249
  "",
1225
1250
  rule(boardColumns),
@@ -5,7 +5,7 @@ export type CommandResult = string | {
5
5
  };
6
6
  export type CommandMap = Record<string, (argv: string[]) => Promise<CommandResult>>;
7
7
  export declare const COMMANDS: CommandMap;
8
- export declare const USAGE = "tickmarkr \u2014 spec-driven orchestration harness for AI coding agents\nusage: tickmarkr <command>\n init guided setup + doctor; init --agent [--force] [--docs] adds agent skills/docs\n doctor re-probe adapters, herdr, auth; print capability matrix (--fix writes the test-runner ignore when a safe edit exists)\n fleet interactive fleet editor (fleet --print for CI drift checks)\n compile <src> spec \u2192 .tickmarkr/graph.json (fails without acceptance criteria)\n scope <intent> draft a compiled native spec beside an answered intent (--force to overwrite)\n plan dry-run routing table + cost estimate + floor lints\n eval run checked-in fixtures against every channel in isolated temp repos\n run execute the graph (--concurrency N --driver auto|herdr|subprocess|orca --route-strict; orca runs only when named)\n status live run state\n verify run the gate battery standalone against merge-base(--base, HEAD)..HEAD \u2014 no daemon, one verdict (--base main --criteria <file> | --task <id> [--files <glob>] [--author adapter:model] [--no-review] [--json])\n resume <id> continue a run from its journal\n report <id> cost/quality report (--md for committable execution record)\n profile show learned routing profile (profile reset = forget history via cursor, keeps telemetry)\n ui open the Fleet Studio TUI (full-screen tabbed cockpit)\n unlock remove a stale/garbage run lock (refuses if the holder is alive)\n beat <tier> record one supervision beat for orchestrator|orchestrator-context|overseer|overseer-context|watch, --seat <identity> required (--stand-down to hand off); a supervising seat's own watcher loop calls it, and status reads the tier STALE once the beats stop\n approve <id> <task> release a park (--uphold sides with the reviewer and funds a fixed attempt; --by <name> --reason <text>); takes effect on resume";
8
+ export declare const USAGE = "tickmarkr \u2014 spec-driven orchestration harness for AI coding agents\nusage: tickmarkr <command>\n init guided setup + doctor; init --agent [--force] [--docs] adds agent skills/docs\n doctor re-probe adapters, herdr, auth; print capability matrix (--fix writes the test-runner ignore when a safe edit exists)\n fleet interactive fleet editor (fleet --print for CI drift checks)\n compile <src> spec \u2192 .tickmarkr/graph.json (fails without acceptance criteria)\n scope <intent> draft a compiled native spec beside an answered intent (--force to overwrite)\n plan dry-run routing table + cost estimate + floor lints\n eval run checked-in fixtures against every channel in isolated temp repos\n run execute the graph (--concurrency N --driver auto|herdr|subprocess|orca --route-strict; orca runs only when named)\n status live run state\n stats all-run channel delivery, red, rescue, author and reviewer statistics\n verify run the gate battery standalone against merge-base(--base, HEAD)..HEAD \u2014 no daemon, one verdict (--base main --criteria <file> | --task <id> [--files <glob>] [--author adapter:model] [--no-review] [--json])\n resume <id> continue a run from its journal\n report <id> cost/quality report (--md for committable execution record)\n profile show learned routing profile (profile reset = forget history via cursor, keeps telemetry)\n ui open the Fleet Studio TUI (full-screen tabbed cockpit)\n unlock remove a stale/garbage run lock (refuses if the holder is alive)\n beat <tier> record one supervision beat for orchestrator|orchestrator-context|overseer|overseer-context|watch, --seat <identity> required (--stand-down to hand off); a supervising seat's own watcher loop calls it, and status reads the tier STALE once the beats stop\n approve <id> <task> release a park (--uphold sides with the reviewer and funds a fixed attempt; --by <name> --reason <text>); takes effect on resume";
9
9
  export declare function dispatch(cmd: string | undefined, argv: string[], commands?: CommandMap): Promise<{
10
10
  out: string;
11
11
  code: number;
package/dist/cli/index.js CHANGED
@@ -14,6 +14,7 @@ import { report } from "./commands/report.js";
14
14
  import { resume } from "./commands/resume.js";
15
15
  import { run } from "./commands/run.js";
16
16
  import { scope } from "./commands/scope.js";
17
+ import { stats } from "./commands/stats.js";
17
18
  import { status } from "./commands/status.js";
18
19
  import { ui } from "./commands/ui.js";
19
20
  import { unlock } from "./commands/unlock.js";
@@ -21,7 +22,7 @@ import { verify } from "./commands/verify.js";
21
22
  import { version } from "./commands/version.js";
22
23
  const normalize = (r) => typeof r === "string" ? { out: r, code: 0 } : r;
23
24
  export const COMMANDS = {
24
- init, doctor, fleet, compile, scope, plan, run, status, resume, report, profile, ui, unlock, approve, beat, version, verify, eval: evalCommand,
25
+ init, doctor, fleet, compile, scope, plan, run, status, stats, resume, report, profile, ui, unlock, approve, beat, version, verify, eval: evalCommand,
25
26
  };
26
27
  const VERSION_FLAGS = new Set(["version", "--version", "-v"]);
27
28
  const HELP_CMDS = new Set(["help", "-h", "--help"]);
@@ -37,6 +38,7 @@ usage: tickmarkr <command>
37
38
  eval run checked-in fixtures against every channel in isolated temp repos
38
39
  run execute the graph (--concurrency N --driver auto|herdr|subprocess|orca --route-strict; orca runs only when named)
39
40
  status live run state
41
+ stats all-run channel delivery, red, rescue, author and reviewer statistics
40
42
  verify run the gate battery standalone against merge-base(--base, HEAD)..HEAD — no daemon, one verdict (--base main --criteria <file> | --task <id> [--files <glob>] [--author adapter:model] [--no-review] [--json])
41
43
  resume <id> continue a run from its journal
42
44
  report <id> cost/quality report (--md for committable execution record)
@@ -117,7 +117,7 @@ const VOCAB_RE = /\b(?:error|fail(?:ed|ure|ing)?)\b/i;
117
117
  // shapes are emitted by the process that was asked to run the oracle; they are deliberately kept in
118
118
  // this runner-output classifier rather than applied to any judge-authored reason text. A real test
119
119
  // failure still dominates below because one regression-shaped line makes the whole output regression.
120
- const INFRA_RE = /\bE(?:AGAIN|MFILE|NFILE|NOMEM|NOSPC)\b|JavaScript heap out of memory|Cannot allocate memory|Resource temporarily unavailable|Token not found in system keyring|Process from config\.webServer was not able to start/i;
120
+ const INFRA_RE = /\bE(?:AGAIN|MFILE|NFILE|NOMEM|NOSPC)\b|JavaScript heap out of memory|Cannot allocate memory|Resource temporarily unavailable|Token not found in system keyring|Process from config\.webServer was not able to start|\[birpc\] rpc is closed, cannot call\b/i;
121
121
  // Capture invalidation is deliberately narrower than the gate's infrastructure vocabulary above:
122
122
  // keyring/config-webServer startup failures remain gate concerns, while this policy is specifically
123
123
  // for evidence that the capture ran while the machine was resource-starved.
@@ -127,6 +127,12 @@ const CAPTURE_EXHAUSTION_RE = /\bE(?:AGAIN|MFILE|NFILE|NOMEM|NOSPC)\b|JavaScript
127
127
  const ERROR_CLASS_RE = /\b[A-Za-z][A-Za-z0-9]*Error\b/;
128
128
  const isInfraLine = (l) => INFRA_RE.test(l) && !ERROR_CLASS_RE.test(l) && !namesFailure(l) && !SUMMARY_FAIL_RE.test(l);
129
129
  const namesRegression = (l) => (isFailureShaped(l) || ERROR_CLASS_RE.test(l)) && !isInfraLine(l);
130
+ // Infrastructure signatures must enter the fingerprint diff too. Otherwise a signature such as
131
+ // birpc's assertion-free RPC death is classified correctly in the raw output but collapses to the
132
+ // content-free UNRECOGNIZED_FAILURE marker before the fresh-failure path can ask the same classifier.
133
+ // This does not make the vocabulary open-ended: INFRA_RE is still the one closed list, and
134
+ // classifyFailureOutput's ERROR_CLASS_RE/namesFailure vetoes still decide mixed lines.
135
+ const isFingerprintShaped = (l) => isFailureShaped(l) || INFRA_RE.test(l);
130
136
  /**
131
137
  * What a nonzero runner exit is evidence OF. `undefined` when the output names neither — the
132
138
  * unreadable-runner case the existing fail-closed path already owns.
@@ -196,10 +202,10 @@ export function fingerprint(output) {
196
202
  // prefixed form keeps fingerprinting too (baseline-recorded package-level reds stay forgivable).
197
203
  const shaped = [];
198
204
  for (const l of lines) {
199
- if (isFailureShaped(l))
205
+ if (isFingerprintShaped(l))
200
206
  shaped.push(l);
201
207
  const stripped = stripTurboPrefix(l);
202
- if (stripped !== undefined && !PASS_LINE_RE.test(stripped) && isFailureShaped(stripped))
208
+ if (stripped !== undefined && !PASS_LINE_RE.test(stripped) && isFingerprintShaped(stripped))
203
209
  shaped.push(stripped);
204
210
  }
205
211
  if (!shaped.length)
@@ -544,21 +550,6 @@ export async function compareToBaseline(cwd, commands, baseline, enabled) {
544
550
  continue;
545
551
  }
546
552
  const raw = (r.stdout + "\n" + r.stderr).split(cwd).join("");
547
- // T9: classify BEFORE the baseline diff, and record it on every nonzero result. An infra-only
548
- // exit means the runner never completed a suite, so there is nothing to forgive and nothing
549
- // verified — it fails, and `meta.infra` marks it so the merge predicate cannot read it as a
550
- // satisfied gate even if some future producer reports it as a pass. Baseline forgiveness stays
551
- // exactly where it belongs: on failures the runner actually reported and the baseline already had.
552
- const classification = classifyFailureOutput(raw);
553
- if (classification === "infra") {
554
- record({
555
- gate: name,
556
- pass: false,
557
- details: `exit ${r.code} on infrastructure alone — the runner never completed a suite, so this gate verified nothing:\n${unrecognizedEvidence(raw) || raw.trim().split("\n").slice(0, 10).join("\n")}`,
558
- meta: { classification, infra: true },
559
- });
560
- continue;
561
- }
562
553
  // OBS-278: only a failure SHAPE is a verdict — everything fingerprint() keeps is one, except the
563
554
  // unrecognized-output marker, which is evidence for the operator and never grounds to reject.
564
555
  // ponytail: ceiling — a runner whose failure output holds no shape above and whose baseline is
@@ -567,6 +558,25 @@ export async function compareToBaseline(cwd, commands, baseline, enabled) {
567
558
  // that runner's position rule (leading verdict + identifier, or identifier + separator + trailing
568
559
  // verdict); loosening back to vocabulary re-opens OBS-278.
569
560
  const { failing, unreadable } = freshFailures(entry, raw);
561
+ // T9: classify the FRESH diff before charging it. The complete runner output can legitimately
562
+ // contain a baseline-recorded assertion beside a newly introduced infrastructure death; letting
563
+ // that known assertion outvote the fresh birpc line turns machine failure into a worker defect.
564
+ // `classifyFailureOutput` remains the single discriminator. When there is no fresh fingerprint,
565
+ // retain the whole-output read so a repeated infra abort can never be baseline-forgiven as green.
566
+ const freshClassification = failing.length ? classifyFailureOutput(failing.join("\n")) : undefined;
567
+ const classification = freshClassification ?? (!failing.length ? classifyFailureOutput(raw) : undefined);
568
+ if (classification === "infra") {
569
+ const evidence = failing.length
570
+ ? failing.slice(0, 10).join("\n")
571
+ : unrecognizedEvidence(raw) || raw.trim().split("\n").slice(0, 10).join("\n");
572
+ record({
573
+ gate: name,
574
+ pass: false,
575
+ details: `exit ${r.code}; the fresh failures carry infrastructure evidence alone — the runner never completed a suite, so this gate verified nothing:\n${evidence}`,
576
+ meta: { classification, infra: true },
577
+ });
578
+ continue;
579
+ }
570
580
  // OBS-534 (T2): only a recorded VERDICT can be forgiven. A green baseline has no red to forgive,
571
581
  // and neither has a capture that was killed at its ceiling — it recorded a cause instead, so it
572
582
  // fails closed on the same branch rather than reading as "only pre-existing failures". Legacy
@@ -521,6 +521,35 @@ async function gateCommitSubject(base, head, wt) {
521
521
  return head; // fail closed to the exact object id if canonicalization fails
522
522
  return createHash("sha256").update(history.stdout).digest("hex");
523
523
  }
524
+ // A CPU total cannot prove a tree empty: a newly launched live process can still have a measured
525
+ // total of zero at the host's clock resolution. This probe asks the narrower cardinality question
526
+ // OBS-737 needs. A readable process table with no marker root is measured empty; a failed or empty
527
+ // table is unmeasurable. Descendants are closed over PPID because not every child retains the
528
+ // dispatch-script marker in its own argv.
529
+ async function observeWorkerProcessTree(marker, cwd) {
530
+ const snapshot = await shGit("ps -Awwo pid=,ppid=,command=", cwd, 15_000);
531
+ if (snapshot.code !== 0)
532
+ return "unmeasurable";
533
+ const rows = [];
534
+ for (const line of snapshot.stdout.split("\n")) {
535
+ const match = /^\s*(\d+)\s+(\d+)\s+(.*)$/.exec(line);
536
+ if (match)
537
+ rows.push({ pid: match[1], ppid: match[2], command: match[3] });
538
+ }
539
+ if (rows.length === 0)
540
+ return "unmeasurable";
541
+ const tree = new Set(rows.filter((row) => row.command.includes(marker)).map((row) => row.pid));
542
+ for (let grew = true; grew;) {
543
+ grew = false;
544
+ for (const row of rows) {
545
+ if (!tree.has(row.pid) && tree.has(row.ppid)) {
546
+ tree.add(row.pid);
547
+ grew = true;
548
+ }
549
+ }
550
+ }
551
+ return tree.size === 0 ? "empty" : "running";
552
+ }
524
553
  const OBSERVE_CHUNK_BYTES = 64 * 1024;
525
554
  const OBSERVE_BUDGET_BYTES = 256 * 1024 * 1024;
526
555
  let observeBudgetBytes = OBSERVE_BUDGET_BYTES;
@@ -605,6 +634,20 @@ async function boundedGitObservation(command, worktree, budget) {
605
634
  return undefined;
606
635
  return result.stdout;
607
636
  }
637
+ // "No worktree delta" means no staged/unstaged/untracked bytes AND no commits beyond the task's
638
+ // dispatch base. Both reads are bounded and preserve the observer's third state: a failed probe is
639
+ // unreadable, never clean.
640
+ async function observeWorktreeDelta(base, worktree) {
641
+ let budget = observeBudgetBytes;
642
+ const status = await boundedGitObservation("GIT_OPTIONAL_LOCKS=0 git status --porcelain=v1 -z --untracked-files=all", worktree, budget);
643
+ if (status === undefined)
644
+ return "unreadable";
645
+ budget -= Buffer.byteLength(status);
646
+ const ahead = await boundedGitObservation(`GIT_OPTIONAL_LOCKS=0 git rev-list --count ${shq(base)}..HEAD`, worktree, budget);
647
+ if (ahead === undefined || !/^\d+$/.test(ahead.trim()))
648
+ return "unreadable";
649
+ return status.length === 0 && ahead.trim() === "0" ? "unchanged" : "changed";
650
+ }
608
651
  // The signature has four git-owned inputs: HEAD, staged blob/mode/path identity, porcelain path and
609
652
  // state, and the set of worktree paths whose bytes Git cannot supply. Those paths contribute their
610
653
  // filesystem identity, mode, symlink text, and content. A failed or over-budget leg is a third
@@ -1091,10 +1134,10 @@ export async function runDaemon(repoRoot, opts = {}) {
1091
1134
  };
1092
1135
  // gateFails/consults are execTask-scoped counters passed in so a park row is a rich verified-failure
1093
1136
  // observation (e.g. ladder-exhausted + gateFails:4); every task-human row has a closed kind, never prose alone.
1094
- const park = async (t, reason, kind, assignment, attempts, startMs, gateFails = 0, consults = 0, tokens, metered = 0, retryMode = "fresh") => {
1137
+ const park = async (t, reason, kind, assignment, attempts, startMs, gateFails = 0, consults = 0, tokens, metered = 0, retryMode = "fresh", details = {}) => {
1095
1138
  graph = setStatus(graph, t.id, "human");
1096
1139
  saveGraph(repoRoot, graph);
1097
- journal.append("task-human", t.id, { reason, kind });
1140
+ journal.append("task-human", t.id, { ...details, reason, kind });
1098
1141
  if (assignment) {
1099
1142
  // OBS-547: `metered` counts CHARGEABLE metered attempts, so an unchargeable dispatch passes 0 and
1100
1143
  // the count is omitted rather than written as 0 or as `1` beside `attempts: 0` — a row claiming
@@ -1172,6 +1215,22 @@ export async function runDaemon(repoRoot, opts = {}) {
1172
1215
  journal.append("worktree-preserved", t.id, { ref });
1173
1216
  return driver.worktree(repoRoot, taskBranch, taskBase);
1174
1217
  };
1218
+ // A dead worker with a clean checkout still needs a durable recovery handle: there may be no
1219
+ // pane left to identify even the commit it was dispatched from. preserveWorktree deliberately
1220
+ // creates no ref for ordinary clean recreations, so this exceptional terminal path pins HEAD
1221
+ // explicitly under the same recovery namespace. Reconfirm the delta immediately before the
1222
+ // ref write; a change or unreadable recheck withdraws the park.
1223
+ const preserveDeadWorker = async (worktree, taskBase) => {
1224
+ const state = await observeWorktreeDelta(taskBase, worktree);
1225
+ if (state !== "unchanged")
1226
+ return { state };
1227
+ const head = await gitHead(worktree);
1228
+ const ref = `refs/tickmarkr/preserved/${head}`;
1229
+ const updated = await shGit(`git update-ref ${shq(ref)} ${shq(head)}`, worktree);
1230
+ if (updated.code !== 0)
1231
+ throw new Error(`could not preserve dead worker HEAD at ${ref}: ${updated.stderr || updated.stdout}`);
1232
+ return { state, ref };
1233
+ };
1175
1234
  const r = route(t, cfg, channels, profile, undefined, demotedChannels);
1176
1235
  for (const lint of r.lints)
1177
1236
  journal.append("routing-lint", t.id, { lint });
@@ -2184,6 +2243,7 @@ export async function runDaemon(repoRoot, opts = {}) {
2184
2243
  // subprocess tree that REACHED its exit marker, not about what the worker claimed.
2185
2244
  let processExited = false;
2186
2245
  let earlyLaunchDead = false;
2246
+ let deadWorkerPark;
2187
2247
  let settleParsed;
2188
2248
  let seedResult;
2189
2249
  // v1.22 T5 / OBS-19: auto-answer a fingerprint-matched trust dialog exactly once per slot.
@@ -2323,6 +2383,8 @@ export async function runDaemon(repoRoot, opts = {}) {
2323
2383
  let quotaStreak = 0;
2324
2384
  let rowSaturationHeld = false; // journaled once per attempt when the kill stands down
2325
2385
  let cpuHeld = false; // likewise for the CPU leg's stand-down (OBS-548)
2386
+ let paneReadHeld = false;
2387
+ let paneStatusHeld = false;
2326
2388
  while (Date.now() - lastProgressAt < stallWindowMs) {
2327
2389
  const sliceStart = Date.now();
2328
2390
  const remaining = stallWindowMs - (sliceStart - lastProgressAt);
@@ -2350,7 +2412,39 @@ export async function runDaemon(repoRoot, opts = {}) {
2350
2412
  break;
2351
2413
  }
2352
2414
  }
2353
- const paneText = await driver.read(slot, PANE_READ_ROWS);
2415
+ // A failed pane read is absence of evidence, never evidence of an absent pane. Keep the
2416
+ // rolling taskTimeoutMinutes window as the backstop and name the held probe once.
2417
+ let paneText;
2418
+ try {
2419
+ paneText = await driver.read(slot, PANE_READ_ROWS);
2420
+ }
2421
+ catch (error) {
2422
+ // Preserve the pre-existing exception path when the independent status probe still
2423
+ // sees a pane. The outer attempt finally owns accountant cleanup on that path. Only
2424
+ // an undetectable status makes the read failure relevant to the death detector, and
2425
+ // that genuinely unmeasurable pair fails open to the rolling timeout.
2426
+ let paneUndetectable = true;
2427
+ try {
2428
+ paneUndetectable = await driver.status(slot) === "unknown";
2429
+ }
2430
+ catch {
2431
+ // Two unreadable pane probes are still unmeasurable, never proof of death.
2432
+ }
2433
+ if (!paneUndetectable)
2434
+ throw error;
2435
+ if (!paneReadHeld) {
2436
+ paneReadHeld = true;
2437
+ journal.append("worker-dead-held", t.id, {
2438
+ slot: slot.name, attempt, reason: "pane-read-unreadable",
2439
+ error: error instanceof Error ? error.message : String(error),
2440
+ });
2441
+ }
2442
+ await armCpuLeg(false);
2443
+ const spent = Date.now() - sliceStart;
2444
+ if (spent < slice)
2445
+ await new Promise((r) => setTimeout(r, Math.min(slice - spent, 1_000)));
2446
+ continue;
2447
+ }
2354
2448
  if (paneText.length > 0)
2355
2449
  everHadOutput = true;
2356
2450
  // OBS-117 (v1.71 T6): zero raw output by the early-launch deadline is a dead channel now.
@@ -2419,7 +2513,24 @@ export async function runDaemon(repoRoot, opts = {}) {
2419
2513
  // gate held. page on "idle" too: herdr's blocked-scrape is strict and proved flaky
2420
2514
  // for TUI dialogs (live check: cursor's trust dialog scraped as idle).
2421
2515
  // "unknown"/"working" never page.
2422
- const st = await driver.status(slot);
2516
+ let st;
2517
+ try {
2518
+ st = await driver.status(slot);
2519
+ }
2520
+ catch (error) {
2521
+ if (!paneStatusHeld) {
2522
+ paneStatusHeld = true;
2523
+ journal.append("worker-dead-held", t.id, {
2524
+ slot: slot.name, attempt, reason: "pane-status-unreadable",
2525
+ error: error instanceof Error ? error.message : String(error),
2526
+ });
2527
+ }
2528
+ await armCpuLeg(false);
2529
+ const spent = Date.now() - sliceStart;
2530
+ if (spent < slice)
2531
+ await new Promise((r) => setTimeout(r, Math.min(slice - spent, 1_000)));
2532
+ continue;
2533
+ }
2423
2534
  if (st !== lastStatus) {
2424
2535
  lastStatus = st;
2425
2536
  journal.append("worker-status", t.id, { slot: slot.name, status: st, attempt });
@@ -2459,6 +2570,95 @@ export async function runDaemon(repoRoot, opts = {}) {
2459
2570
  // daemon has nothing left to do — the pane falls back under the fast-kill and page
2460
2571
  // watchdogs like any other, instead of riding the whole rolling window untended.
2461
2572
  const nudgePending = nudgeable && (!nudged || nudgeDeadline !== undefined);
2573
+ // OBS-737: disposition for the one fully measured death state. `unknown` alone is not
2574
+ // absence (status parsing can fail), and an empty read alone is not absence (a live pane
2575
+ // can be quiet); together they are the existing driver-level pane absence witness. The
2576
+ // process probe and both worktree observations retain their own third states. Only the
2577
+ // explicit conjunction parks; every other state falls through to the unchanged rolling
2578
+ // timeout below. Seeded launches are excluded because their process marker is knowingly
2579
+ // unmeasurable (armCpuLeg documents that contract above).
2580
+ const paneAbsentCandidate = st === "unknown" && paneText.trim().length === 0;
2581
+ // A subprocess can exit between waitOutput and read while its stdout is still draining.
2582
+ // Confirm emptiness in this same poll before paying for `ps`; then confirm once more
2583
+ // after the process probe yielded the event loop. Any bytes or read error withdraw the
2584
+ // absence witness, so a fast completed worker cannot be parked in that drain race.
2585
+ let paneAbsent = paneAbsentCandidate;
2586
+ if (paneAbsent) {
2587
+ try {
2588
+ const confirmation = await driver.read(slot, PANE_READ_ROWS);
2589
+ paneAbsent = confirmation.trim().length === 0;
2590
+ if (!paneAbsent) {
2591
+ everHadOutput = true;
2592
+ if (stallProgress.observe({ paneText: confirmation, contextTokens }))
2593
+ lastProgressAt = Date.now();
2594
+ }
2595
+ }
2596
+ catch (error) {
2597
+ paneAbsent = false;
2598
+ if (!paneReadHeld) {
2599
+ paneReadHeld = true;
2600
+ journal.append("worker-dead-held", t.id, {
2601
+ slot: slot.name, attempt, reason: "pane-read-unreadable",
2602
+ error: error instanceof Error ? error.message : String(error),
2603
+ });
2604
+ }
2605
+ }
2606
+ }
2607
+ const processTree = paneAbsent && worktreeSinceLaunch === "unchanged" && !hasSeed
2608
+ ? await observeWorkerProcessTree(dispatchScript, wt)
2609
+ : "unmeasurable";
2610
+ if (processTree === "empty") {
2611
+ try {
2612
+ const confirmation = await driver.read(slot, PANE_READ_ROWS);
2613
+ paneAbsent = confirmation.trim().length === 0;
2614
+ if (!paneAbsent) {
2615
+ everHadOutput = true;
2616
+ if (stallProgress.observe({ paneText: confirmation, contextTokens }))
2617
+ lastProgressAt = Date.now();
2618
+ }
2619
+ }
2620
+ catch (error) {
2621
+ paneAbsent = false;
2622
+ if (!paneReadHeld) {
2623
+ paneReadHeld = true;
2624
+ journal.append("worker-dead-held", t.id, {
2625
+ slot: slot.name, attempt, reason: "pane-read-unreadable",
2626
+ error: error instanceof Error ? error.message : String(error),
2627
+ });
2628
+ }
2629
+ }
2630
+ }
2631
+ const worktreeDelta = processTree === "empty"
2632
+ && paneAbsent ? await observeWorktreeDelta(taskBase, wt)
2633
+ : "unreadable";
2634
+ // The first process snapshot can race a just-starting child after the dispatch pane
2635
+ // disappeared. Re-read it after the pane and worktree legs have both held: preservation
2636
+ // is terminal, so a process appearing in that interval must withdraw the park rather
2637
+ // than be orphaned by it. The final worktree recheck remains inside preserveDeadWorker.
2638
+ const confirmedProcessTree = worktreeDelta === "unchanged"
2639
+ ? await observeWorkerProcessTree(dispatchScript, wt)
2640
+ : "unmeasurable";
2641
+ const deathCertain = paneAbsent
2642
+ && processTree === "empty"
2643
+ && confirmedProcessTree === "empty"
2644
+ && worktreeDelta === "unchanged";
2645
+ if (deathCertain) {
2646
+ const preservation = await preserveDeadWorker(wt, taskBase);
2647
+ if (!preservation.ref) {
2648
+ journal.append("worker-dead-held", t.id, {
2649
+ slot: slot.name, attempt, reason: `worktree-${preservation.state}`,
2650
+ });
2651
+ continue;
2652
+ }
2653
+ const ref = preservation.ref;
2654
+ const reason = `worker is unambiguously dead: pane absent, process tree empty, and worktree unchanged; preserved at ${ref}`;
2655
+ deadWorkerPark = { ref, reason };
2656
+ journal.append("worktree-preserved", t.id, { ref });
2657
+ journal.append("worker-dead-held", t.id, {
2658
+ slot: slot.name, attempt, reason: "unambiguous-worker-death", ref,
2659
+ });
2660
+ break;
2661
+ }
2462
2662
  // T1 review fix: the kill's "no output growth" leg clocks off the RAW growth signals,
2463
2663
  // never lastProgressAt alone — the flat-token rule (stall.ts) deliberately suppresses
2464
2664
  // the re-arm report on row growth once tokens stick, and contextTokens is sticky across
@@ -2632,7 +2832,13 @@ export async function runDaemon(repoRoot, opts = {}) {
2632
2832
  if (!finished && exitCode === null) {
2633
2833
  // timed out (or only ever saw false positives): harvest whatever the pane holds now
2634
2834
  timedOut = Date.now() - lastProgressAt >= stallWindowMs;
2635
- output = await driver.read(slot, PANE_READ_ROWS);
2835
+ try {
2836
+ output = await driver.read(slot, PANE_READ_ROWS);
2837
+ }
2838
+ catch {
2839
+ // The poll loop already recorded the unreadable pane. Retain the last readable bytes
2840
+ // so this ambiguous path still reaches the ordinary timeout/consult backstop.
2841
+ }
2636
2842
  finished = new RegExp(trailerPattern(nonce)).test(output);
2637
2843
  const exit = exitRe.exec(output);
2638
2844
  exitCode = exit ? Number(exit[1]) : null;
@@ -2766,6 +2972,10 @@ export async function runDaemon(repoRoot, opts = {}) {
2766
2972
  tokens = addUsage(tokens, attemptUsage);
2767
2973
  metered++;
2768
2974
  }
2975
+ if (deadWorkerPark) {
2976
+ await park(t, deadWorkerPark.reason, "stall", assignment, attempt + 1, startMs, gateFails, consults, tokens, metered, retryMode, { ref: deadWorkerPark.ref });
2977
+ return;
2978
+ }
2769
2979
  let result = settleParsed ?? adapter.parse(output, nonce);
2770
2980
  const workerFinished = finished;
2771
2981
  const workerCause = classifyWorkerResultCause({ output, ok: result.ok, finished, exitCode, summary: result.summary, timedOut, deadChannel: deadChannelKilled });
@@ -29,6 +29,11 @@ export declare const ATTEMPT_CAP_RELEASE: "attempt-cap";
29
29
  export declare const GATE_SATISFIED_RELEASE: "gate-satisfied";
30
30
  export declare const REVIEW_UPHELD_RELEASE: "review-upheld";
31
31
  export declare const RECHECK_RELEASE: "recheck";
32
+ export interface PreservedRef {
33
+ ref: string;
34
+ diffCommand: string;
35
+ }
36
+ export declare function preservedRefsByTask(events: JournalEvent[]): Map<string, PreservedRef[]>;
32
37
  export declare function reviewRoundsSinceApproval(events: JournalEvent[], taskId: string): number;
33
38
  export declare function upheldFeedbackByTask(events: JournalEvent[]): Map<string, string>;
34
39
  export interface StructuredFinding {
@@ -2,7 +2,7 @@ import { AsyncLocalStorage } from "node:async_hooks";
2
2
  import { appendFileSync, existsSync, mkdirSync, readFileSync, readdirSync } from "node:fs";
3
3
  import { join } from "node:path";
4
4
  import { z } from "zod";
5
- import { channelKey, TokenUsageSchema } from "../adapters/types.js";
5
+ import { channelKey, shq, TokenUsageSchema } from "../adapters/types.js";
6
6
  import { stateDirName, taskContentDigest, tickmarkrDir } from "../graph/graph.js";
7
7
  import { GATE_NAMES, TIERS } from "../graph/schema.js";
8
8
  import { buildProfile, classify } from "../route/profile.js";
@@ -56,6 +56,24 @@ export const REVIEW_UPHELD_RELEASE = "review-upheld";
56
56
  // so the corrected declaration is the thing that earns the green. Budget semantics match attempt-cap
57
57
  // (fresh attempts, tried survives) because the park cost the task its remaining budget.
58
58
  export const RECHECK_RELEASE = "recheck";
59
+ // OBS-738: one authority for every recovery surface. The ref is accepted only from the row that
60
+ // preservation itself writes; task-human prose, branch heads and commit history are deliberately
61
+ // absent from this fold. Keep every row in journal order — one task can be recreated more than once,
62
+ // and the terminal record owes the operator every resulting recovery handle, not merely the newest.
63
+ export function preservedRefsByTask(events) {
64
+ const byTask = new Map();
65
+ for (const event of events) {
66
+ if (event.event !== "worktree-preserved" || !event.taskId || typeof event.data.ref !== "string"
67
+ || event.data.ref === "")
68
+ continue;
69
+ const ref = event.data.ref;
70
+ byTask.set(event.taskId, [
71
+ ...(byTask.get(event.taskId) ?? []),
72
+ { ref, diffCommand: `git diff ${shq(`${ref}^!`)}` },
73
+ ]);
74
+ }
75
+ return byTask;
76
+ }
59
77
  // OBS-189: review rounds are scoped to the current ENGAGEMENT — the stretch since the newest operator
60
78
  // approval for the task. A whole-journal count re-parks an upheld task before its funded attempt can
61
79
  // dispatch (measured live on run-20260726-213539), making a fresh journal the only escape. A T15
@@ -1069,19 +1087,36 @@ export class Journal {
1069
1087
  ? undefined
1070
1088
  : DecisionEventSchema.parse({ ...eventOrDecision, ts: new Date().toISOString() });
1071
1089
  const event = decisionRow?.event ?? eventOrDecision;
1090
+ const rowTaskId = decisionRow && "taskId" in decisionRow ? decisionRow.taskId : taskId;
1072
1091
  const inputData = decisionRow?.data ?? data;
1092
+ // OBS-738: terminal and resume records reduce the journal that precedes them. Neither re-derives
1093
+ // recovery facts from task-human prose: preserved refs come from preservedRefsByTask, and the
1094
+ // upheld brief comes from the established prompt/replay reducer.
1095
+ const priorEvents = event === "run-end" || event === "resume-restore" ? this.read() : [];
1096
+ const reducedData = event === "run-end"
1097
+ ? (() => {
1098
+ const preservedRefs = [...preservedRefsByTask(priorEvents)].flatMap(([preservedTaskId, refs]) => refs.map(({ ref, diffCommand }) => ({ taskId: preservedTaskId, ref, diffCommand })));
1099
+ return preservedRefs.length > 0 ? { ...inputData, preservedRefs } : inputData;
1100
+ })()
1101
+ : event === "resume-restore" && rowTaskId && upheldFeedbackByTask(priorEvents).has(rowTaskId)
1102
+ ? {
1103
+ ...inputData,
1104
+ upheldFeedbackRestoredFor: rowTaskId,
1105
+ summary: `upheld feedback restored for ${rowTaskId}`,
1106
+ }
1107
+ : inputData;
1073
1108
  const evidence = judgePersistence.getStore();
1074
1109
  const failed = evidence?.invocations.filter((invocation) => invocation.transcript !== undefined) ?? [];
1075
1110
  const persistedData = event === "judge-retry" && failed.length > 0
1076
1111
  ? {
1077
- ...inputData,
1112
+ ...reducedData,
1078
1113
  transcript: failed[0].transcript,
1079
1114
  ...(failed[1] ? { retryTranscript: failed[1].transcript } : {}),
1080
1115
  }
1081
- : inputData;
1082
- const row = decisionRow ?? {
1083
- ts: new Date().toISOString(), event, ...(taskId ? { taskId } : {}), data: persistedData,
1084
- };
1116
+ : reducedData;
1117
+ const row = decisionRow
1118
+ ? { ...decisionRow, data: persistedData }
1119
+ : { ts: new Date().toISOString(), event, ...(taskId ? { taskId } : {}), data: persistedData };
1085
1120
  // T3 secret redaction: only the persisted bytes are masked — the caller's data stays untouched in
1086
1121
  // memory. The narrator receives the persisted (masked) row so a pane sink never shows a credential.
1087
1122
  const line = redactSecrets(JSON.stringify(row));
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "tickmarkr",
3
- "version": "2.1.5",
3
+ "version": "2.1.6",
4
4
  "description": "Spec in, verified work out.",
5
5
  "type": "module",
6
6
  "license": "MIT",
@@ -103,7 +103,7 @@ through brief lineage. **An executor choice nobody made is still an executor cho
103
103
  fraction (`ORCH · v1.19 4/5`, updated on every task-done); tickmarkr opens ONE TAB PER TASK, labelled
104
104
  with the task id and holding that task's worker plus its judge/review/consult panes (tickmarkr
105
105
  updates it). Never long context strings or ✓-chains.
106
- 2. **Orchestrator**: Launch the orchestrator with your agent host. Spawning on current herdr is two-step — the one-shot `agent start --cwd` form was removed in the herdr CLI redesign and now fails with `unknown option` (OBS-138): first create the pane with `herdr tab create --workspace <ws> --cwd <repo> --label "ORCH · <version>"` and parse `result.root_pane.pane_id` from its JSON, then start the agent in it. For Claude Code, use `herdr agent start orchestrator --kind claude --pane <root-pane-id> -- --permission-mode bypassPermissions` (append `--model <m>` after the `--` if the operator has a policy). For Codex, use `herdr agent start orchestrator --kind codex --pane <root-pane-id> -- --dangerously-bypass-approvals-and-sandbox` (add `--model <m>` to specify the model). The unsandboxed flag is REQUIRED: codex's `workspace-write` sandbox keeps `.git` refs read-only, so a sandboxed orchestrator's `tickmarkr run` dies at integration-branch creation — do not downgrade it. Workers you never spawn — tickmarkr spawns its own visible worker panes. Auxiliary agents you do spawn (consultants, reviewers, scouts) follow the same forms: never launch a claude session in plan mode or default permission mode for autonomous work — both stall on per-command approval prompts nobody is watching; claude is always `--permission-mode bypassPermissions --settings '{"promptSuggestionEnabled":false}'`, and a read-only codex consultant may use `--sandbox read-only`. **That `--settings` pair is not cosmetic and it is not optional:** claude-code's AUTOSUGGEST renders context-plausible ghost text into an idle seat's prompt line that is BYTE-IDENTICAL to a typed draft in text-format reads (OBS-482), so a supervising tier cannot tell a seat's own unsent work from a rendering artifact without `agent read --format ansi`. Turning the suggester off at spawn removes the ambiguity at its source instead of paying for the discrimination at every read. Verified against the shipped binary: `claude --settings '{"promptSuggestionEnabled":false}' -p …` exits 0 with a real response, and the key appears in the binary's own settings schema. **For kimi, pass `-y`** (`herdr agent start <name> --kind kimi --pane <id> -- -y`) — the adapter already launches its own workers that way (`src/adapters/kimi.ts:204`), and a kimi seat spawned without it sits on an approval prompt having done nothing. **Herdr cannot see that state**: it reports a kimi pane as `agent_status: working` with `screen_detection_skipped: true` while the prompt is up, so the BLOCKED-STATE watcher below is blind on this vendor and the spawn flag is the ONLY control. Every vendor you spawn needs its auto-approve form named here; a vendor absent from this list is a seat that will hang.
106
+ 2. **Orchestrator**: Launch the orchestrator with your agent host. Spawning on current herdr is two-step — the one-shot `agent start --cwd` form was removed in the herdr CLI redesign and now fails with `unknown option` (OBS-138): first create the pane with `herdr tab create --workspace <ws> --cwd <repo> --label "ORCH · <version>"` and parse `result.root_pane.pane_id` from its JSON, then start the agent in it. For Claude Code, use `herdr agent start orchestrator --kind claude --pane <root-pane-id> -- --permission-mode bypassPermissions` (append `--model <m>` after the `--` if the operator has a policy). For Codex, use `herdr agent start orchestrator --kind codex --pane <root-pane-id> -- --dangerously-bypass-approvals-and-sandbox` (add `--model <m>` to specify the model). The unsandboxed flag is REQUIRED: codex's `workspace-write` sandbox keeps `.git` refs read-only, so a sandboxed orchestrator's `tickmarkr run` dies at integration-branch creation — do not downgrade it. Workers you never spawn — tickmarkr spawns its own visible worker panes. Auxiliary agents you do spawn (consultants, reviewers, scouts) follow the same forms: never launch a claude session in plan mode or default permission mode for autonomous work — both stall on per-command approval prompts nobody is watching; claude is always `--permission-mode bypassPermissions --settings '{"promptSuggestionEnabled":false}'`. **For a codex consultant, use `-a never --sandbox workspace-write` — NOT `--sandbox read-only`.** ⚠ **`--sandbox read-only` CONTRADICTS this skill's own completion protocol and will hang the seat.** Every seat you spawn is told to deliver an ARTIFACT ending in a terminal MARKER, because that is the only completion signal the artifact watcher can key on (`done` is turn end). A read-only sandbox cannot write that artifact, so codex blocks on `Would you like to make the following edits?` for its OWN report — and the report exists ONLY in the pending edit, so abandoning the prompt destroys the work rather than merely delaying it. Measured 2026-08-28: a consultant spawned `--sandbox read-only` finished a 14,604-byte verdict, sat blocked on the write, and the operator saw the prompt before the supervising tier did. `read-only` is correct ONLY for a seat that writes nothing at all — which, under the artifact+marker rule, is no seat this skill tells you to spawn. When the prompt does appear, answer **"Yes, and don't ask again for these files"** rather than plain yes: plain yes re-blocks on the next write of the same file. **That `--settings` pair is not cosmetic and it is not optional:** claude-code's AUTOSUGGEST renders context-plausible ghost text into an idle seat's prompt line that is BYTE-IDENTICAL to a typed draft in text-format reads (OBS-482), so a supervising tier cannot tell a seat's own unsent work from a rendering artifact without `agent read --format ansi`. Turning the suggester off at spawn removes the ambiguity at its source instead of paying for the discrimination at every read. Verified against the shipped binary: `claude --settings '{"promptSuggestionEnabled":false}' -p …` exits 0 with a real response, and the key appears in the binary's own settings schema. **For kimi, pass `-y`** (`herdr agent start <name> --kind kimi --pane <id> -- -y`) — the adapter already launches its own workers that way (`src/adapters/kimi.ts:204`), and a kimi seat spawned without it sits on an approval prompt having done nothing. **Herdr cannot see that state**: it reports a kimi pane as `agent_status: working` with `screen_detection_skipped: true` while the prompt is up, so the BLOCKED-STATE watcher below is blind on this vendor and the spawn flag is the ONLY control. Every vendor you spawn needs its auto-approve form named here; a vendor absent from this list is a seat that will hang.
107
107
  3. **Standing instructions travel as a brief FILE, never as pane text** — PTY input truncates at ~1024B and a
108
108
  truncated brief silently drops policy. Write the full brief to `<repo>/.tickmarkr/overseer/ORCH-BRIEF.md`
109
109
  (inside the tickmarkr state dir — already self-gitignored, no exclude step needed), then send one line:
@@ -1053,6 +1053,17 @@ twice.** They are mission-independent on purpose: nothing here names a task, a l
1053
1053
  concluded both forms were valid. The tell is unavailable unless the tool volunteers it. Corollary —
1054
1054
  an instrument that takes an input must be handed a DELIBERATELY BAD one before its clean runs are
1055
1055
  worth anything (rule 11 applied to tools, not just to gates).
1056
+ ⚠ **AND THE COMMONEST WRONG INPUT IS A BASE REF: after the first merge, a task's diff against the
1057
+ run's `baseRef` is NEVER that task's diff.** Workers branch from the INTEGRATION TIP, so once any task
1058
+ has merged, `git diff baseRef..HEAD` in a later worktree reports that task PLUS every task merged
1059
+ before it, and the number looks entirely plausible. Diff from the task's OWN base — the integration
1060
+ commit it branched from — and say which base you used whenever you quote a size.
1061
+ **Measured 2026-08-28 in one run, twice, in both directions.** A supervising seat quoted "712
1062
+ insertions across 7 files" for a task whose real contribution was **300 across 2**; the surplus was two
1063
+ other tasks' merged work. On the next task the same trap was **larger** — 920 across 11 versus a true
1064
+ 167 across 4 — and it was caught only because the other tier had just been burned by it. A scope
1065
+ judgement, a cost claim, or a review-size argument built on the baseRef diff is measuring three tasks
1066
+ and calling it one.
1056
1067
  14. **A unit is not a measurement.** A configured timeout is a KILL CEILING, not a duration — never compare
1057
1068
  it to a wall clock or quote it to an operator as an estimate.
1058
1069
  15. **Verify through the path that LOADS, not the path you edited.** Mirrored trees and symlinks mean your