tickmarkr 2.1.4 → 2.1.6

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/run/git.js CHANGED
@@ -63,6 +63,53 @@ export const deriveForkCap = (concurrency, cores = availableParallelism()) => Ma
63
63
  export const runWithForkBudget = (concurrency, fn) => forkBudget.run(String(deriveForkCap(concurrency)), fn);
64
64
  /** The cap owned by the run on this async context; the standalone default outside one. */
65
65
  export const resolvedForkCap = () => forkBudget.getStore() ?? DEFAULT_FORK_CAP;
66
+ const positiveInt = (v) => typeof v === "number" && Number.isInteger(v) && v > 0;
67
+ /**
68
+ * Three states, never two. A record carrying NO capacity is an older record from before this stamp
69
+ * existed: it keeps exactly the verdict it has today. A record carrying a capacity it cannot state —
70
+ * half the pair, an empty container, a zero, a negative, an unparseable value — is a NEWER record
71
+ * that is malformed, and reading it as an older one is how a fail-closed guard stops firing silently.
72
+ */
73
+ export function readCapacity(value) {
74
+ if (value === undefined)
75
+ return { state: "absent" };
76
+ if (value === null || typeof value !== "object")
77
+ return { state: "malformed" };
78
+ const { forkCap, cores } = value;
79
+ return positiveInt(forkCap) && positiveInt(cores)
80
+ ? { state: "present", capacity: { forkCap, cores } }
81
+ : { state: "malformed" };
82
+ }
83
+ /**
84
+ * May a verdict recorded under `recorded` be reused — forgiven, cached, replayed — by a session
85
+ * running under `current`? Absent → yes, unchanged. Present and identical → yes. Malformed, a
86
+ * different capacity, or a current capacity the caller could not state → no.
87
+ */
88
+ export function sameCapacity(recorded, current) {
89
+ const read = readCapacity(recorded);
90
+ if (read.state === "absent")
91
+ return true;
92
+ if (read.state === "malformed" || current === undefined)
93
+ return false;
94
+ return read.capacity.forkCap === current.forkCap && read.capacity.cores === current.cores;
95
+ }
96
+ export const describeCapacity = (value) => {
97
+ const read = readCapacity(value);
98
+ return read.state === "present"
99
+ ? `fork cap ${read.capacity.forkCap} of ${read.capacity.cores} cores`
100
+ : read.state === "absent" ? "an unrecorded capacity" : "a malformed capacity";
101
+ };
102
+ /**
103
+ * The capacity a child spawned on THIS async context would receive: the same precedence `shell`
104
+ * applies below — an operator export of the cap wins over the run's own derived value — beside the
105
+ * cores it was divided from. A caller holding a command's own result reads the capacity off THAT
106
+ * result (`ShResult.capacity`, stamped where the child's environment was built); this is for the
107
+ * decisions taken BEFORE any child exists — a cache hit, a reuse predicate.
108
+ */
109
+ export const resolvedCapacity = () => ({
110
+ forkCap: Number(FORK_CAP_ENV in process.env ? process.env[FORK_CAP_ENV] : resolvedForkCap()),
111
+ cores: availableParallelism(),
112
+ });
66
113
  /** The shipped shell ceiling: the fallback every caller gets when nothing measured a better one. */
67
114
  export const DEFAULT_SHELL_TIMEOUT_MS = 600000;
68
115
  /**
@@ -103,6 +150,13 @@ function shell(cmd, cwd, timeoutMs, login) {
103
150
  // OBS-110: apply the run's own fork cap only when the operator has not already set one.
104
151
  if (!(FORK_CAP_ENV in env))
105
152
  env[FORK_CAP_ENV] = resolvedForkCap();
153
+ // T7: the capacity every result of this shell carries, read HERE — off the environment the child
154
+ // is about to receive, after the precedence above has settled. An operator export is already in
155
+ // `env`, so what gets recorded is the operator's number, which is the case a release was re-taken
156
+ // for; re-deriving the run's own budget after the command returned would stamp a cap no child ran
157
+ // under. `Number` of an unparseable export is NaN, which every reader treats as malformed and
158
+ // therefore fails closed — the honest direction when the cap in play cannot be stated.
159
+ const capacity = { forkCap: Number(env[FORK_CAP_ENV]), cores: availableParallelism() };
106
160
  const attempt = () => new Promise((resolve) => {
107
161
  const startedAt = Date.now();
108
162
  // detached: bash gets its own process group so a timeout can kill the whole tree —
@@ -120,7 +174,7 @@ function shell(cmd, cwd, timeoutMs, login) {
120
174
  clearTimeout(timer);
121
175
  stdout += stdoutDecoder.end();
122
176
  stderr += stderrDecoder.end();
123
- resolve({ code, stdout, stderr: err ?? stderr, timedOut, durationMs: Date.now() - startedAt });
177
+ resolve({ code, stdout, stderr: err ?? stderr, timedOut, durationMs: Date.now() - startedAt, capacity });
124
178
  };
125
179
  const timer = setTimeout(() => {
126
180
  timedOut = true;
@@ -170,7 +224,7 @@ function shell(cmd, cwd, timeoutMs, login) {
170
224
  // Bounded, and the bound is what makes a persisting shortage a REPORTED failure rather than a
171
225
  // wedged daemon: past it the caller gets the refusal's own text under exit 127, as before.
172
226
  if (n >= SPAWN_ATTEMPT_LIMIT) {
173
- return { code: 127, stdout: "", stderr: String(r.refused), durationMs: Date.now() - startedAt };
227
+ return { code: 127, stdout: "", stderr: String(r.refused), durationMs: Date.now() - startedAt, capacity };
174
228
  }
175
229
  await new Promise((wake) => setTimeout(wake, SPAWN_RETRY_BACKOFF_MS * n));
176
230
  }
@@ -29,6 +29,11 @@ export declare const ATTEMPT_CAP_RELEASE: "attempt-cap";
29
29
  export declare const GATE_SATISFIED_RELEASE: "gate-satisfied";
30
30
  export declare const REVIEW_UPHELD_RELEASE: "review-upheld";
31
31
  export declare const RECHECK_RELEASE: "recheck";
32
+ export interface PreservedRef {
33
+ ref: string;
34
+ diffCommand: string;
35
+ }
36
+ export declare function preservedRefsByTask(events: JournalEvent[]): Map<string, PreservedRef[]>;
32
37
  export declare function reviewRoundsSinceApproval(events: JournalEvent[], taskId: string): number;
33
38
  export declare function upheldFeedbackByTask(events: JournalEvent[]): Map<string, string>;
34
39
  export interface StructuredFinding {
@@ -36,6 +41,7 @@ export interface StructuredFinding {
36
41
  path: string;
37
42
  symbol: string;
38
43
  note: string;
44
+ rationale?: string;
39
45
  fingerprint: string;
40
46
  }
41
47
  export declare const UNIDENTIFIED = "<unidentified>";
@@ -52,6 +58,15 @@ export declare const UNIDENTIFIED = "<unidentified>";
52
58
  * normalized words (see toFinding).
53
59
  */
54
60
  export declare function structuredFindings(gate: string, details: string, _scopeFiles?: string[]): StructuredFinding[];
61
+ export declare function isDeferredFinding(finding: StructuredFinding): boolean;
62
+ /**
63
+ * The findings a PASSING review DEFERRED — the rows a blocking-only projection drops on the floor.
64
+ * A passing review's details are prose; without this the deferral has no identity a later round can
65
+ * match, and every structured reader of the journal is blind to a defect the reviewer itself named.
66
+ */
67
+ export declare function deferredReviewFindings(details: string): StructuredFinding[];
68
+ /** The exact review.ts details fragment represented by a structured review finding. */
69
+ export declare function renderStructuredReviewFinding(finding: StructuredFinding): string;
55
70
  export interface PriorRunJournal {
56
71
  runId: string;
57
72
  events: JournalEvent[];
@@ -126,6 +141,19 @@ export declare function journaledFailureBrief(events: JournalEvent[], taskId: st
126
141
  * finding was dropped at the exact moment the operator paid for another attempt to fix it. A review
127
142
  * that DECLINED (`skipped`) is not a verdict and neither adds nor retires — fail closed. Findings are
128
143
  * keyed by fingerprint, so a reviewer restating one across rounds carries it once, not once per round.
144
+ *
145
+ * v2.1.5 T2: a passing review settles the findings it BLOCKED on. It does not settle the ones it
146
+ * DEFERRED — those it saw, declined to block on, and recorded a rationale for, and nothing has fixed
147
+ * them. So a pass retires the blocking set and re-seats its own deferrals, and the two retirements
148
+ * stay distinguishable: the blocking finding is gone, the deferral travels on as accepted work.
149
+ *
150
+ * A deferral's bound is the SAME single release as a blocking finding's — the operator accepting the
151
+ * review gate itself (`GATE_SATISFIED_RELEASE` stamped `gate: "review"`), the one approval in which a
152
+ * human actually looked at what the reviewer waved through. It is deliberately NOT bounded by a round
153
+ * count or by a time window: both retire a finding by arithmetic nobody read, which is the silent drop
154
+ * this fold exists to refuse. Nor can it accumulate — a reviewer restating the same path/note round
155
+ * after round re-seats ONE fingerprint, and a revised rationale replaces the prior rationale on that
156
+ * row. N rounds of the same concern therefore carry the newest accepted explanation once, not N rows.
129
157
  */
130
158
  export declare function outstandingReviewFindings(events: JournalEvent[], taskId: string): StructuredFinding[];
131
159
  /** The findings a funded repair must carry into the next dispatch, or undefined if none is pending. */
@@ -2,7 +2,7 @@ import { AsyncLocalStorage } from "node:async_hooks";
2
2
  import { appendFileSync, existsSync, mkdirSync, readFileSync, readdirSync } from "node:fs";
3
3
  import { join } from "node:path";
4
4
  import { z } from "zod";
5
- import { channelKey, TokenUsageSchema } from "../adapters/types.js";
5
+ import { channelKey, shq, TokenUsageSchema } from "../adapters/types.js";
6
6
  import { stateDirName, taskContentDigest, tickmarkrDir } from "../graph/graph.js";
7
7
  import { GATE_NAMES, TIERS } from "../graph/schema.js";
8
8
  import { buildProfile, classify } from "../route/profile.js";
@@ -56,6 +56,24 @@ export const REVIEW_UPHELD_RELEASE = "review-upheld";
56
56
  // so the corrected declaration is the thing that earns the green. Budget semantics match attempt-cap
57
57
  // (fresh attempts, tried survives) because the park cost the task its remaining budget.
58
58
  export const RECHECK_RELEASE = "recheck";
59
+ // OBS-738: one authority for every recovery surface. The ref is accepted only from the row that
60
+ // preservation itself writes; task-human prose, branch heads and commit history are deliberately
61
+ // absent from this fold. Keep every row in journal order — one task can be recreated more than once,
62
+ // and the terminal record owes the operator every resulting recovery handle, not merely the newest.
63
+ export function preservedRefsByTask(events) {
64
+ const byTask = new Map();
65
+ for (const event of events) {
66
+ if (event.event !== "worktree-preserved" || !event.taskId || typeof event.data.ref !== "string"
67
+ || event.data.ref === "")
68
+ continue;
69
+ const ref = event.data.ref;
70
+ byTask.set(event.taskId, [
71
+ ...(byTask.get(event.taskId) ?? []),
72
+ { ref, diffCommand: `git diff ${shq(`${ref}^!`)}` },
73
+ ]);
74
+ }
75
+ return byTask;
76
+ }
59
77
  // OBS-189: review rounds are scoped to the current ENGAGEMENT — the stretch since the newest operator
60
78
  // approval for the task. A whole-journal count re-parks an upheld task before its funded attempt can
61
79
  // dispatch (measured live on run-20260726-213539), making a fresh journal the only escape. A T15
@@ -114,10 +132,11 @@ export function upheldFeedbackByTask(events) {
114
132
  // for a resolved one is the silent-lie shape the gates exist to refuse, so the field says so outright.
115
133
  export const UNIDENTIFIED = "<unidentified>";
116
134
  const ANCHORED_RE = /^- (\S+?):(\d+) — (.*)$/; // "## Anchored review" rows (llm.ts)
117
- const REVIEW_ROW_RE = /^- \[([^\]]+)\] (.*)$/; // "- [material] …" (review.ts)
135
+ const REVIEW_ROW_START_RE = /^- \[([^\]\r\n]+)\] /gm; // "- [material] …" (review.ts)
118
136
  const JUDGE_ROW_RE = /^✗ ([\w.-]+): (.*)$/; // "✗ c1: …" (acceptance.ts) — id, then reason
119
137
  const PATH_RE = /\b((?:[\w.@~+-]+\/)+[\w.@~+-]+\.\w{1,6})\b/;
120
138
  const LINE_REF_RE = /(:\d+(?::\d+)?\b)|(\bline \d+\b)/gi;
139
+ const REVIEW_RATIONALE_SEPARATOR = " — rationale: ";
121
140
  // ponytail: repo-relative tail from the first known top-level directory — enough to make an absolute
122
141
  // worktree path and its repo-relative twin the same identity. Widen the marker list if a run ever
123
142
  // names findings outside these roots.
@@ -143,10 +162,46 @@ function identifierIn(note) {
143
162
  // code identity still has one stable identity of its own: its own words, with the volatile tokens
144
163
  // swept out so line/path churn cannot mint a new symbol for the same finding. It is the reviewer's
145
164
  // own bytes, never a guess, and it can never fuse two different findings into one.
146
- function toFinding(cls, note, path, symbol) {
165
+ function toFinding(cls, note, path, symbol, rationale) {
147
166
  const p = path || UNIDENTIFIED;
148
167
  const s = symbol || normalizeGateFailure(note) || UNIDENTIFIED;
149
- return { class: cls, path: p, symbol: s, note, fingerprint: `${cls}|${p}|${s}` };
168
+ return {
169
+ class: cls, path: p, symbol: s, note,
170
+ ...(rationale !== undefined ? { rationale } : {}),
171
+ fingerprint: `${cls}|${p}|${s}`,
172
+ };
173
+ }
174
+ /**
175
+ * Decode review.ts's row rendering without treating physical lines as findings. A review finding's
176
+ * note and rationale are JSON strings before rendering and may therefore contain newlines; the next
177
+ * typed row (or the anchored-review block) is the record boundary. The fixed rationale separator is
178
+ * removed before identity is computed, so changing only why a concern was accepted re-seats it.
179
+ */
180
+ function reviewDetailFindings(details) {
181
+ const anchoredAt = details.indexOf("\n\n## Anchored review");
182
+ let prose = anchoredAt === -1 ? details : details.slice(0, anchoredAt);
183
+ const inconsistencyAt = prose.search(/\nreview (?:finding|verdict) inconsistent:/);
184
+ if (inconsistencyAt !== -1)
185
+ prose = prose.slice(0, inconsistencyAt);
186
+ const starts = [...prose.matchAll(REVIEW_ROW_START_RE)];
187
+ return starts.map((start, i) => {
188
+ const contentStart = start.index + start[0].length;
189
+ const contentEnd = starts[i + 1]?.index ?? prose.length;
190
+ let content = prose.slice(contentStart, contentEnd);
191
+ // The newline before the next typed row is framing, while every earlier newline belongs to the
192
+ // reviewer's field. At EOF/anchored-review there is no framing newline to remove.
193
+ if (starts[i + 1] && content.endsWith("\n"))
194
+ content = content.slice(0, -1);
195
+ const deferred = String(start[1]).startsWith("deferred/");
196
+ const separatorAt = deferred ? content.indexOf(REVIEW_RATIONALE_SEPARATOR) : -1;
197
+ return separatorAt === -1
198
+ ? { label: start[1], note: content }
199
+ : {
200
+ label: start[1],
201
+ note: content.slice(0, separatorAt),
202
+ rationale: content.slice(separatorAt + REVIEW_RATIONALE_SEPARATOR.length),
203
+ };
204
+ });
150
205
  }
151
206
  /**
152
207
  * Structured findings for a BLOCKING review/judge gate result, parsed from the details the gate
@@ -163,24 +218,22 @@ function toFinding(cls, note, path, symbol) {
163
218
  export function structuredFindings(gate, details, _scopeFiles = []) {
164
219
  const lines = details.split("\n");
165
220
  const rows = [];
166
- const push = (cls, note, ownPath, fallbackSymbol = "") => {
221
+ const push = (cls, note, ownPath, fallbackSymbol = "", rationale) => {
167
222
  const own = canonicalPath(ownPath || PATH_RE.exec(note)?.[1] || "");
168
223
  const sym = identifierIn(note) || fallbackSymbol;
169
- rows.push(toFinding(cls, note, own, sym));
224
+ rows.push(toFinding(cls, note, own, sym, rationale));
170
225
  };
226
+ if (gate === "review") {
227
+ for (const finding of reviewDetailFindings(details)) {
228
+ push(`review:${finding.label}`, finding.note, "", "", finding.rationale);
229
+ }
230
+ }
171
231
  for (const line of lines) {
172
232
  const a = ANCHORED_RE.exec(line);
173
233
  if (a) {
174
234
  push(`${gate}:anchored`, a[3], a[1]);
175
235
  continue;
176
236
  }
177
- if (gate === "review") {
178
- const r = REVIEW_ROW_RE.exec(line);
179
- if (r) {
180
- push(`review:${r[1]}`, r[2], "");
181
- continue;
182
- }
183
- }
184
237
  if (gate === "acceptance") {
185
238
  const j = JUDGE_ROW_RE.exec(line);
186
239
  // the criterion id IS a stable symbol for an unmet acceptance criterion — the same criterion is
@@ -197,6 +250,29 @@ export function structuredFindings(gate, details, _scopeFiles = []) {
197
250
  }
198
251
  return rows;
199
252
  }
253
+ // v2.1.5 T2: the reviewer's DEFERRAL channel, kept structured. `classifyReviewFindings`
254
+ // (gates/review.ts) renders a deferred finding as `- [deferred/<severity>] <note> — rationale: …`.
255
+ // The parser above preserves multiline fields and separates rationale from identity; an older journal
256
+ // whose row holds only prose still degrades through the same parse rather than to nothing. A deferred
257
+ // finding is a concern the reviewer SAW and chose not to block on; it is not a concern that was fixed.
258
+ const DEFERRED_CLASS_RE = /^review:deferred\b/;
259
+ export function isDeferredFinding(finding) {
260
+ return DEFERRED_CLASS_RE.test(finding.class);
261
+ }
262
+ /**
263
+ * The findings a PASSING review DEFERRED — the rows a blocking-only projection drops on the floor.
264
+ * A passing review's details are prose; without this the deferral has no identity a later round can
265
+ * match, and every structured reader of the journal is blind to a defect the reviewer itself named.
266
+ */
267
+ export function deferredReviewFindings(details) {
268
+ return structuredFindings("review", details).filter(isDeferredFinding);
269
+ }
270
+ /** The exact review.ts details fragment represented by a structured review finding. */
271
+ export function renderStructuredReviewFinding(finding) {
272
+ const label = finding.class.startsWith("review:") ? finding.class.slice("review:".length) : finding.class;
273
+ const rationale = finding.rationale === undefined ? "" : `${REVIEW_RATIONALE_SEPARATOR}${finding.rationale}`;
274
+ return `- [${label}] ${finding.note}${rationale}`;
275
+ }
200
276
  const findingRows = (event, gate) => {
201
277
  if (Array.isArray(event.data.findings)) {
202
278
  const rows = event.data.findings.filter((finding) => {
@@ -205,6 +281,7 @@ const findingRows = (event, gate) => {
205
281
  const row = finding;
206
282
  return typeof row.class === "string" && typeof row.path === "string"
207
283
  && typeof row.symbol === "string" && typeof row.note === "string"
284
+ && (row.rationale === undefined || typeof row.rationale === "string")
208
285
  && typeof row.fingerprint === "string";
209
286
  });
210
287
  if (rows.length > 0)
@@ -509,6 +586,19 @@ export function journaledFailureBrief(events, taskId) {
509
586
  * finding was dropped at the exact moment the operator paid for another attempt to fix it. A review
510
587
  * that DECLINED (`skipped`) is not a verdict and neither adds nor retires — fail closed. Findings are
511
588
  * keyed by fingerprint, so a reviewer restating one across rounds carries it once, not once per round.
589
+ *
590
+ * v2.1.5 T2: a passing review settles the findings it BLOCKED on. It does not settle the ones it
591
+ * DEFERRED — those it saw, declined to block on, and recorded a rationale for, and nothing has fixed
592
+ * them. So a pass retires the blocking set and re-seats its own deferrals, and the two retirements
593
+ * stay distinguishable: the blocking finding is gone, the deferral travels on as accepted work.
594
+ *
595
+ * A deferral's bound is the SAME single release as a blocking finding's — the operator accepting the
596
+ * review gate itself (`GATE_SATISFIED_RELEASE` stamped `gate: "review"`), the one approval in which a
597
+ * human actually looked at what the reviewer waved through. It is deliberately NOT bounded by a round
598
+ * count or by a time window: both retire a finding by arithmetic nobody read, which is the silent drop
599
+ * this fold exists to refuse. Nor can it accumulate — a reviewer restating the same path/note round
600
+ * after round re-seats ONE fingerprint, and a revised rationale replaces the prior rationale on that
601
+ * row. N rounds of the same concern therefore carry the newest accepted explanation once, not N rows.
512
602
  */
513
603
  export function outstandingReviewFindings(events, taskId) {
514
604
  const open = new Map();
@@ -522,8 +612,18 @@ export function outstandingReviewFindings(events, taskId) {
522
612
  }
523
613
  if (e.event !== "gate-result" || e.data.gate !== "review" || e.data.skipped === true)
524
614
  continue;
525
- if (e.data.pass !== false)
526
- open.clear(); // a later review PASSED on this task: nothing is outstanding
615
+ if (e.data.pass !== false) {
616
+ // a later review PASSED on this task: every finding it BLOCKED on is settled …
617
+ for (const [key, finding] of open)
618
+ if (!isDeferredFinding(finding))
619
+ open.delete(key);
620
+ // … and no deferral is, whether or not this pass restated it. A pass is silent about a
621
+ // deferral it does not mention: the concern is unfixed either way, and the reviewer that
622
+ // waved it through is not the release that accepts it. Retiring on omission would drop it on
623
+ // the very next round — the same silent drop by a different door.
624
+ for (const finding of findingRows(e, "review").filter(isDeferredFinding))
625
+ open.set(finding.fingerprint, finding);
626
+ }
527
627
  else
528
628
  for (const finding of findingRows(e, "review"))
529
629
  open.set(finding.fingerprint, finding);
@@ -987,19 +1087,36 @@ export class Journal {
987
1087
  ? undefined
988
1088
  : DecisionEventSchema.parse({ ...eventOrDecision, ts: new Date().toISOString() });
989
1089
  const event = decisionRow?.event ?? eventOrDecision;
1090
+ const rowTaskId = decisionRow && "taskId" in decisionRow ? decisionRow.taskId : taskId;
990
1091
  const inputData = decisionRow?.data ?? data;
1092
+ // OBS-738: terminal and resume records reduce the journal that precedes them. Neither re-derives
1093
+ // recovery facts from task-human prose: preserved refs come from preservedRefsByTask, and the
1094
+ // upheld brief comes from the established prompt/replay reducer.
1095
+ const priorEvents = event === "run-end" || event === "resume-restore" ? this.read() : [];
1096
+ const reducedData = event === "run-end"
1097
+ ? (() => {
1098
+ const preservedRefs = [...preservedRefsByTask(priorEvents)].flatMap(([preservedTaskId, refs]) => refs.map(({ ref, diffCommand }) => ({ taskId: preservedTaskId, ref, diffCommand })));
1099
+ return preservedRefs.length > 0 ? { ...inputData, preservedRefs } : inputData;
1100
+ })()
1101
+ : event === "resume-restore" && rowTaskId && upheldFeedbackByTask(priorEvents).has(rowTaskId)
1102
+ ? {
1103
+ ...inputData,
1104
+ upheldFeedbackRestoredFor: rowTaskId,
1105
+ summary: `upheld feedback restored for ${rowTaskId}`,
1106
+ }
1107
+ : inputData;
991
1108
  const evidence = judgePersistence.getStore();
992
1109
  const failed = evidence?.invocations.filter((invocation) => invocation.transcript !== undefined) ?? [];
993
1110
  const persistedData = event === "judge-retry" && failed.length > 0
994
1111
  ? {
995
- ...inputData,
1112
+ ...reducedData,
996
1113
  transcript: failed[0].transcript,
997
1114
  ...(failed[1] ? { retryTranscript: failed[1].transcript } : {}),
998
1115
  }
999
- : inputData;
1000
- const row = decisionRow ?? {
1001
- ts: new Date().toISOString(), event, ...(taskId ? { taskId } : {}), data: persistedData,
1002
- };
1116
+ : reducedData;
1117
+ const row = decisionRow
1118
+ ? { ...decisionRow, data: persistedData }
1119
+ : { ts: new Date().toISOString(), event, ...(taskId ? { taskId } : {}), data: persistedData };
1003
1120
  // T3 secret redaction: only the persisted bytes are masked — the caller's data stays untouched in
1004
1121
  // memory. The narrator receives the persisted (masked) row so a pane sink never shows a credential.
1005
1122
  const line = redactSecrets(JSON.stringify(row));
package/dist/run/merge.js CHANGED
@@ -3,7 +3,7 @@ import { join } from "node:path";
3
3
  import { shq } from "../adapters/types.js";
4
4
  import { ceilingKillResult, classifyFailureOutput, effectiveCeilingMs, fingerprint, freshFailures, } from "../gates/baseline.js";
5
5
  import { tickmarkrDir } from "../graph/graph.js";
6
- import { gitHead, linkNodeModules, resolveIntegrationBranch, sh, shGit, shGitOk, WORKTREES_DIR } from "./git.js";
6
+ import { describeCapacity, gitHead, linkNodeModules, resolveIntegrationBranch, sameCapacity, sh, shGit, shGitOk, WORKTREES_DIR } from "./git.js";
7
7
  export function integrationBranch(cfg, runId) {
8
8
  return `${cfg.integrationBranchPrefix}${runId}`;
9
9
  }
@@ -99,7 +99,13 @@ export async function verifyIntegrationTip(intWt, commands, runDir, baseline) {
99
99
  // Battery parity on the infra rule too (T9): infrastructure-only output means the runner never
100
100
  // completed a suite — nothing was verified, so nothing is forgivable, however familiar its
101
101
  // fingerprints. Stricter-than-battery edge kept: unreadable output never forgives.
102
- const forgiven = r.code !== 0 && baselineRed && failing.length === 0 && !unreadable && cause !== "infra";
102
+ // T7: the SECOND reader of a baseline entry, and the same rule as the battery's (baseline.ts).
103
+ // Evidence crosses a session boundary here — the capture that would forgive this red is very
104
+ // often the previous session's — so a capture taken under a different resolved capacity forgives
105
+ // nothing at the tip either. An absent capacity is a pre-T7 baseline and keeps today's verdict.
106
+ const comparable = sameCapacity(entry?.capacity, r.capacity);
107
+ const forgiven = r.code !== 0 && baselineRed && failing.length === 0 && !unreadable && cause !== "infra"
108
+ && comparable;
103
109
  const pass = r.code === 0 || forgiven;
104
110
  if (!pass)
105
111
  writeFileSync(artifact, raw);
@@ -111,7 +117,11 @@ export async function verifyIntegrationTip(intWt, commands, runDir, baseline) {
111
117
  fingerprints: r.code !== 0 ? fingerprint(stripped) : [],
112
118
  details: r.code === 0 ? "exit 0"
113
119
  : forgiven ? `exit ${r.code} but only baseline-recorded failures (forgiven vs baseline)`
114
- : `exit ${r.code}`,
120
+ : !comparable && baselineRed && failing.length === 0
121
+ ? `exit ${r.code}; every failure is baseline-recorded, but that capture ran under `
122
+ + `${describeCapacity(entry?.capacity)} and this verification ran under ${describeCapacity(r.capacity)} `
123
+ + `— forgiveness across a changed capacity is not evidence`
124
+ : `exit ${r.code}`,
115
125
  ...(forgiven ? { forgiven: true } : {}),
116
126
  ...(cause ? { cause } : {}),
117
127
  ...(pass ? {} : { artifact }),
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "tickmarkr",
3
- "version": "2.1.4",
3
+ "version": "2.1.6",
4
4
  "description": "Spec in, verified work out.",
5
5
  "type": "module",
6
6
  "license": "MIT",
@@ -103,7 +103,7 @@ through brief lineage. **An executor choice nobody made is still an executor cho
103
103
  fraction (`ORCH · v1.19 4/5`, updated on every task-done); tickmarkr opens ONE TAB PER TASK, labelled
104
104
  with the task id and holding that task's worker plus its judge/review/consult panes (tickmarkr
105
105
  updates it). Never long context strings or ✓-chains.
106
- 2. **Orchestrator**: Launch the orchestrator with your agent host. Spawning on current herdr is two-step — the one-shot `agent start --cwd` form was removed in the herdr CLI redesign and now fails with `unknown option` (OBS-138): first create the pane with `herdr tab create --workspace <ws> --cwd <repo> --label "ORCH · <version>"` and parse `result.root_pane.pane_id` from its JSON, then start the agent in it. For Claude Code, use `herdr agent start orchestrator --kind claude --pane <root-pane-id> -- --permission-mode bypassPermissions` (append `--model <m>` after the `--` if the operator has a policy). For Codex, use `herdr agent start orchestrator --kind codex --pane <root-pane-id> -- --dangerously-bypass-approvals-and-sandbox` (add `--model <m>` to specify the model). The unsandboxed flag is REQUIRED: codex's `workspace-write` sandbox keeps `.git` refs read-only, so a sandboxed orchestrator's `tickmarkr run` dies at integration-branch creation — do not downgrade it. Workers you never spawn — tickmarkr spawns its own visible worker panes. Auxiliary agents you do spawn (consultants, reviewers, scouts) follow the same forms: never launch a claude session in plan mode or default permission mode for autonomous work — both stall on per-command approval prompts nobody is watching; claude is always `--permission-mode bypassPermissions --settings '{"promptSuggestionEnabled":false}'`, and a read-only codex consultant may use `--sandbox read-only`. **That `--settings` pair is not cosmetic and it is not optional:** claude-code's AUTOSUGGEST renders context-plausible ghost text into an idle seat's prompt line that is BYTE-IDENTICAL to a typed draft in text-format reads (OBS-482), so a supervising tier cannot tell a seat's own unsent work from a rendering artifact without `agent read --format ansi`. Turning the suggester off at spawn removes the ambiguity at its source instead of paying for the discrimination at every read. Verified against the shipped binary: `claude --settings '{"promptSuggestionEnabled":false}' -p …` exits 0 with a real response, and the key appears in the binary's own settings schema. **For kimi, pass `-y`** (`herdr agent start <name> --kind kimi --pane <id> -- -y`) — the adapter already launches its own workers that way (`src/adapters/kimi.ts:204`), and a kimi seat spawned without it sits on an approval prompt having done nothing. **Herdr cannot see that state**: it reports a kimi pane as `agent_status: working` with `screen_detection_skipped: true` while the prompt is up, so the BLOCKED-STATE watcher below is blind on this vendor and the spawn flag is the ONLY control. Every vendor you spawn needs its auto-approve form named here; a vendor absent from this list is a seat that will hang.
106
+ 2. **Orchestrator**: Launch the orchestrator with your agent host. Spawning on current herdr is two-step — the one-shot `agent start --cwd` form was removed in the herdr CLI redesign and now fails with `unknown option` (OBS-138): first create the pane with `herdr tab create --workspace <ws> --cwd <repo> --label "ORCH · <version>"` and parse `result.root_pane.pane_id` from its JSON, then start the agent in it. For Claude Code, use `herdr agent start orchestrator --kind claude --pane <root-pane-id> -- --permission-mode bypassPermissions` (append `--model <m>` after the `--` if the operator has a policy). For Codex, use `herdr agent start orchestrator --kind codex --pane <root-pane-id> -- --dangerously-bypass-approvals-and-sandbox` (add `--model <m>` to specify the model). The unsandboxed flag is REQUIRED: codex's `workspace-write` sandbox keeps `.git` refs read-only, so a sandboxed orchestrator's `tickmarkr run` dies at integration-branch creation — do not downgrade it. Workers you never spawn — tickmarkr spawns its own visible worker panes. Auxiliary agents you do spawn (consultants, reviewers, scouts) follow the same forms: never launch a claude session in plan mode or default permission mode for autonomous work — both stall on per-command approval prompts nobody is watching; claude is always `--permission-mode bypassPermissions --settings '{"promptSuggestionEnabled":false}'`. **For a codex consultant, use `-a never --sandbox workspace-write` — NOT `--sandbox read-only`.** ⚠ **`--sandbox read-only` CONTRADICTS this skill's own completion protocol and will hang the seat.** Every seat you spawn is told to deliver an ARTIFACT ending in a terminal MARKER, because that is the only completion signal the artifact watcher can key on (`done` is turn end). A read-only sandbox cannot write that artifact, so codex blocks on `Would you like to make the following edits?` for its OWN report — and the report exists ONLY in the pending edit, so abandoning the prompt destroys the work rather than merely delaying it. Measured 2026-08-28: a consultant spawned `--sandbox read-only` finished a 14,604-byte verdict, sat blocked on the write, and the operator saw the prompt before the supervising tier did. `read-only` is correct ONLY for a seat that writes nothing at all — which, under the artifact+marker rule, is no seat this skill tells you to spawn. When the prompt does appear, answer **"Yes, and don't ask again for these files"** rather than plain yes: plain yes re-blocks on the next write of the same file. **That `--settings` pair is not cosmetic and it is not optional:** claude-code's AUTOSUGGEST renders context-plausible ghost text into an idle seat's prompt line that is BYTE-IDENTICAL to a typed draft in text-format reads (OBS-482), so a supervising tier cannot tell a seat's own unsent work from a rendering artifact without `agent read --format ansi`. Turning the suggester off at spawn removes the ambiguity at its source instead of paying for the discrimination at every read. Verified against the shipped binary: `claude --settings '{"promptSuggestionEnabled":false}' -p …` exits 0 with a real response, and the key appears in the binary's own settings schema. **For kimi, pass `-y`** (`herdr agent start <name> --kind kimi --pane <id> -- -y`) — the adapter already launches its own workers that way (`src/adapters/kimi.ts:204`), and a kimi seat spawned without it sits on an approval prompt having done nothing. **Herdr cannot see that state**: it reports a kimi pane as `agent_status: working` with `screen_detection_skipped: true` while the prompt is up, so the BLOCKED-STATE watcher below is blind on this vendor and the spawn flag is the ONLY control. Every vendor you spawn needs its auto-approve form named here; a vendor absent from this list is a seat that will hang.
107
107
  3. **Standing instructions travel as a brief FILE, never as pane text** — PTY input truncates at ~1024B and a
108
108
  truncated brief silently drops policy. Write the full brief to `<repo>/.tickmarkr/overseer/ORCH-BRIEF.md`
109
109
  (inside the tickmarkr state dir — already self-gitignored, no exclude step needed), then send one line:
@@ -249,6 +249,111 @@ detect: **the ruling would have made the worker commit the violation the task wa
249
249
 
250
250
  ### Context is a supervised resource, for BOTH tiers
251
251
 
252
+ **THE TIERS CLEAR EACH OTHER AT 50%. Neither tier clears itself on its own notice.** Operator directive,
253
+ 2026-08-28, and it exists because **a seat cannot reliably observe its own exhaustion** — the seat that
254
+ most needs clearing is the one least able to notice, and this project has now measured that three ways:
255
+ an overseer ran nine hours at 86% unable to read its own number; a context watcher went **alive and blind**
256
+ when the run's own status text pushed the percentage off the statusline; and an orchestrator went
257
+ **366k → 970k of 1M between two checks** while its ACT wake sat unread in a detached log.
258
+
259
+ The protocol, in both directions:
260
+
261
+ 1. **Overseer sees orch at ≥50%** → nudge it: write `HANDOFF-ORCH-<ver>.md`, then `/clear`, then re-read
262
+ its brief **and** its handoff, then **re-arm every watcher it listed** (a cleared session has none).
263
+ 2. **The returning orch, now fresh, checks the OVERSEER.** If the overseer is at ≥50%, it directs the
264
+ overseer to write its handoff and clear, and **points it at `HANDOFF-OVERSEER-<ver>.md` by path**.
265
+ 3. Whichever seat is fresh performs the check. **Never both at once** — the run keeps one supervising tier
266
+ at all times, and the seat holding the endgame goes second.
267
+ 4. **The duty to clear the other tier must SURVIVE a clear**, so it belongs in BOTH handoff files as a
268
+ standing re-arm item — not only in the message that ordered it. Earned 2026-08-28: the orchestrator
269
+ performed the check on the overseer BEFORE its own clear, precisely because clearing first would have
270
+ wiped the instruction to do it, and its handoff did not record the duty.
271
+ 5. **A seat cannot `/clear` ITSELF — so the OTHER TIER SENDS IT.** `/clear` is a CLI command typed into a
272
+ session and no tool invokes it in your own pane, but it is just text in someone else's: the partner
273
+ tier types it into your pane. **This is the whole reason the protocol is mutual**, and it means the
274
+ loop closes without the operator. The exchange, both directions, in this exact order:
275
+
276
+ ```bash
277
+ # 1. the seat crossing 50% writes its handoff FIRST, ending with its terminal marker
278
+ # 2. it asks the partner, naming its own pane and handoff path:
279
+ herdr pane run <partner> "I am at <N>%. Clear me: send /clear to <my-pane>, then point me at <my-handoff>."
280
+ # 3. the PARTNER sends the clear, then VERIFIES before pointing:
281
+ herdr pane run <my-pane> "/clear"
282
+ # read the pane back — a cleared claude session shows an empty prompt and a reset context gauge
283
+ # 4. and only THEN, as a SEPARATE send, the re-orientation:
284
+ herdr pane run <my-pane> "You were cleared at <N>%. Read <handoff> and <brief>, re-arm EVERY watcher
285
+ they name — a cleared session has none — then confirm you are back."
286
+ ```
287
+
288
+ ⚠ **Steps 3 and 4 are two sends, never one.** A pointer batched with the clear lands *during* it and is
289
+ lost with the context it was meant to survive. Verify the clear landed by reading the prompt line
290
+ before sending the pointer — the same read-back every other send in this skill requires.
291
+ ⚠ **The partner must not clear itself in the same window.** One supervising tier stays live at all
292
+ times; the seat holding the endgame goes second.
293
+ ⚠ **Step 4 must state an EXPECTED-RETURN DEADLINE**, e.g. *"confirm you are back within 10 minutes."*
294
+ A clear order without one is an unbounded wait: see rule 6.
295
+
296
+ 6. **THE RETURN LEG — the returning seat's FIRST act after re-arming is a verified notice to its partner.**
297
+ Not its second, not once the next milestone lands. **Measured 2026-08-28 (OBS-743):** an overseer cleared
298
+ at 00:05Z, was back at 00:07Z, *read the orchestrator's pane at 00:10Z to take its percentage* — and said
299
+ nothing. The orchestrator's last line had been *"T7 is mine until you're back."* It then held the task
300
+ alone for **53 minutes** across a reviewer flake, a retry, a merge and the whole tip verify, with no
301
+ signal that its supervising tier existed. The notice went out only because the operator noticed the gap.
302
+
303
+ **A one-way read is not a handshake.** The adopt step tells you to READ the partner's pane, which feels
304
+ like contact and transmits nothing — that is exactly how this gets skipped by a seat following the
305
+ protocol correctly.
306
+
307
+ The notice is ONE line (a newline submits early), sent with `herdr pane run` and **read back**, and it
308
+ carries four things:
309
+ ```bash
310
+ herdr pane run <partner> "RETURN NOTICE <seat>: back on <my-pane> since <HH:MM>Z. WATCHERS I NOW HOLD:
311
+ <list>. SWEPT: <what was dead>. MISSION STATE AS I READ IT: <one clause>. Reply with every watcher YOU
312
+ still hold so we deconflict — do not arm anything I just named."
313
+ ```
314
+ - **where and since when**, so the partner can stop holding your duties;
315
+ - **the watcher inventory you now hold** — coverage is the thing both tiers silently assume about each
316
+ other, and a returning seat that re-arms without saying so produces double-coverage that reads as
317
+ redundancy and is actually two tiers each trusting the other;
318
+ - **anything you swept**, because a dead watcher the partner armed is *its* belief about coverage, not
319
+ yours, and it will keep believing it;
320
+ - **an explicit deconfliction request.** Ask for the partner's inventory back; do not infer it.
321
+
322
+ ⚠ **RE-ADOPT EVERY DETACHED WATCHER ON RETURN, and prove it from disk.** A detached watcher (`ppid 1`)
323
+ is the one kind that SURVIVES your clear, which is exactly why it rots unattended: `stat` its heartbeat
324
+ and treat **stale as dead**. Measured the same morning (OBS-742): a detached resolution watcher's
325
+ heartbeat was **157 minutes stale** and the task it existed to report had resolved **2h04m after its
326
+ last beat** — present, silent, and indistinguishable from healthy to anyone who did not look. Sweep it,
327
+ archive the heartbeat rather than deleting it so the gap stays measurable, and name it in the notice.
328
+
329
+ ⚠ **A CLEAR AND A DEATH ARE THE SAME SILENCE.** A seat that clears and never returns — wrong pane,
330
+ crashed host, operator closed the tab — is indistinguishable from one mid-`/clear`. That is why step 4
331
+ states a deadline: **partner silent past it → escalate to the operator.** Without the deadline the
332
+ protocol's most dangerous state has no timeout, and the surviving tier waits forever on a peer that no
333
+ longer exists.
334
+
335
+ ⚠ **`∑ NNNk tok` ON A CLAUDE STATUSLINE IS CUMULATIVE SESSION SPEND, NOT CONTEXT FILL.** The percentage
336
+ is the fill; the token total is what has been spent across every turn and keeps climbing after a compaction
337
+ or a clear. **Measured 2026-08-28, expensively:** this seat built a fallback watcher on `∑ Nk tok`, read
338
+ `970k` as 97% of a 1M window, and sent an urgent clear-order to an orchestrator that was actually at
339
+ **33%** — which then began a handoff and offered to discard a session two-thirds fresh, mid-endgame.
340
+ **Read the `%`. Never derive fill from the token total, and never build an instrument on a signal whose
341
+ SEMANTICS you have not verified against a second source.**
342
+
343
+ **Why 50% and not 85%:** a handoff written at 50% is written by a seat whose judgment is intact. One
344
+ written at 85% is written by a seat already degraded, about the decisions it is least able to summarise.
345
+ The threshold buys judgment, not headroom.
346
+
347
+ **Why the OTHER tier issues it:** a self-issued clear competes with whatever the seat is doing and loses.
348
+ An instruction from the other tier arrives as work, and the tier issuing it is not the tier that has to
349
+ overcome its own momentum to obey.
350
+
351
+ ⚠ **A detached watcher gives COVERAGE and takes away NOTIFICATION.** A wake written to a log file that no
352
+ seat reads is not a wake. If the watcher must outlive a turn, it also needs a path that reaches a seat —
353
+ a beat the other tier reads, a notification, or an artifact the other tier watches. **Measured 2026-08-28:
354
+ `ACT: orch-215 at 970k/1000k — handoff + /clear NOW` fired correctly and sat unread in a scratchpad log
355
+ while the orchestrator kept working.**
356
+
252
357
  Arm a context watcher on the orchestrator at spawn time and treat a threshold wake as a first-class event:
253
358
  finish the step, write a handoff, `/clear` **plus a fresh brief — never `/compact`**, because a compaction
254
359
  is a lossy summary nobody trusts while a clean session re-oriented from disk-verifiable state is reliable.
@@ -257,8 +362,9 @@ good, not after. If your own context cannot be read by the watcher, say so to th
257
362
  number — an unmeasured budget is not a small budget.
258
363
 
259
364
  ```bash
260
- .claude/skills/tickmarkr-overseer/scripts/watch-context.sh orchestrator <orchestrator-agent-or-pane> 60 75 <handoff-file>
261
- .claude/skills/tickmarkr-overseer/scripts/watch-context.sh overseer <overseer-agent-or-pane> 60 75 <handoff-file>
365
+ # WARN 50 / ACT 50 — the mutual-clear threshold above, not a headroom alarm.
366
+ .claude/skills/tickmarkr-overseer/scripts/watch-context.sh orchestrator <orchestrator-agent-or-pane> 50 50 <handoff-file>
367
+ .claude/skills/tickmarkr-overseer/scripts/watch-context.sh overseer <overseer-agent-or-pane> 50 50 <handoff-file>
262
368
  ```
263
369
 
264
370
  The first argument chooses the closed per-seat tier (`orchestrator-context` or `overseer-context`),
@@ -947,6 +1053,17 @@ twice.** They are mission-independent on purpose: nothing here names a task, a l
947
1053
  concluded both forms were valid. The tell is unavailable unless the tool volunteers it. Corollary —
948
1054
  an instrument that takes an input must be handed a DELIBERATELY BAD one before its clean runs are
949
1055
  worth anything (rule 11 applied to tools, not just to gates).
1056
+ ⚠ **AND THE COMMONEST WRONG INPUT IS A BASE REF: after the first merge, a task's diff against the
1057
+ run's `baseRef` is NEVER that task's diff.** Workers branch from the INTEGRATION TIP, so once any task
1058
+ has merged, `git diff baseRef..HEAD` in a later worktree reports that task PLUS every task merged
1059
+ before it, and the number looks entirely plausible. Diff from the task's OWN base — the integration
1060
+ commit it branched from — and say which base you used whenever you quote a size.
1061
+ **Measured 2026-08-28 in one run, twice, in both directions.** A supervising seat quoted "712
1062
+ insertions across 7 files" for a task whose real contribution was **300 across 2**; the surplus was two
1063
+ other tasks' merged work. On the next task the same trap was **larger** — 920 across 11 versus a true
1064
+ 167 across 4 — and it was caught only because the other tier had just been burned by it. A scope
1065
+ judgement, a cost claim, or a review-size argument built on the baseRef diff is measuring three tasks
1066
+ and calling it one.
950
1067
  14. **A unit is not a measurement.** A configured timeout is a KILL CEILING, not a duration — never compare
951
1068
  it to a wall clock or quote it to an operator as an estimate.
952
1069
  15. **Verify through the path that LOADS, not the path you edited.** Mirrored trees and symlinks mean your