tickmarkr 1.84.0 → 1.85.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -25,9 +25,61 @@ export declare const GATE_SATISFIED_RELEASE: "gate-satisfied";
25
25
  export declare const REVIEW_UPHELD_RELEASE: "review-upheld";
26
26
  export declare const RECHECK_RELEASE: "recheck";
27
27
  export declare function reviewRoundsSinceApproval(events: JournalEvent[], taskId: string): number;
28
+ export declare function upheldFeedbackByTask(events: JournalEvent[]): Map<string, string>;
29
+ export interface StructuredFinding {
30
+ class: string;
31
+ path: string;
32
+ symbol: string;
33
+ note: string;
34
+ fingerprint: string;
35
+ }
36
+ export declare const UNIDENTIFIED = "<unidentified>";
37
+ /**
38
+ * Structured findings for a BLOCKING review/judge gate result, parsed from the details the gate
39
+ * already writes (D-03: no gate-module change, so an older gate's prose degrades to one unclassified
40
+ * finding rather than to none). Never empty for a blocking result — a finding the journal cannot
41
+ * classify is still a finding the next retry must not lose.
42
+ *
43
+ * Rule: a finding's path is its verdict row's own evidence path. An inline path and an anchored row's
44
+ * path therefore resolve; a different anchor or the task's declared scope never substitutes for a
45
+ * pathless finding. That row fails closed as UNIDENTIFIED instead of manufacturing an R4 identity.
46
+ * A symbol the row's own prose does not name falls back to the criterion id, then to the row's own
47
+ * normalized words (see toFinding).
48
+ */
49
+ export declare function structuredFindings(gate: string, details: string, _scopeFiles?: string[]): StructuredFinding[];
50
+ /** Normalized identity of a gate failure: the same defect, seen twice, normalizes to the same bytes. */
51
+ export declare function normalizeGateFailure(details: string): string;
52
+ export declare const GATE_FINGERPRINT_CAP = 2;
53
+ export declare function identicalGateFailures(events: JournalEvent[], taskId: string, gate: string, normalized: string): number;
54
+ /** Repair attempts this engagement has already funded — journal-derived, so a resume inherits it. */
55
+ export declare function repairsSinceApproval(events: JournalEvent[], taskId: string): number;
56
+ /**
57
+ * Why the last attempt failed, one row per journaled cause, in the daemon's own `source: details`
58
+ * shape. The daemon builds that brief in a loop-local variable, which dies with the process: a resumed
59
+ * or `--retry-failed` run rebuilt the prompt from nothing and dispatched a retry that had lost the
60
+ * reason it was retrying — OBS-254's class, one layer below the upheld brief. Re-derived here so the
61
+ * bytes the journal already holds cannot be taken away by any reset of attempt or channel state.
62
+ *
63
+ * The same rule governs a dead DISPATCH: its exact task-failed error is retained until
64
+ * `worker-launch`, never retired at
65
+ * `task-dispatch`: everything between the two — worktree recreation, setup, prompt write, slot
66
+ * allocation, the launch itself — can still die with no worker having read a word, and clearing at
67
+ * task-dispatch meant `--retry-failed` after exactly that death rebuilt the prompt without the gate
68
+ * failures OR the delivery failure that preceded it. `task-approved` also clears (an operator approval
69
+ * retires the findings it settled — the uphold case re-derives its own brief separately).
70
+ */
71
+ export declare function journaledFailureBrief(events: JournalEvent[], taskId: string): string[];
72
+ /** The findings a funded repair must carry into the next dispatch, or undefined if none is pending. */
73
+ export declare function pendingRepairFindings(events: JournalEvent[], taskId: string): string | undefined;
74
+ /**
75
+ * The gate whose identical failure banned an identical retry of the NEXT dispatch — bound to the
76
+ * channel that produced it, so a verdict that has already moved the work elsewhere is not refused for
77
+ * a channel it is no longer using, and a later unrelated failure is not parked under a stale reason.
78
+ */
79
+ export declare function activeRetryBan(events: JournalEvent[], taskId: string, channel: string): string | undefined;
28
80
  export declare const PARK_KINDS: readonly ["human-gate", "ladder-exhausted", "attempt-cap", "gate-fail", "quota", "reroute-exhausted", "setup", "stall", "merge-conflict", "tip-moved", "infra", "dispatch"];
29
81
  export type ParkKind = (typeof PARK_KINDS)[number];
30
- export declare const RETRY_MODES: readonly ["resume", "fresh"];
82
+ export declare const RETRY_MODES: readonly ["resume", "fresh", "repair"];
31
83
  export type RetryMode = (typeof RETRY_MODES)[number];
32
84
  export declare const WORKER_RESULT_CAUSES: readonly ["provider-death", "stall-timeout", "malformed-trailer", "clean-exit-no-trailer"];
33
85
  export type WorkerResultCause = (typeof WORKER_RESULT_CAUSES)[number];
@@ -66,13 +118,13 @@ export declare const TelemetryRowSchema: z.ZodObject<{
66
118
  "attempt-cap": "attempt-cap";
67
119
  "gate-fail": "gate-fail";
68
120
  quota: "quota";
121
+ dispatch: "dispatch";
69
122
  "human-gate": "human-gate";
70
123
  "reroute-exhausted": "reroute-exhausted";
71
124
  stall: "stall";
72
125
  "merge-conflict": "merge-conflict";
73
126
  "tip-moved": "tip-moved";
74
127
  infra: "infra";
75
- dispatch: "dispatch";
76
128
  }>>;
77
129
  tokens: z.ZodCatch<z.ZodOptional<z.ZodObject<{
78
130
  input: z.ZodNumber;
@@ -87,13 +139,14 @@ export declare const TelemetryRowSchema: z.ZodObject<{
87
139
  retryMode: z.ZodOptional<z.ZodEnum<{
88
140
  resume: "resume";
89
141
  fresh: "fresh";
142
+ repair: "repair";
90
143
  }>>;
91
144
  signalQuality: z.ZodOptional<z.ZodUnion<readonly [z.ZodLiteral<0>, z.ZodLiteral<0.25>, z.ZodLiteral<0.5>, z.ZodLiteral<0.75>, z.ZodLiteral<1>]>>;
92
145
  signalBasis: z.ZodOptional<z.ZodEnum<{
146
+ "judge-only": "judge-only";
93
147
  skipped: "skipped";
94
148
  proved: "proved";
95
149
  "review-agree": "review-agree";
96
- "judge-only": "judge-only";
97
150
  legacy: "legacy";
98
151
  vacuous: "vacuous";
99
152
  }>>;
@@ -70,6 +70,277 @@ export function reviewRoundsSinceApproval(events, taskId) {
70
70
  }
71
71
  return rounds;
72
72
  }
73
+ // OBS-189/OBS-254: the uphold brief is the operator's funded decision, not attempt state. ONE fold,
74
+ // two consumers — replayResumeState seeds it, and the daemon re-derives it from the journal at
75
+ // prompt-build time so no reset of attempt/channel state can take the findings with it (OBS-254 deleted
76
+ // the whole resume entry and dispatched a funded attempt with an empty "fix these specifically" heading).
77
+ export function upheldFeedbackByTask(events) {
78
+ const upheld = new Map();
79
+ const lastReviewFail = new Map(); // newest failed review details per task
80
+ for (const e of events) {
81
+ if (!e.taskId)
82
+ continue;
83
+ if (e.event === "gate-result" && e.data.gate === "review" && e.data.pass === false
84
+ && typeof e.data.details === "string") {
85
+ lastReviewFail.set(e.taskId, e.data.details);
86
+ }
87
+ else if (e.event === "task-approved") {
88
+ // any later approval supersedes: a plain accept-the-diff approval retires the uphold brief.
89
+ if (e.data.release === REVIEW_UPHELD_RELEASE) {
90
+ const details = lastReviewFail.get(e.taskId);
91
+ if (details)
92
+ upheld.set(e.taskId, details);
93
+ else
94
+ upheld.delete(e.taskId);
95
+ }
96
+ else {
97
+ upheld.delete(e.taskId);
98
+ }
99
+ }
100
+ }
101
+ return upheld;
102
+ }
103
+ // Reserved for a finding whose OWN evidence names no path. Reporting a blank path a reader would take
104
+ // for a resolved one is the silent-lie shape the gates exist to refuse, so the field says so outright.
105
+ export const UNIDENTIFIED = "<unidentified>";
106
+ const ANCHORED_RE = /^- (\S+?):(\d+) — (.*)$/; // "## Anchored review" rows (llm.ts)
107
+ const REVIEW_ROW_RE = /^- \[([^\]]+)\] (.*)$/; // "- [material] …" (review.ts)
108
+ const JUDGE_ROW_RE = /^✗ ([\w.-]+): (.*)$/; // "✗ c1: …" (acceptance.ts) — id, then reason
109
+ const PATH_RE = /\b((?:[\w.@~+-]+\/)+[\w.@~+-]+\.\w{1,6})\b/;
110
+ const LINE_REF_RE = /(:\d+(?::\d+)?\b)|(\bline \d+\b)/gi;
111
+ // ponytail: repo-relative tail from the first known top-level directory — enough to make an absolute
112
+ // worktree path and its repo-relative twin the same identity. Widen the marker list if a run ever
113
+ // names findings outside these roots.
114
+ function canonicalPath(raw) {
115
+ const cleaned = raw.replace(/^["'`(]+/, "").replace(/["'`),.]+$/, "").replace(/^\.\//, "");
116
+ const m = /(?:^|\/)((?:src|tests|scripts|docs|fixtures|specs|schema|skills|assets)\/.+)$/.exec(cleaned);
117
+ return m ? m[1] : cleaned;
118
+ }
119
+ // The code identity a finding names, if it names one: a backticked identifier, then a call/member
120
+ // expression. Line references are stripped first so no identity can carry one. "" means the prose
121
+ // named no symbol — the caller decides what stands in, rather than this guessing from prose.
122
+ function identifierIn(note) {
123
+ const text = note.replace(LINE_REF_RE, " ");
124
+ const ticked = /`([^`]{1,80})`/.exec(text);
125
+ if (ticked)
126
+ return ticked[1].trim();
127
+ // no whitespace before the paren: "the brief (see …)" is prose, not a call expression, and a prose
128
+ // word standing in for a symbol is the guessing this function exists to refuse.
129
+ const call = /\b([A-Za-z_$][\w$]*(?:\.[A-Za-z_$][\w$]*)*)\(/.exec(text);
130
+ return call ? call[1] : "";
131
+ }
132
+ // The SYMBOL of last resort. R4 admits "stable symbol/test title" — a finding whose prose names no
133
+ // code identity still has one stable identity of its own: its own words, with the volatile tokens
134
+ // swept out so line/path churn cannot mint a new symbol for the same finding. It is the reviewer's
135
+ // own bytes, never a guess, and it can never fuse two different findings into one.
136
+ function toFinding(cls, note, path, symbol) {
137
+ const p = path || UNIDENTIFIED;
138
+ const s = symbol || normalizeGateFailure(note) || UNIDENTIFIED;
139
+ return { class: cls, path: p, symbol: s, note, fingerprint: `${cls}|${p}|${s}` };
140
+ }
141
+ /**
142
+ * Structured findings for a BLOCKING review/judge gate result, parsed from the details the gate
143
+ * already writes (D-03: no gate-module change, so an older gate's prose degrades to one unclassified
144
+ * finding rather than to none). Never empty for a blocking result — a finding the journal cannot
145
+ * classify is still a finding the next retry must not lose.
146
+ *
147
+ * Rule: a finding's path is its verdict row's own evidence path. An inline path and an anchored row's
148
+ * path therefore resolve; a different anchor or the task's declared scope never substitutes for a
149
+ * pathless finding. That row fails closed as UNIDENTIFIED instead of manufacturing an R4 identity.
150
+ * A symbol the row's own prose does not name falls back to the criterion id, then to the row's own
151
+ * normalized words (see toFinding).
152
+ */
153
+ export function structuredFindings(gate, details, _scopeFiles = []) {
154
+ const lines = details.split("\n");
155
+ const rows = [];
156
+ const push = (cls, note, ownPath, fallbackSymbol = "") => {
157
+ const own = canonicalPath(ownPath || PATH_RE.exec(note)?.[1] || "");
158
+ const sym = identifierIn(note) || fallbackSymbol;
159
+ rows.push(toFinding(cls, note, own, sym));
160
+ };
161
+ for (const line of lines) {
162
+ const a = ANCHORED_RE.exec(line);
163
+ if (a) {
164
+ push(`${gate}:anchored`, a[3], a[1]);
165
+ continue;
166
+ }
167
+ if (gate === "review") {
168
+ const r = REVIEW_ROW_RE.exec(line);
169
+ if (r) {
170
+ push(`review:${r[1]}`, r[2], "");
171
+ continue;
172
+ }
173
+ }
174
+ if (gate === "acceptance") {
175
+ const j = JUDGE_ROW_RE.exec(line);
176
+ // the criterion id IS a stable symbol for an unmet acceptance criterion — the same criterion is
177
+ // the same finding however the judge rephrases its reason — so it backs the prose-derived one.
178
+ if (j) {
179
+ push("acceptance:unmet", j[2], "", j[1]);
180
+ continue;
181
+ }
182
+ }
183
+ }
184
+ if (rows.length === 0) {
185
+ const head = lines.map((l) => l.trim()).find(Boolean) ?? "";
186
+ push(`${gate}:unclassified`, head, "");
187
+ }
188
+ return rows;
189
+ }
190
+ // v1.85 T3: volatile tokens carry no information about WHY a gate failed — ~663m across 5 runs went to
191
+ // re-dispatching against failures that differed only in these. Every rule below erases a token PROVEN
192
+ // to be a diagnostic location or a clock reading; nothing erases a value the failure asserts ABOUT.
193
+ // Ordered: styling, then timestamps (they contain colon-digits), then paths (they end before a :line),
194
+ // then line refs, durations, long hex.
195
+ const VOLATILE_TOKENS = [
196
+ [/\u001b\[[0-9;]*[a-zA-Z]/g, ""], // ANSI styling
197
+ [/\b\d{4}-\d{2}-\d{2}[T ][\d:]+(?:\.\d+)?(?:Z|[+-]\d{2}:?\d{2})?\b/g, "<ts>"], // timestamps
198
+ [/\brun-\d{8}-\d{6}\b/g, "<run>"], // run identifiers
199
+ [/\b0x[0-9a-fA-F]+\b/g, "<addr>"], // memory addresses
200
+ // An absolute path INTO the repo keeps its repo-relative tail — that tail IS identity (a defect in
201
+ // daemon.ts is not a defect in journal.ts); only the machine/worktree prefix ahead of it is volatile.
202
+ [/\/(?:[\w.@~+%-]+\/)*((?:src|tests|scripts|docs|fixtures|specs|schema|skills|assets)\/[\w.@~+%/-]+)/g, "<path>/$1"],
203
+ // Rule: an absolute diagnostic path's machine/worktree prefix is volatile, but its named file is
204
+ // identity. Therefore paths outside the repo-marker set keep their final segment: two machines
205
+ // naming parse.js normalize together, while parse.js and render.js can never spend one another's
206
+ // retry budget. A path-shaped VALUE ("/api/v1/users") is rooted nowhere real and survives.
207
+ [/\/(?:tmp|private|var|Users|home|opt|workspace|w)(?:\/[\w.@~+%-]+)*\/([\w.@~+%-]+)\/?/g, "<path>/$1"],
208
+ // A line[:col] ref counts as one only when it hangs off a file-ish token (a dot or a slash in it):
209
+ // R4 says the line number is evidence, not identity. "exit 1" and "expected 3" are neither.
210
+ [/([\w.@~+%-]*[./][\w.@~+%-]*):\d+(?::\d+)?\b/g, "$1:<line>"],
211
+ [/\bline \d+\b/gi, "line <line>"],
212
+ [/\b\d+(?:[.,]\d+)?\s?(?:ms|µs|us|ns|s|sec|secs|m|min|mins|h|hrs)\b/g, "<dur>"], // durations
213
+ [/\b[0-9a-f]{12,40}\b/g, "<hex>"], // sha / worktree ids
214
+ ];
215
+ // Rule: a quoted span is protected IFF it is assertion payload. Quoting alone is ordinary diagnostic
216
+ // rendering, so paths/timestamps inside ENOENT and worker messages still normalize. A value introduced
217
+ // by an assertion cue is payload whether quoted or bare: `expected /tmp/actual-a to be /tmp/want-a`
218
+ // must not collapse with an assertion about actual-b.
219
+ //
220
+ // The asymmetry is deliberate: a missed cap costs one extra round, a false cap bans a legitimate retry.
221
+ const ASSERTION_CUE = "expected|received|actual|got|to be|to equal|to match|to contain|instead of|but was|but got|but received";
222
+ const PAYLOAD_SPAN = new RegExp(`(?<=\\b(?:${ASSERTION_CUE})[:=]?[ \\t])(?:'[^'\\n]*'|"[^"\\n]*"|\`[^\`\\n]*\`|[^\\s,;)]+)`, "gi");
223
+ const eraseVolatile = (text) => VOLATILE_TOKENS.reduce((out, [re, replacement]) => out.replace(re, replacement), text);
224
+ // Whitespace RUNS are rendering, so they collapse — but only outside a payload, exactly like every
225
+ // other rule here. Inside one it is part of what the failure asserts: `expected "a b"` and
226
+ // `expected "a b"` are two different assertions, and collapsing the joined string erased that
227
+ // difference and banned a retry that was never redundant. Newlines survive (payload spans cannot
228
+ // cross one) and the line-wise trim below finishes the job.
229
+ const collapseRuns = (text) => text.replace(/[^\S\n]+/g, " ");
230
+ /** Normalized identity of a gate failure: the same defect, seen twice, normalizes to the same bytes. */
231
+ export function normalizeGateFailure(details) {
232
+ let out = "";
233
+ let last = 0;
234
+ for (const m of details.matchAll(PAYLOAD_SPAN)) {
235
+ out += collapseRuns(eraseVolatile(details.slice(last, m.index))) + m[0];
236
+ last = m.index + m[0].length;
237
+ }
238
+ out += collapseRuns(eraseVolatile(details.slice(last)));
239
+ return out.split("\n").map((l) => l.trim()).filter(Boolean).join("\n");
240
+ }
241
+ // Two normalized-identical failures of one gate on one task buy no more rounds (the ladder cannot fix
242
+ // what it already re-ran verbatim). Engagement-scoped exactly like reviewRoundsSinceApproval: an
243
+ // operator approval is a new engagement, and nothing else resets the count.
244
+ export const GATE_FINGERPRINT_CAP = 2;
245
+ export function identicalGateFailures(events, taskId, gate, normalized) {
246
+ let n = 0;
247
+ for (const e of events) {
248
+ if (e.taskId !== taskId)
249
+ continue;
250
+ if (e.event === "task-approved")
251
+ n = 0;
252
+ else if (e.event === "gate-result" && e.data.gate === gate && e.data.pass === false
253
+ && typeof e.data.details === "string"
254
+ && normalizeGateFailure(e.data.details) === normalized)
255
+ n++;
256
+ }
257
+ return n;
258
+ }
259
+ /** Repair attempts this engagement has already funded — journal-derived, so a resume inherits it. */
260
+ export function repairsSinceApproval(events, taskId) {
261
+ let n = 0;
262
+ for (const e of events) {
263
+ if (e.taskId !== taskId)
264
+ continue;
265
+ if (e.event === "task-approved")
266
+ n = 0;
267
+ else if (e.event === "repair-attempt")
268
+ n++;
269
+ }
270
+ return n;
271
+ }
272
+ // Both retry decisions below govern exactly ONE dispatch: the next one. So both are read back from the
273
+ // journal at the moment that dispatch is built, never carried in a process variable — a stop between
274
+ // the decision and the dispatch (OBS-254's shape, one layer up) would otherwise send a normal prompt
275
+ // with the findings gone, or re-run an assignment that was banned.
276
+ //
277
+ // Rule: retry state is spent iff a worker actually launches. The expiry is `worker-launch`, NOT
278
+ // `task-dispatch`: task-dispatch is journaled before worktree
279
+ // recreation, setup, prompt writing and slot allocation, so spending the decision there hands it to a
280
+ // dispatch that may still die before any worker sees it — and `--retry-failed` would then send a fresh
281
+ // prompt with the repair findings gone, or re-run the banned channel. worker-launch is appended only
282
+ // once the prompt has actually been delivered to a worker, which is the dispatch the decision governs.
283
+ const DECISION_SPENT = "worker-launch";
284
+ function decisionForNextDispatch(events, taskId, event) {
285
+ let pending;
286
+ for (const e of events) {
287
+ if (e.taskId !== taskId)
288
+ continue;
289
+ if (e.event === event)
290
+ pending = e;
291
+ else if (e.event === DECISION_SPENT)
292
+ pending = undefined;
293
+ }
294
+ return pending;
295
+ }
296
+ /**
297
+ * Why the last attempt failed, one row per journaled cause, in the daemon's own `source: details`
298
+ * shape. The daemon builds that brief in a loop-local variable, which dies with the process: a resumed
299
+ * or `--retry-failed` run rebuilt the prompt from nothing and dispatched a retry that had lost the
300
+ * reason it was retrying — OBS-254's class, one layer below the upheld brief. Re-derived here so the
301
+ * bytes the journal already holds cannot be taken away by any reset of attempt or channel state.
302
+ *
303
+ * The same rule governs a dead DISPATCH: its exact task-failed error is retained until
304
+ * `worker-launch`, never retired at
305
+ * `task-dispatch`: everything between the two — worktree recreation, setup, prompt write, slot
306
+ * allocation, the launch itself — can still die with no worker having read a word, and clearing at
307
+ * task-dispatch meant `--retry-failed` after exactly that death rebuilt the prompt without the gate
308
+ * failures OR the delivery failure that preceded it. `task-approved` also clears (an operator approval
309
+ * retires the findings it settled — the uphold case re-derives its own brief separately).
310
+ */
311
+ export function journaledFailureBrief(events, taskId) {
312
+ let rows = [];
313
+ for (const e of events) {
314
+ if (e.taskId !== taskId)
315
+ continue;
316
+ if (e.event === "worker-launch" || e.event === "task-approved")
317
+ rows = [];
318
+ else if (e.event === "gate-result" && e.data.pass === false && e.data.skipped !== true
319
+ && typeof e.data.details === "string")
320
+ rows.push(`${e.data.gate}: ${e.data.details}`);
321
+ else if (e.event === "delivery-readiness-failed" && typeof e.data.transcript === "string") {
322
+ rows.push(`dispatch: delivery readiness failed after ${e.data.waitedMs}ms; pane transcript:\n${e.data.transcript}`);
323
+ }
324
+ else if (e.event === "task-failed" && e.data.kind === "dispatch" && typeof e.data.error === "string") {
325
+ rows.push(`dispatch: ${e.data.error}`);
326
+ }
327
+ }
328
+ return rows;
329
+ }
330
+ /** The findings a funded repair must carry into the next dispatch, or undefined if none is pending. */
331
+ export function pendingRepairFindings(events, taskId) {
332
+ const e = decisionForNextDispatch(events, taskId, "repair-attempt");
333
+ return typeof e?.data.findings === "string" ? e.data.findings : undefined;
334
+ }
335
+ /**
336
+ * The gate whose identical failure banned an identical retry of the NEXT dispatch — bound to the
337
+ * channel that produced it, so a verdict that has already moved the work elsewhere is not refused for
338
+ * a channel it is no longer using, and a later unrelated failure is not parked under a stale reason.
339
+ */
340
+ export function activeRetryBan(events, taskId, channel) {
341
+ const e = decisionForNextDispatch(events, taskId, "gate-fingerprint-cap");
342
+ return e && e.data.channel === channel && typeof e.data.gate === "string" ? e.data.gate : undefined;
343
+ }
73
344
  // Fail-closed shape for a dispatched assignment (journal.ts:75-90 posture): a malformed assignment in
74
345
  // one dispatch degrades that single task toward today's behavior — counts toward attempts, contributes
75
346
  // nothing to tried, poisons only lastAssignment — never crashes resume, never poisons other tasks.
@@ -81,7 +352,10 @@ const DispatchAssignmentSchema = z.object({
81
352
  });
82
353
  export const PARK_KINDS = ["human-gate", "ladder-exhausted", "attempt-cap", "gate-fail", "quota",
83
354
  "reroute-exhausted", "setup", "stall", "merge-conflict", "tip-moved", "infra", "dispatch"];
84
- export const RETRY_MODES = ["resume", "fresh"];
355
+ // v1.85 T3: "repair" is a third dispatch mode beside the v1.29 session pair — a fix-only attempt that
356
+ // carries the failing findings and the diff CONTENT of the work already landed, instead of re-buying
357
+ // ~20m of onboarding to rediscover them (62 of 68 measured re-dispatches were fresh).
358
+ export const RETRY_MODES = ["resume", "fresh", "repair"];
85
359
  export const WORKER_RESULT_CAUSES = ["provider-death", "stall-timeout", "malformed-trailer", "clean-exit-no-trailer"];
86
360
  // Status consumes the routing profile's existing quality split directly: verified park kinds classify
87
361
  // to 0, while availability/recovery noise classifies to null. Keep the synthetic row here at the
@@ -6,6 +6,14 @@ export interface StallProgressSample {
6
6
  seedSubmitted?: boolean;
7
7
  contextTokens?: number;
8
8
  }
9
+ export declare const NUDGEABLE_ADAPTERS: Set<string>;
10
+ export declare const QUOTA_BANNER_TAIL_ROWS = 12;
11
+ export declare function stallSnapshotTail(text: string, rows?: number): string;
12
+ export declare function stallSnapshotBannerRows(text: string, rows?: number): string;
13
+ export declare const ROW_REARM_TOKEN_FLAT_MS: number;
14
+ export declare function setRowRearmTokenFlatMsForTests(ms: number): void;
15
+ export declare function resetRowRearmTokenFlatMsForTests(): void;
16
+ export declare const PANE_READ_ROWS = 1000;
9
17
  /**
10
18
  * Monotonic worker-progress measure for the stall watchdog.
11
19
  *
@@ -13,12 +21,38 @@ export interface StallProgressSample {
13
21
  * evidence of work. A rendered transcript is only known to have grown when it occupies more
14
22
  * non-empty rows than any prior sample. Same-row rewrites are deliberately ambiguous and do not
15
23
  * advance the clock: a recoverable early consult is safer than silencing the watchdog forever.
24
+ *
25
+ * CEILING: `transcriptRows` is a monotone high-water over the daemon's bounded pane read
26
+ * (PANE_READ_ROWS lines), so it is a one-way ratchet whose signal goes blind once the pane's
27
+ * content exceeds the read window — the same full window slides and `observe()` can never
28
+ * report row growth again. Past that point a `false` return means "unmeasurable", not "no
29
+ * output" — consumers making a kill decision (the dead-channel fast-kill) must check
30
+ * `rowSignalSaturated` and stand down on it.
16
31
  */
17
32
  export declare class StallProgressTracker {
18
33
  private transcriptRows;
34
+ private rawWindowLines;
19
35
  private seedSubmitted;
20
36
  private contextTokens;
21
- observe(sample: StallProgressSample): boolean;
37
+ private lastTokenGrowthAt;
38
+ private rowGrowthAt;
39
+ /** True once a sample FILLED the bounded read window on RAW lines (blanks and chrome-only
40
+ * rows included): the pane's real extent is then unknown — genuinely new content scrolls out
41
+ * of the read and the row high-water can never advance again — so a flat tracker is blindness,
42
+ * not silence. The raw window is the saturation signal, NOT the normalized non-empty count: a
43
+ * production `read(slot, PANE_READ_ROWS)` returns at most PANE_READ_ROWS lines including blank
44
+ * and chrome-only rows (measured 730 non-empty of 1000 on the codex-mcp-spinner fixture), so
45
+ * comparing the non-empty high-water against PANE_READ_ROWS could never engage and the
46
+ * fast-kill's stand-down was unreachable. Sticky by construction (the high-water never
47
+ * decreases). */
48
+ get rowSignalSaturated(): boolean;
49
+ /** Raw row-growth clock: the last observe() that advanced the row high-water, recorded even
50
+ * when the flat-token rule suppresses the progress REPORT (observe returns false). T1 review:
51
+ * the dead-channel fast-kill's "no output growth" leg must read this, not the suppressed
52
+ * progress clock — a metered adapter whose sticky token counter freezes the report while the
53
+ * pane keeps streaming rows is alive, and only this clock sees it. */
54
+ get lastRowGrowthAt(): number | undefined;
55
+ observe(sample: StallProgressSample, now?: number): boolean;
22
56
  }
23
57
  /** Filter transcript text bound for an LLM prompt (consult dossiers, gate prompts). The classify
24
58
  * seam exists for fault injection in tests only — production callers pass text alone. */
package/dist/run/stall.js CHANGED
@@ -18,6 +18,62 @@ const ELAPSED_RE = /(?<![\w.])\d+(?:\.\d+)?(?:ms|[hms])(?!\w)/g;
18
18
  export function normalizeStallSnapshot(text) {
19
19
  return text.replace(ANSI_RE, "").replace(SPINNER_RE, "").replace(ELAPSED_RE, "");
20
20
  }
21
+ // T1 (OBS-262): the rescue nudge's adapter scope — claude-code only (steering path proven,
22
+ // OBS-122). Widening it is a future fixture-capture chore (an occupied-frame capture per adapter,
23
+ // OBS-181 scar), never a drive-by edit. Lives in the stall module so the watchdog's policy and its
24
+ // scope constant cannot drift apart.
25
+ export const NUDGEABLE_ADAPTERS = new Set(["claude-code"]);
26
+ // T1 (OBS-263): a LIVE provider banner is the last thing the pane printed — the worker stopped
27
+ // underneath it. A "quota"/"rate limit" mention inside the task prompt, a diff hunk, or earlier
28
+ // output sits ABOVE the transcript tail and must never fail an attempt over, so the in-loop quota
29
+ // classifier reads only this many trailing non-empty rows instead of the whole retained snapshot.
30
+ // ponytail: rows, not a banner grammar — the ceiling is a worker frozen with a quota mention as its
31
+ // literal last output; the two-consecutive-slices + tracker-silence gates bound that cost to one
32
+ // failover within the routing floor. Upgrade path is a per-adapter banner fixture if it ever bites.
33
+ export const QUOTA_BANNER_TAIL_ROWS = 12;
34
+ export function stallSnapshotTail(text, rows = QUOTA_BANNER_TAIL_ROWS) {
35
+ return normalizeStallSnapshot(text)
36
+ .split("\n")
37
+ .filter((line) => line.trim().length > 0)
38
+ .slice(-rows)
39
+ .join("\n");
40
+ }
41
+ // T1 review (chrome-blind-matcher class, OBS-152/155): the tail of a RENDERED TUI frame is not
42
+ // "what the pane printed last" — its bottom rows are fixed composer/welcome chrome. Codex pins
43
+ // "• You have 3 usage limit resets available." there, so a raw-tail QUOTA_RE match fires on every
44
+ // frame of a wedged pane (verified against all 8 frames of tests/fixtures/codex-mcp-spinner) and
45
+ // would fail a live worker over mid-work. Filter the KNOWN chrome instead of everything on screen
46
+ // at some anchor: a novelty baseline cannot distinguish "chrome that was already there" from "a
47
+ // real banner the CLI printed before the first poll read" — a channel throttled at launch paints
48
+ // its banner inside the first BLOCKED_POLL_MS slice, so the banner BECOMES the baseline and is
49
+ // exculpated forever (proven by execution: banner-from-the-first-loop-read fails over on shipped
50
+ // 843328b0, parks human under the baseline). This line is semantically the opposite of exhaustion
51
+ // — resets AVAILABLE — so matching it out can never hide a real banner. Closed allowlist, same
52
+ // philosophy as the normalizer's: a new adapter's quota-flavored chrome is a fixture-capture
53
+ // chore, never a drive-by edit.
54
+ const QUOTA_CHROME_RE = /usage limit resets? available/i;
55
+ export function stallSnapshotBannerRows(text, rows = QUOTA_BANNER_TAIL_ROWS) {
56
+ return stallSnapshotTail(text, rows)
57
+ .split("\n")
58
+ .filter((line) => !QUOTA_CHROME_RE.test(line))
59
+ .join("\n");
60
+ }
61
+ // T1 (OBS-262/263, speed-spec §2): past fifteen minutes with a FLAT token-usage counter, row
62
+ // growth alone no longer re-arms the inactivity window — cosmetic repaint rows are not paid work.
63
+ // Token usage is the paid-work signal the tracker already samples; token growth always re-arms.
64
+ export const ROW_REARM_TOKEN_FLAT_MS = 15 * 60_000;
65
+ // Test seam, same pattern as the daemon's timing seams: production reads the constant.
66
+ let rowRearmTokenFlatMs = ROW_REARM_TOKEN_FLAT_MS;
67
+ export function setRowRearmTokenFlatMsForTests(ms) {
68
+ rowRearmTokenFlatMs = ms;
69
+ }
70
+ export function resetRowRearmTokenFlatMsForTests() {
71
+ rowRearmTokenFlatMs = ROW_REARM_TOKEN_FLAT_MS;
72
+ }
73
+ // T1 review (read-ceiling blindness): the daemon samples panes through `driver.read(slot, N)` —
74
+ // a bounded window. This constant IS that N, and the daemon's pane reads must use it (never a
75
+ // literal) so the tracker's saturation check below cannot drift away from the real read depth.
76
+ export const PANE_READ_ROWS = 1000;
21
77
  /**
22
78
  * Monotonic worker-progress measure for the stall watchdog.
23
79
  *
@@ -25,32 +81,86 @@ export function normalizeStallSnapshot(text) {
25
81
  * evidence of work. A rendered transcript is only known to have grown when it occupies more
26
82
  * non-empty rows than any prior sample. Same-row rewrites are deliberately ambiguous and do not
27
83
  * advance the clock: a recoverable early consult is safer than silencing the watchdog forever.
84
+ *
85
+ * CEILING: `transcriptRows` is a monotone high-water over the daemon's bounded pane read
86
+ * (PANE_READ_ROWS lines), so it is a one-way ratchet whose signal goes blind once the pane's
87
+ * content exceeds the read window — the same full window slides and `observe()` can never
88
+ * report row growth again. Past that point a `false` return means "unmeasurable", not "no
89
+ * output" — consumers making a kill decision (the dead-channel fast-kill) must check
90
+ * `rowSignalSaturated` and stand down on it.
28
91
  */
29
92
  export class StallProgressTracker {
30
- transcriptRows = 0;
93
+ transcriptRows = 0; // non-empty high-water over the bounded read — the growth signal
94
+ rawWindowLines = 0; // raw-line high-water — the SATURATION signal (see the getter)
31
95
  seedSubmitted = false;
32
96
  contextTokens;
33
- observe(sample) {
34
- let advanced = false;
97
+ lastTokenGrowthAt; // undefined until the first token sample anchors the flat-clock
98
+ rowGrowthAt; // raw row-growth clock — NEVER suppressed by the flat-token rule
99
+ /** True once a sample FILLED the bounded read window on RAW lines (blanks and chrome-only
100
+ * rows included): the pane's real extent is then unknown — genuinely new content scrolls out
101
+ * of the read and the row high-water can never advance again — so a flat tracker is blindness,
102
+ * not silence. The raw window is the saturation signal, NOT the normalized non-empty count: a
103
+ * production `read(slot, PANE_READ_ROWS)` returns at most PANE_READ_ROWS lines including blank
104
+ * and chrome-only rows (measured 730 non-empty of 1000 on the codex-mcp-spinner fixture), so
105
+ * comparing the non-empty high-water against PANE_READ_ROWS could never engage and the
106
+ * fast-kill's stand-down was unreachable. Sticky by construction (the high-water never
107
+ * decreases). */
108
+ get rowSignalSaturated() {
109
+ return this.rawWindowLines >= PANE_READ_ROWS;
110
+ }
111
+ /** Raw row-growth clock: the last observe() that advanced the row high-water, recorded even
112
+ * when the flat-token rule suppresses the progress REPORT (observe returns false). T1 review:
113
+ * the dead-channel fast-kill's "no output growth" leg must read this, not the suppressed
114
+ * progress clock — a metered adapter whose sticky token counter freezes the report while the
115
+ * pane keeps streaming rows is alive, and only this clock sees it. */
116
+ get lastRowGrowthAt() {
117
+ return this.rowGrowthAt;
118
+ }
119
+ observe(sample, now = Date.now()) {
120
+ let rowsAdvanced = false;
121
+ // raw window high-water first — this is the saturation signal (rowSignalSaturated), and it
122
+ // must see the sample exactly as the bounded read returned it, blanks and chrome included.
123
+ const rawLines = sample.paneText.split("\n").length;
124
+ if (rawLines > this.rawWindowLines)
125
+ this.rawWindowLines = rawLines;
35
126
  const rows = normalizeStallSnapshot(sample.paneText)
36
127
  .split("\n")
37
128
  .filter((line) => line.trim().length > 0)
38
129
  .length;
39
130
  if (rows > this.transcriptRows) {
40
131
  this.transcriptRows = rows;
41
- advanced = true;
132
+ rowsAdvanced = true;
133
+ this.rowGrowthAt = now; // raw signal — advances even when the report below is suppressed
42
134
  }
135
+ let seedAdvanced = false;
43
136
  if (sample.seedSubmitted && !this.seedSubmitted) {
44
137
  this.seedSubmitted = true;
45
- advanced = true;
138
+ seedAdvanced = true;
46
139
  }
140
+ let tokensAdvanced = false;
47
141
  const tokens = sample.contextTokens;
48
142
  if (tokens !== undefined && Number.isFinite(tokens)) {
49
143
  if (tokens > (this.contextTokens ?? 0))
50
- advanced = true;
51
- this.contextTokens = Math.max(this.contextTokens ?? 0, tokens);
144
+ tokensAdvanced = true;
145
+ // T1 review: ANY movement re-anchors the flat-clock, not just a new high-water mark — a
146
+ // context compaction drops the counter, and the climb back below the old peak is still paid
147
+ // work. A high-water comparison would freeze the anchor forever on the first decrease. (The
148
+ // first token sample always counts as movement: contextTokens starts undefined.)
149
+ if (tokens !== this.contextTokens)
150
+ this.lastTokenGrowthAt = now;
151
+ this.contextTokens = tokens;
152
+ }
153
+ if (tokensAdvanced || seedAdvanced)
154
+ return true;
155
+ if (rowsAdvanced) {
156
+ // T1: row growth past the flat-token cap is cosmetic — the paid-work counter has not moved,
157
+ // so the inactivity window must NOT re-arm on it. A tracker that never sees a token sample
158
+ // (unmetered adapter) keeps the old row-growth behavior.
159
+ if (this.lastTokenGrowthAt !== undefined && now - this.lastTokenGrowthAt >= rowRearmTokenFlatMs)
160
+ return false;
161
+ return true;
52
162
  }
53
- return advanced;
163
+ return false;
54
164
  }
55
165
  }
56
166
  // ─── v1.65 T2: LLM-bound transcript filter ──────────────────────────────────────────────────────
@@ -39,6 +39,8 @@ export declare const KEYBAR_KEYS: {
39
39
  export type JournalRow = {
40
40
  readonly id: string;
41
41
  readonly time: string;
42
+ /** The complete recorded instant behind `time`; absent only on non-event/legacy rows. */
43
+ readonly timestamp?: string;
42
44
  readonly state: ComponentState;
43
45
  readonly text: string;
44
46
  };