tickmarkr 1.83.0 → 1.85.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (37) hide show
  1. package/dist/adapters/claude-code.d.ts +1 -0
  2. package/dist/adapters/claude-code.js +57 -1
  3. package/dist/adapters/fake.js +9 -0
  4. package/dist/adapters/types.d.ts +3 -0
  5. package/dist/adapters/types.js +21 -0
  6. package/dist/cli/commands/status.js +160 -30
  7. package/dist/compile/collateral.d.ts +86 -2
  8. package/dist/compile/collateral.js +294 -3
  9. package/dist/config/config.d.ts +62 -0
  10. package/dist/config/config.js +157 -2
  11. package/dist/drivers/herdr.d.ts +20 -3
  12. package/dist/drivers/herdr.js +288 -105
  13. package/dist/gates/baseline.d.ts +1 -0
  14. package/dist/gates/baseline.js +91 -13
  15. package/dist/gates/review.d.ts +7 -0
  16. package/dist/gates/review.js +99 -6
  17. package/dist/gates/run-gates.d.ts +9 -0
  18. package/dist/gates/run-gates.js +285 -41
  19. package/dist/run/daemon.d.ts +48 -2
  20. package/dist/run/daemon.js +1417 -315
  21. package/dist/run/journal.d.ts +56 -3
  22. package/dist/run/journal.js +275 -1
  23. package/dist/run/stall.d.ts +35 -1
  24. package/dist/run/stall.js +118 -8
  25. package/dist/tui/cockpit/components.d.ts +30 -1
  26. package/dist/tui/cockpit/components.js +19 -3
  27. package/dist/tui/cockpit/derive.d.ts +29 -2
  28. package/dist/tui/cockpit/derive.js +219 -23
  29. package/dist/tui/cockpit/layout.d.ts +75 -6
  30. package/dist/tui/cockpit/layout.js +97 -19
  31. package/dist/tui/cockpit/live.d.ts +14 -1
  32. package/dist/tui/cockpit/live.js +223 -29
  33. package/dist/tui/cockpit/pointer.d.ts +261 -0
  34. package/dist/tui/cockpit/pointer.js +610 -0
  35. package/dist/tui/cockpit/run-cockpit.d.ts +36 -5
  36. package/dist/tui/cockpit/run-cockpit.js +270 -51
  37. package/package.json +1 -1
@@ -1,4 +1,4 @@
1
- import { randomBytes } from "node:crypto";
1
+ import { createHash, randomBytes } from "node:crypto";
2
2
  import { shq } from "../adapters/types.js";
3
3
  import { existsSync, mkdirSync, mkdtempSync, readFileSync, rmSync, writeFileSync } from "node:fs";
4
4
  import { tmpdir } from "node:os";
@@ -8,7 +8,7 @@ import { classifyDeadChannel, NO_TRAILER_SUMMARY, trailerPattern, UNPARSEABLE_TR
8
8
  import { allAdapters, discoverChannels, getAdapter, probeAll, readDoctor } from "../adapters/registry.js";
9
9
  import { addUsage, channelKey, matchesTrustDialog, QUOTA_RE } from "../adapters/types.js";
10
10
  import { bannerShell, paneDispatchCommand } from "../brand.js";
11
- import { globalConfigDir, loadConfigWithMode, readOverlayFile, repoOverlayPath, } from "../config/config.js";
11
+ import { DEFAULT_DIFF_CAP, globalConfigDir, loadConfigWithMode, readOverlayFile, repoOverlayPath, } from "../config/config.js";
12
12
  import { DeliveryReadinessError } from "../drivers/herdr.js";
13
13
  import { herdrSealShellPrefix, SubprocessDriver } from "../drivers/subprocess.js";
14
14
  import { formatOwnedName } from "../drivers/types.js";
@@ -20,13 +20,13 @@ import { augmentRetryBrief, consult, renderRetryGuidance } from "./consult.js";
20
20
  import { runEnvironment } from "./environment.js";
21
21
  import { cleanupRunWorktrees, gitHead, linkNodeModules, npmDependencyInstallCommand, npmDependencyManifestChanged, sh, shGit, WORKTREE_LAYOUT_CONTRACT, worktreePath } from "./git.js";
22
22
  import { runInteractiveSeed } from "./interactive-seed.js";
23
- import { classifyTaskFailure, classifyWorkerResultCause, engagementComparable, Journal, loadRoutingProfile, newRunId, phaseForGate, recordedTaskFailureKind, reviewRoundsSinceApproval } from "./journal.js";
23
+ import { activeRetryBan, classifyTaskFailure, classifyWorkerResultCause, engagementComparable, GATE_FINGERPRINT_CAP, identicalGateFailures, journaledFailureBrief, Journal, loadRoutingProfile, newRunId, normalizeGateFailure, pendingRepairFindings, phaseForGate, recordedTaskFailureKind, repairsSinceApproval, reviewRoundsSinceApproval, structuredFindings, upheldFeedbackByTask } from "./journal.js";
24
24
  import { isDiffCapPark } from "../gates/review.js";
25
25
  import { acquireRunLock, releaseRunLock } from "./lock.js";
26
26
  import { ensureIntegration, integrationBranch, integrationHead, mergeTask, verifyIntegrationTip } from "./merge.js";
27
27
  import { nextChannel, route } from "../route/router.js";
28
28
  import { desiredPanes } from "./reconcile.js";
29
- import { StallProgressTracker } from "./stall.js";
29
+ import { NUDGEABLE_ADAPTERS, PANE_READ_ROWS, StallProgressTracker, stallSnapshotBannerRows } from "./stall.js";
30
30
  const MODE_RANK = { "staff-led": 0, "risk-based": 1, "partner-led": 2 };
31
31
  // An override (flag/spec) re-resolves through loadConfigWithMode itself, via a synthesized repo overlay
32
32
  // carrying routing.mode — floors, explore, lints, and provenance all come from config.ts's preset
@@ -69,6 +69,107 @@ export function formatSummary(s) {
69
69
  return `done: ${s.done.length}, failed: ${s.failed.length}, human: ${s.human.length}, blocked: ${s.blocked.length}, pending: ${s.pending.length}\nintegration branch: ${s.branch}${tip}`;
70
70
  }
71
71
  const MAX_ATTEMPTS = 10; // ponytail: hard cap so a pathological ladder can never loop forever
72
+ // v1.85 T3 (retry economics): two repairs per engagement, then the fresh ladder. A repair re-uses the
73
+ // findings and the landed diff instead of re-buying onboarding; when two of them have not closed the
74
+ // battery, the cheaper next move is the ladder's channel change, not a third fix-only pass.
75
+ const MAX_REPAIRS = 2;
76
+ /** A named oracle decided this acceptance failure — deterministic, unlike an LLM judge verdict. */
77
+ const isOracleFailure = (g) => g.details.startsWith("oracle failed:");
78
+ /**
79
+ * R3 (OBS-186): a gate that DECLINED to run is not a gate that failed. The review gate's skip branch
80
+ * no longer forges `pass: true` to buy passage, so the merge decision has to read the same predicate
81
+ * the run surfaces already read (src/run/activity.ts): pass, or an honest declared skip. Without this
82
+ * the honesty change would silently park every judge-only task at merge — an unrun gate blocking work
83
+ * it was never asked to review. `skipped` is set only by a gate that says so about ITSELF; a red
84
+ * verdict from a review that actually ran still fails here, exactly as before.
85
+ *
86
+ * ONE pair of predicates, every fold. `!g.pass` was correct only while the sole `pass:false` producer
87
+ * was a gate that actually failed; the moment a decline can be recorded red, every `!g.pass` in this
88
+ * file — the retry feedback brief, the review-fix eligibility test, the failing-battery list the
89
+ * ladder and the fingerprint cap are scored on, the structured findings attached to a blocking
90
+ * verdict — reads an unrun gate as a defect. `gateFailed` is the seam they now share, and the journal
91
+ * write below is the seam every OUT-of-file fold shares.
92
+ */
93
+ const gateSatisfied = (g) => g.pass || g.meta?.skipped === true;
94
+ const gateFailed = (g) => !gateSatisfied(g);
95
+ // v1.85 T3: the gates whose failure IS a deterministic measurement — a machine re-ran a command over a
96
+ // tree and printed the same bytes. Those are the failures the fingerprint cap governs (the ruling names
97
+ // it a "deterministic-gate" cap): a third identical answer to a question already answered twice is the
98
+ // ~663m-across-5-runs loop, whatever rung the ladder happens to stand on. An LLM verdict is a different
99
+ // object — two reviewers, or a judge asked twice, can restate one another without the question being
100
+ // closed — and each already carries a tighter bound of its own: REVIEW_ROUND_CAP parks review at the
101
+ // OPERATOR in two rounds, and a judge verdict rides the ladder and the attempt cap. The boundary is a
102
+ // property of the GATE, never of the ladder rung or of the move that would follow the failure.
103
+ const DETERMINISTIC_GATES = new Set(["build", "test", "lint", "evidence", "scope"]);
104
+ const isDeterministicFailure = (g) => DETERMINISTIC_GATES.has(g.gate) || (g.gate === "acceptance" && isOracleFailure(g));
105
+ /**
106
+ * Is the failing battery narrow enough that a fix-only pass can close it? The ruling's three cases:
107
+ * review-only, a single deterministic test/lint gate, or acceptance decided by a named oracle.
108
+ * Unparseable verdicts and diff-cap trips are excluded exactly as they are from the review-fix retry —
109
+ * neither names anything a worker can fix.
110
+ */
111
+ function narrowRepairBattery(failing) {
112
+ if (failing.length !== 1)
113
+ return false;
114
+ const g = failing[0];
115
+ if (g.gate === "review")
116
+ return g.meta?.unparseable !== true && !isDiffCapPark(g);
117
+ if (g.gate === "test" || g.gate === "lint")
118
+ return true;
119
+ if (g.gate === "acceptance")
120
+ return isOracleFailure(g) && g.meta?.unparseable !== true;
121
+ return false;
122
+ }
123
+ /**
124
+ * T4 (OBS-265): the journal with the review objections a round did NOT hinge on removed. Judge and
125
+ * review are now launched together, so a round can journal a failed review that the serial walk would
126
+ * never have asked for — it returned at the judge. Those verdicts stay on the record (they are real,
127
+ * and the retry brief carries them), but they must not spend the OPERATOR-facing review round budget:
128
+ * otherwise concurrency alone parks a task rounds early for objections the old pipeline never bought.
129
+ * A round is the gate-result span opened by each `gates` phase-start, per task.
130
+ */
131
+ export function decisiveReviewRounds(events) {
132
+ const open = new Map();
133
+ const spent = new Set();
134
+ const close = (taskId) => {
135
+ const round = open.get(taskId) ?? [];
136
+ if (round.some((e) => e.data.gate !== "review" && e.data.pass === false)) {
137
+ for (const e of round)
138
+ if (e.data.gate === "review" && e.data.pass === false)
139
+ spent.add(e);
140
+ }
141
+ open.delete(taskId);
142
+ };
143
+ for (const e of events) {
144
+ if (!e.taskId)
145
+ continue;
146
+ if (e.event === "phase-start" && e.data.phase === "gates")
147
+ close(e.taskId);
148
+ else if (e.event === "gate-result")
149
+ open.set(e.taskId, [...(open.get(e.taskId) ?? []), e]);
150
+ }
151
+ for (const taskId of open.keys())
152
+ close(taskId); // Map iteration tolerates deleting the current key
153
+ return events.filter((e) => !spent.has(e));
154
+ }
155
+ /** The fix-only contract: the findings verbatim, then the diff content of the work already landed. */
156
+ function repairBrief(findings, diff, baseRef) {
157
+ const repair = { findings, diff };
158
+ return [
159
+ "## Repair attempt — fix ONLY what these findings name",
160
+ "The commits from your prior attempt are already in this worktree and their diff is reproduced"
161
+ + " below. Do NOT re-implement that work, do not start over, and do not revert it: make the"
162
+ + " smallest change that resolves every finding, then commit.",
163
+ "",
164
+ "### Failing gate findings (verbatim)",
165
+ repair.findings,
166
+ "",
167
+ `### The work under review (git diff ${baseRef.slice(0, 12)}..HEAD)`,
168
+ "```diff",
169
+ repair.diff,
170
+ "```",
171
+ ].join("\n");
172
+ }
72
173
  // v1.70 T5: request-changes review rounds a single task may draw before it parks for a human decision
73
174
  // instead of cycling. Well below MAX_ATTEMPTS so review non-convergence is caught long before the
74
175
  // global cap. ponytail: literal constant; lift to cfg.review.roundCap only if a second knob-turner appears.
@@ -92,19 +193,23 @@ export function setEarlyLaunchLivenessMsForTests(ms) {
92
193
  export function resetEarlyLaunchLivenessMsForTests() {
93
194
  earlyLaunchLivenessMs = EARLY_LAUNCH_LIVENESS_MS;
94
195
  }
95
- // OBS-201: the liveness nudge — the daemon's ACTIVE response to an idle worker holding no trailer,
96
- // replacing page-a-human-then-burn-the-window (289 of 692 worker-minutes in one measured day).
97
- // Gate: herdr classifies the pane idle AND the monotonic tracker has seen nothing for
98
- // NUDGE_AFTER_SILENT_MS (a worker grinding inside a tool run is `working` and never reaches the
99
- // gate). One nudge per attempt; if the grace passes still idle with no progress, the wait concludes
100
- // as a stall NOW and the consult sees the un-answered nudge instead of an hour of silence.
101
- // Allowlist: claude-code first (steering path proven, OBS-122); widen per adapter only with a
102
- // captured occupied-frame fixture (OBS-181 scar). The message builder takes no nonce — the
196
+ // OBS-201 + T1 (OBS-262): the liveness nudge — the daemon's ACTIVE response to a worker holding no
197
+ // trailer, replacing page-a-human-then-burn-the-window (289 of 692 worker-minutes in one measured
198
+ // day). Gate: NUDGE_AFTER_SILENT_MS of monotonic-tracker silence, regardless of the herdr status
199
+ // reading (unknown/working no longer suppress it — a wedged TUI often scrapes as either); a
200
+ // `blocked` pane still pages instead, since nudging a dialog prompt can't help. One nudge per
201
+ // attempt; if the grace passes with no progress, the wait concludes as a stall NOW and the consult
202
+ // sees the un-answered nudge instead of an hour of silence. Scope allowlist lives in stall.ts
203
+ // (claude-code only; widening is a fixture-capture chore). The message builder takes no nonce — the
103
204
  // self-reference guard holds by construction, an echoed bare token can never match the wait regex.
104
- export const NUDGEABLE_ADAPTERS = new Set(["claude-code"]);
205
+ export { NUDGEABLE_ADAPTERS } from "./stall.js";
105
206
  export const WORKER_NUDGE_MESSAGE = "tickmarkr liveness check: if the task is complete, print your TICKMARKR_RESULT completion trailer exactly as specified in your prompt now. If not, state your next concrete action and continue working.";
106
- const NUDGE_AFTER_SILENT_MS = 3 * 60_000;
207
+ const NUDGE_AFTER_SILENT_MS = 10 * 60_000; // T1 (OBS-262): >=10m tracker silence — was 3m behind an unreachable status gate
107
208
  const WORKER_NUDGE_GRACE_MS = 4 * 60_000;
209
+ // T1 review: a false return from driver.nudge is a DELIVERY outcome (missing pin, readiness
210
+ // stable-frame timeout, read-back hiccup), not proof of an unreachable channel — so one failure
211
+ // is retried once in-slice after this settle, and only a failed retry latches nudgeFailed.
212
+ const NUDGE_REDELIVER_MS = 2_000;
108
213
  let nudgeAfterSilentMs = NUDGE_AFTER_SILENT_MS;
109
214
  let workerNudgeGraceMs = WORKER_NUDGE_GRACE_MS;
110
215
  /** Test seam — shrink the nudge gate and grace without minute-long sleeps. */
@@ -116,6 +221,375 @@ export function resetNudgeTimingForTests() {
116
221
  nudgeAfterSilentMs = NUDGE_AFTER_SILENT_MS;
117
222
  workerNudgeGraceMs = WORKER_NUDGE_GRACE_MS;
118
223
  }
224
+ // T1 (OBS-263): in-loop quota-banner classification — the banner IS output, so the empty-output
225
+ // rules can never catch it and the post-loop QUOTA_RE check only runs after the full window. Two
226
+ // consecutive matching slices plus this much monotonic-tracker silence classify (a worker whose
227
+ // diff merely quotes "rate limit" keeps working undisturbed).
228
+ const QUOTA_BANNER_SILENT_MS = 3 * 60_000;
229
+ let quotaBannerSilentMs = QUOTA_BANNER_SILENT_MS;
230
+ /** Test seam — shrink the quota-banner silence gate without minute-long sleeps. */
231
+ export function setQuotaBannerSilentMsForTests(ms) {
232
+ quotaBannerSilentMs = ms;
233
+ }
234
+ export function resetQuotaBannerSilentMsForTests() {
235
+ quotaBannerSilentMs = QUOTA_BANNER_SILENT_MS;
236
+ }
237
+ // T1 (OBS-262): the operator page is UNLATCHED — every eligible slice journals a page, and the
238
+ // notification is delivered again on a status change or once this cadence elapses. The cadence is
239
+ // an operator-spam guard only; it can no longer turn a stall into a single page forever. It sits
240
+ // BELOW the dead-channel fast-kill window on purpose (T1 review): at 5m == 5m the second delivery
241
+ // raced the kill on the same slice boundary, so an idle non-nudgeable pane holding no delta — the
242
+ // exact class the repeat exists for — got exactly one page in production.
243
+ const PAGE_REPEAT_MS = 2 * 60_000;
244
+ let pageRepeatMs = PAGE_REPEAT_MS;
245
+ /** Test seam — shrink the repeat-page cadence without minute-long sleeps. */
246
+ export function setPageRepeatMsForTests(ms) {
247
+ pageRepeatMs = ms;
248
+ }
249
+ export function resetPageRepeatMsForTests() {
250
+ pageRepeatMs = PAGE_REPEAT_MS;
251
+ }
252
+ // T1 (R1 dead-channel fast-kill): a worker with no trailer, no worktree delta, and no output
253
+ // growth for this long is dead — conclude immediately instead of burning the rolling window.
254
+ // Per-task timeoutMinutes stays the escape valve for slow-but-live workers.
255
+ const DEAD_CHANNEL_FAST_KILL_MS = 5 * 60_000;
256
+ let deadChannelFastKillMs = DEAD_CHANNEL_FAST_KILL_MS;
257
+ /** Test seam — shrink the fast-kill window without minute-long sleeps. */
258
+ export function setDeadChannelFastKillMsForTests(ms) {
259
+ deadChannelFastKillMs = ms;
260
+ }
261
+ export function resetDeadChannelFastKillMsForTests() {
262
+ deadChannelFastKillMs = DEAD_CHANNEL_FAST_KILL_MS;
263
+ }
264
+ // T2 (OBS-264): finished work is harvested, never redone. 18 of 18 observed stalls carried 2-33
265
+ // commits, and the redispatch then re-bought verification of work that had already landed. The
266
+ // liveness triad — commits made by this attempt, a FLAT worker-tree CPU delta, and this much
267
+ // monotonic-tracker silence — CONCLUDES the wait. Conclude, never kill: the pane is harvested by
268
+ // the same tail a window expiry uses, and the carried worktree goes straight to gates. Set at the
269
+ // fast-kill's window on purpose (the OBS-264 arithmetic is "a ~36m stall + ~15m redo becomes a
270
+ // ~5m gate pass"); a worker that is merely thinking still burns CPU and is never concluded here.
271
+ const HARVEST_SILENT_MS = 5 * 60_000;
272
+ let harvestSilentMs = HARVEST_SILENT_MS;
273
+ /** Test seam — shrink the harvest silence gate without minute-long sleeps. */
274
+ export function setHarvestSilentMsForTests(ms) {
275
+ harvestSilentMs = ms;
276
+ }
277
+ export function resetHarvestSilentMsForTests() {
278
+ harvestSilentMs = HARVEST_SILENT_MS;
279
+ }
280
+ // A CPU delta needs two samples separated in WALL CLOCK, and the CPU clock is QUANTIZED: darwin's
281
+ // `ps` prints hundredths ("0:00.03"), linux's prints whole seconds ("00:00:01"). Equality across a
282
+ // window shorter than the quantum is not evidence of anything — a worker throttled to a low duty
283
+ // cycle accrues less than one tick per sample and reads flat while genuinely working. So the flat
284
+ // observation must span the LARGER of a floor and this many ticks of the clock actually in use:
285
+ // crossing 30 ticks means the tree burned <1 tick in 30, i.e. under ~3% of one core. On a
286
+ // hundredths host that is a 3s window; on a whole-second host it is 30s — still nothing against the
287
+ // ~15m redispatch it replaces. Resolution is read off the sampled rows, never assumed.
288
+ const HARVEST_CPU_FLAT_MS = 3_000;
289
+ const HARVEST_CPU_FLAT_TICKS = 30;
290
+ // Once the flat window opens, retain descendants often enough to observe brief tool processes that
291
+ // can start and exit between the daemon's ordinary wait slices. This sampler exists only during an
292
+ // eligible silence window; it is stopped on progress or as soon as the worker wait concludes.
293
+ const HARVEST_CPU_ACCOUNTING_POLL_MS = 100;
294
+ // T2 review (material): that 100ms cadence forks a shell plus `ps` ten times a second, and on a host
295
+ // where `ps` is unsupported or denied (the managed-sandbox class) EVERY sample fails — tens of
296
+ // thousands of processes per silent attempt, multiplied by daemon concurrency, for a probe that can
297
+ // never conclude anything. Persistent failure is structural, not transient, so the sampler STOPS
298
+ // after this many consecutive unreadable snapshots. It stays stopped for the silence window it was
299
+ // started for: read() then reports no CPU, the triad refuses to conclude and journals the gap, and a
300
+ // later window (after real progress clears the accountant) starts a fresh one that pays the same
301
+ // bounded probe again.
302
+ const HARVEST_CPU_UNMEASURABLE_SAMPLE_CAP = 20;
303
+ let harvestCpuFlatMs;
304
+ export function harvestCpuFlatWindowMs(resolutionMs) {
305
+ return harvestCpuFlatMs ?? Math.max(HARVEST_CPU_FLAT_MS, resolutionMs * HARVEST_CPU_FLAT_TICKS);
306
+ }
307
+ /** Test seam — pin the flat window so a probe case need not sit through a real one. */
308
+ export function setHarvestCpuFlatMsForTests(ms) {
309
+ harvestCpuFlatMs = ms;
310
+ }
311
+ export function resetHarvestCpuFlatMsForTests() {
312
+ harvestCpuFlatMs = undefined;
313
+ }
314
+ // Once the silence gate is met the CPU probe owns the poll cadence: the trailer-wait slice is 30s,
315
+ // so two samples would otherwise cost a minute of wall clock apiece. Below the gate the only rule
316
+ // is not to sleep PAST it — at the shipped 5m gate that changes no slice a worker sees today.
317
+ const HARVEST_POLL_MS = 2_000;
318
+ function harvestSliceMs(silentMs) {
319
+ return Math.max(100, silentMs >= harvestSilentMs ? HARVEST_POLL_MS : harvestSilentMs - silentMs);
320
+ }
321
+ // The synthesized result a carried no-trailer harvest hands to the gates. Distinct from anything a
322
+ // worker can claim: it never comes from adapter.parse, and it is journaled under its own event.
323
+ export const HARVESTED_RESULT_SUMMARY = "harvested: the worktree carries committed work; the worker emitted no TICKMARKR_RESULT trailer";
324
+ /** T4 (OBS-266): identity of the command SET a tip verify ran — a changed command is a different verify. */
325
+ export function commandsHash(commands) {
326
+ return createHash("sha256").update(JSON.stringify(Object.entries(commands).sort())).digest("hex").slice(0, 12);
327
+ }
328
+ /**
329
+ * T4 (OBS-266): the journal's LAST verification cycle — the (tip, cmdHash) pair the most recent run
330
+ * of the verify commands spoke for, the gates it got a pass from, and whether anything failed in it.
331
+ *
332
+ * The LAST one, never a history of every pair ever green. "The last GREEN verified SHA" is what the
333
+ * spec licenses a skip against, and only the last cycle is a statement about the state the run is in
334
+ * now: after A→B→A the tip really moved, and after commands A→B→A the last thing that ran on this
335
+ * SHA was command set B — both re-verify. A cycle is the contiguous run of events sharing one pair,
336
+ * so a cycle cut short by a killed process is missing gates and can never satisfy the caller. A
337
+ * legacy event (no tip/cmdHash) is unattributable and breaks the chain outright.
338
+ */
339
+ function lastVerifyCycle(events) {
340
+ let cur;
341
+ let afterRunEnd = false;
342
+ for (const e of events) {
343
+ // New journals delimit every attempt explicitly. run-end is the legacy delimiter: it starts a
344
+ // new cycle only when another verify event follows, while preserving the just-closed cycle as
345
+ // the cache candidate for an otherwise unmoved next run-end.
346
+ if (e.event === "tip-verify-start") {
347
+ const { tip, cmdHash } = e.data;
348
+ cur = typeof tip === "string" && typeof cmdHash === "string"
349
+ ? { tip, cmdHash, gates: new Set(), failed: false }
350
+ : undefined;
351
+ afterRunEnd = false;
352
+ continue;
353
+ }
354
+ if (e.event === "run-end") {
355
+ afterRunEnd = true;
356
+ continue;
357
+ }
358
+ if (e.event !== "tip-verify" && e.event !== "tip-verify-failed")
359
+ continue;
360
+ const { tip, gate, cmdHash } = e.data;
361
+ if (typeof tip !== "string" || typeof gate !== "string" || typeof cmdHash !== "string") {
362
+ cur = undefined;
363
+ afterRunEnd = false;
364
+ continue;
365
+ }
366
+ if (!cur || afterRunEnd || cur.tip !== tip || cur.cmdHash !== cmdHash) {
367
+ cur = { tip, cmdHash, gates: new Set(), failed: false };
368
+ }
369
+ afterRunEnd = false;
370
+ if (e.event === "tip-verify-failed")
371
+ cur.failed = true;
372
+ else
373
+ cur.gates.add(gate);
374
+ }
375
+ return cur;
376
+ }
377
+ /**
378
+ * OBS-34's strict tip verify, but it stops re-paying for an unmoved tip (~334m corpus-wide; 69.5m in
379
+ * one park-heavy run whose 18 resume cycles merged nothing new). The verify journals the SHA it
380
+ * verified and the hash of the command set, so a later run-end can recognize the same verified state:
381
+ * head equals the LAST green verified SHA, commands unchanged, tree clean → journal
382
+ * `tip-verify-cached` and skip. ANY doubt — moved head, changed commands, dirty tree, a gate missing
383
+ * from that cycle's green set, a failure recorded in it, a cycle older than the last one — runs the
384
+ * full verify. The tip-verify-before-green law is untouched: a cached green is a verified green OF
385
+ * THAT EXACT COMMIT, established by the most recent real run of the same commands.
386
+ * Returns whether the tip is failing.
387
+ */
388
+ export async function verifyIntegrationTipCached(intWt, commands, journal, opts = {}) {
389
+ const cmdHash = commandsHash(commands);
390
+ const tip = await gitHead(intWt);
391
+ const porcelain = await shGit("git status --porcelain", intWt);
392
+ const clean = porcelain.code === 0 && porcelain.stdout.trim() === "";
393
+ const last = lastVerifyCycle(journal.read());
394
+ const cached = last !== undefined && !last.failed && last.tip === tip && last.cmdHash === cmdHash
395
+ && Object.keys(commands).every((g) => last.gates.has(g));
396
+ // A pair can be verified red and then green without either SHA or command hash changing (for
397
+ // example, an external service or ignored fixture recovers). Delimit attempts explicitly so that
398
+ // the earlier red cannot remain latched into the later complete green cycle.
399
+ journal.append("tip-verify-start", undefined, { tip, cmdHash, gates: Object.keys(commands), cached: clean && cached });
400
+ if (clean && cached) {
401
+ journal.append("tip-verify-cached", undefined, { tip, cmdHash, gates: Object.keys(commands) });
402
+ // The skip must not read as a red. Every surface derives the tip's verdict from this cycle's
403
+ // `tip-verify` events (cockpit derive.ts tipVerificationPassed: a run-end claiming "passed" with
404
+ // ZERO events is fail-closed to FALSE), so a carried-forward green still journals its per-gate
405
+ // pass — `cached: true` keeps it honest about not having re-run the command.
406
+ for (const gate of Object.keys(commands)) {
407
+ journal.append("tip-verify", undefined, { gate, cmd: commands[gate], pass: true, exitCode: 0, cached: true, tip, cmdHash });
408
+ }
409
+ return false;
410
+ }
411
+ let tipFailed = false;
412
+ for (const r of await verifyIntegrationTip(intWt, commands, journal.dir)) {
413
+ if (r.pass) {
414
+ journal.append("tip-verify", undefined, { gate: r.gate, cmd: r.cmd, pass: true, exitCode: r.exitCode, details: r.details, tip, cmdHash });
415
+ }
416
+ else {
417
+ journal.append("tip-verify-failed", undefined, {
418
+ gate: r.gate,
419
+ cmd: r.cmd,
420
+ exitCode: r.exitCode,
421
+ fingerprints: r.fingerprints,
422
+ artifact: r.artifact,
423
+ lastMergedTask: opts.lastMergedTask,
424
+ tip,
425
+ cmdHash,
426
+ });
427
+ tipFailed = true;
428
+ }
429
+ }
430
+ return tipFailed;
431
+ }
432
+ // `ps` CPU time: "[[dd-]hh:]mm:ss[.frac]" (darwin prints "0:00.03", linux "00:00:01", both print
433
+ // "1-02:03:04" past a day). Anything else is a header or a row this parser must not guess at.
434
+ // `frac` reports whether THIS host prints sub-second digits — the quantum the flat window is sized
435
+ // against, measured rather than assumed (a darwin sample is 10ms, a linux one 1000ms).
436
+ function parsePsCpu(raw) {
437
+ const m = /^(?:(\d+)-)?(?:(\d+):)?(\d+):(\d+(?:\.\d+)?)$/.exec(raw);
438
+ if (!m)
439
+ return undefined;
440
+ const ms = ((Number(m[1] ?? 0) * 24 + Number(m[2] ?? 0)) * 60 + Number(m[3])) * 60_000 + Math.round(Number(m[4]) * 1000);
441
+ return { ms, frac: m[4].includes(".") };
442
+ }
443
+ let linuxClockTickMs;
444
+ function linuxProcessCpuMs(pid, cwd) {
445
+ if (!existsSync("/proc/self/stat"))
446
+ return Promise.resolve(undefined);
447
+ // shGit, not sh: the accountant samples this path at a 100ms cadence, and a LOGIN shell would
448
+ // re-run the operator's profile (nvm/pyenv/direnv side effects included) on every sample.
449
+ linuxClockTickMs ??= shGit("getconf CLK_TCK", cwd, 15_000).then((r) => {
450
+ const ticks = r.code === 0 ? Number(r.stdout.trim()) : Number.NaN;
451
+ return Number.isFinite(ticks) && ticks > 0 ? 1_000 / ticks : undefined;
452
+ });
453
+ return linuxClockTickMs.then((resolutionMs) => {
454
+ if (resolutionMs === undefined)
455
+ return undefined;
456
+ try {
457
+ // `/proc/<pid>/stat` fields 14-17 are user/system jiffies for the process and its waited-for
458
+ // children. The child totals retain tools that start and exit wholly between live-tree polls.
459
+ // Split after the LAST ')' because comm may contain spaces or parentheses; field 3 is rest[0].
460
+ const stat = readFileSync(`/proc/${pid}/stat`, "utf8");
461
+ const fields = stat.slice(stat.lastIndexOf(")") + 2).trim().split(/\s+/);
462
+ const ticks = Number(fields[11]) + Number(fields[12]) + Number(fields[13]) + Number(fields[14]);
463
+ return Number.isFinite(ticks) ? { ms: ticks * resolutionMs, resolutionMs } : undefined;
464
+ }
465
+ catch {
466
+ return undefined; // process exited between ps ancestry capture and the precise CPU read
467
+ }
468
+ });
469
+ }
470
+ // T2 (OBS-264): the triad's CPU leg. Every non-seeded process of an attempt descends from that
471
+ // attempt's own dispatch script, whose path is unique — print, argv-interactive and resume launches
472
+ // all start there. (interactive-seed is intentionally fail-open below because its adapter-owned
473
+ // launch bypasses this script.) ONE `ps` snapshot finds the root and all current descendants:
474
+ // the agent CLI is a CHILD of the script's shell, so the root's own TIME never moves while the CLI
475
+ // thinks. `resolutionMs` is the sampled clock's quantum, which sizes the caller's flat window.
476
+ // Returns 0 when nothing matches: a worker whose process tree is gone is the strongest possible
477
+ // "not working". Returns undefined when the snapshot itself failed or parsed to nothing —
478
+ // unmeasurable CPU is never evidence a worker stopped, and the caller refuses to conclude on it.
479
+ async function workerTreeCpuSnapshot(marker, cwd) {
480
+ // shGit, not sh: same login-shell cost as the CLK_TCK probe above — `ps` needs no profile.
481
+ const snapshot = await shGit("ps -Awwo pid=,ppid=,time=,command=", cwd, 15_000);
482
+ if (snapshot.code !== 0)
483
+ return undefined;
484
+ const rows = [];
485
+ for (const line of snapshot.stdout.split("\n")) {
486
+ const m = /^\s*(\d+)\s+(\d+)\s+(\S+)\s+(.*)$/.exec(line);
487
+ if (!m)
488
+ continue;
489
+ const cpu = parsePsCpu(m[3]);
490
+ if (cpu !== undefined)
491
+ rows.push({ pid: m[1], ppid: m[2], cpuMs: cpu.ms, frac: cpu.frac, cmd: m[4] });
492
+ }
493
+ if (rows.length === 0)
494
+ return undefined;
495
+ const tree = new Set(rows.filter((p) => p.cmd.includes(marker)).map((p) => p.pid));
496
+ // `ps` output is not topologically ordered — relax the parent→child closure until it stops growing.
497
+ for (let grew = true; grew;) {
498
+ grew = false;
499
+ for (const p of rows) {
500
+ if (!tree.has(p.pid) && tree.has(p.ppid)) {
501
+ tree.add(p.pid);
502
+ grew = true;
503
+ }
504
+ }
505
+ }
506
+ const precise = new Map();
507
+ let preciseResolutionMs;
508
+ for (const p of rows) {
509
+ if (!tree.has(p.pid))
510
+ continue;
511
+ const cpu = await linuxProcessCpuMs(p.pid, cwd);
512
+ precise.set(p.pid, cpu?.ms ?? p.cpuMs);
513
+ if (cpu !== undefined)
514
+ preciseResolutionMs = cpu.resolutionMs;
515
+ }
516
+ // Even an empty worker tree needs the host's actual measurement quantum: on Linux the /proc
517
+ // jiffy clock remains available after the worker exits, while `ps time` only prints whole seconds.
518
+ if (preciseResolutionMs === undefined && existsSync("/proc/self/stat")) {
519
+ preciseResolutionMs = (await linuxProcessCpuMs(String(process.pid), cwd))?.resolutionMs;
520
+ }
521
+ return {
522
+ processes: precise,
523
+ resolutionMs: preciseResolutionMs ?? (rows.some((p) => p.frac) ? 10 : 1_000),
524
+ };
525
+ }
526
+ export async function workerTreeCpuMs(marker, cwd) {
527
+ const snapshot = await workerTreeCpuSnapshot(marker, cwd);
528
+ if (snapshot === undefined)
529
+ return undefined;
530
+ return {
531
+ ms: [...snapshot.processes.values()].reduce((sum, cpuMs) => sum + cpuMs, 0),
532
+ resolutionMs: snapshot.resolutionMs,
533
+ };
534
+ }
535
+ // Sparse live-tree totals forget a tool's CPU as soon as that tool exits. This attempt-local
536
+ // accountant instead adds each observed process's CPU DELTA to a monotonic total and replaces only
537
+ // the live-PID cursor on each sample. When a PID disappears, its contribution stays in `totalMs`;
538
+ // if that PID is later reused, its fresh total is added from zero because it left `live` in between.
539
+ class WorkerTreeCpuAccountant {
540
+ marker;
541
+ cwd;
542
+ active = false;
543
+ loop;
544
+ live = new Map();
545
+ totalMs = 0;
546
+ gaps = 0;
547
+ consecutiveGaps = 0;
548
+ latest;
549
+ constructor(marker, cwd) {
550
+ this.marker = marker;
551
+ this.cwd = cwd;
552
+ }
553
+ async sample() {
554
+ const snapshot = await workerTreeCpuSnapshot(this.marker, this.cwd);
555
+ if (snapshot === undefined) {
556
+ this.gaps++;
557
+ this.live.clear();
558
+ this.latest = undefined;
559
+ // Stop forking `ps` at 10Hz once the host has proved it cannot answer — see the cap's comment.
560
+ if (++this.consecutiveGaps >= HARVEST_CPU_UNMEASURABLE_SAMPLE_CAP)
561
+ this.active = false;
562
+ return;
563
+ }
564
+ this.consecutiveGaps = 0;
565
+ for (const [pid, cpuMs] of snapshot.processes) {
566
+ const prior = this.live.get(pid);
567
+ this.totalMs += prior === undefined || cpuMs < prior ? cpuMs : cpuMs - prior;
568
+ }
569
+ this.live = snapshot.processes;
570
+ this.latest = { ms: this.totalMs, resolutionMs: snapshot.resolutionMs };
571
+ }
572
+ async start() {
573
+ if (this.active)
574
+ return;
575
+ this.active = true;
576
+ await this.sample();
577
+ this.loop = (async () => {
578
+ while (this.active) {
579
+ await new Promise((resolve) => setTimeout(resolve, HARVEST_CPU_ACCOUNTING_POLL_MS));
580
+ if (this.active)
581
+ await this.sample();
582
+ }
583
+ })();
584
+ }
585
+ read() {
586
+ return { cpu: this.latest, gaps: this.gaps };
587
+ }
588
+ async stop() {
589
+ this.active = false;
590
+ await this.loop;
591
+ }
592
+ }
119
593
  async function commitsAheadOf(base, wt) {
120
594
  const head = await gitHead(wt);
121
595
  if (head === base)
@@ -125,6 +599,26 @@ async function commitsAheadOf(base, wt) {
125
599
  return [];
126
600
  return r.stdout.trim().split("\n").filter(Boolean);
127
601
  }
602
+ // T1 (R1 fast-kill): worktree delta = commits ahead of the task base OR any uncommitted change.
603
+ // The fast-kill may only run once the ABSENCE of work has been positively established, so every
604
+ // probe fails OPEN toward "delta": a throwing rev-parse, a non-zero `git log`, and a non-zero
605
+ // `git status` all report delta. commitsAheadOf() is deliberately NOT reused here — it converts a
606
+ // failed log into an empty list, which reads as "no commits" and would kill a live worker.
607
+ async function worktreeHasDelta(wt, base) {
608
+ try {
609
+ const head = await gitHead(wt);
610
+ if (head !== base) {
611
+ const log = await shGit(`git log --reverse --format=%H ${shq(base)}..${shq(head)}`, wt);
612
+ if (log.code !== 0 || log.stdout.trim().length > 0)
613
+ return true;
614
+ }
615
+ const r = await shGit("GIT_OPTIONAL_LOCKS=0 git status --porcelain", wt);
616
+ return r.code !== 0 || r.stdout.trim().length > 0;
617
+ }
618
+ catch {
619
+ return true; // an unreadable worktree is never evidence that the worker did nothing
620
+ }
621
+ }
128
622
  async function cherryPickCommits(wt, commits) {
129
623
  const carried = [];
130
624
  for (const hash of commits) {
@@ -295,7 +789,12 @@ export async function runDaemon(repoRoot, opts = {}) {
295
789
  for (const [id, st] of journal.replayStatuses()) {
296
790
  if (opts.retryFailed && st === "failed" && recordedTaskFailureKind(replayEvents, id) === "dispatch") {
297
791
  graph = setStatus(graph, id, "pending");
298
- resume.delete(id);
792
+ // OBS-254: clear ATTEMPT AND CHANNEL STATE ONLY. Deleting the whole entry also deleted
793
+ // upheldFeedback — the operator's funded brief — and the next dispatch advertised an empty
794
+ // "fix these specifically" heading. A dispatch that died before worker-result then re-ran the
795
+ // worker with the uphold's findings silently gone.
796
+ const prior = resume.get(id);
797
+ resume.set(id, { attempts: 0, tried: [], ...(prior?.upheldFeedback ? { upheldFeedback: prior.upheldFeedback } : {}) });
299
798
  continue;
300
799
  }
301
800
  // operator release: a graph.json edit back to "pending" beats a replayed human/failed park (locked decision 12)
@@ -485,7 +984,7 @@ export async function runDaemon(repoRoot, opts = {}) {
485
984
  // reviewer-exclusion list (badReviewers), never a second parallel counter. OBS-189: scoped to the
486
985
  // current engagement — an operator approval (uphold or accept) resets the round budget, so an upheld
487
986
  // task can dispatch its funded attempt instead of re-parking against the whole journal's history.
488
- const reviewRoundsDrawn = () => reviewRoundsSinceApproval(journal.read(), t.id);
987
+ const reviewRoundsDrawn = () => reviewRoundsSinceApproval(decisiveReviewRounds(journal.read()), t.id);
489
988
  // OBS-193: journal the in-gate review retry (mirrors judge-retry) and exclude the flaked seat from
490
989
  // later attempts' reviewer picks. One helper, called from both onGate sites (satisfied-gate + main).
491
990
  const noteReviewRetry = (g) => {
@@ -498,9 +997,85 @@ export async function runDaemon(repoRoot, opts = {}) {
498
997
  badReviewers.push(rr.flaked);
499
998
  }
500
999
  };
1000
+ // v1.85 T3 (ruling R4): every BLOCKING review/judge result lands its findings in the journal
1001
+ // structured — class + canonical path + stable symbol — so a retry, a consult or an auto-uphold
1002
+ // decision reads identity instead of re-parsing prose, and line-number churn is not a new finding.
1003
+ // One helper, both onGate sites (satisfied-gate resume + main attempt loop).
1004
+ const journalGateResult = (g) => {
1005
+ const blocking = gateFailed(g) && (g.gate === "review" || g.gate === "acceptance");
1006
+ // R3 (OBS-186): a gate that DECLINED has no verdict to state, and this row is the ONE seam every
1007
+ // fold outside this file shares. Writing `pass: false` for a decline is what turned a skip into
1008
+ // a failure at all of them at once — the engagement round budget (reviewRoundsSinceApproval,
1009
+ // journal.ts), the operator's failed-gate list (cli/commands/approve.ts), the record's
1010
+ // gate-failure total (cli/commands/report.ts), the cockpit's gate rows (tui/cockpit/derive.ts).
1011
+ // Each keys on `pass === false`; none of them is reachable from this task's file scope, and
1012
+ // patching five copies of the same question would be the wrong fix even if they were. So the
1013
+ // ledger simply does not claim a verdict it does not have.
1014
+ // The legacy baseline declines (a build command the repo never configured) have always written
1015
+ // `pass: true` beside `skipped: true` and every consumer already reads them right, so their row
1016
+ // is untouched: only a decline that would otherwise be recorded RED changes shape here.
1017
+ const unverdicted = g.meta?.skipped === true && !g.pass;
1018
+ journal.append("gate-result", t.id, {
1019
+ gate: g.gate, ...(unverdicted ? {} : { pass: g.pass }), details: g.details,
1020
+ ...(g.meta?.skipped === true ? { skipped: true } : {}),
1021
+ // R3 (OBS-186): a declined review is journal truth, not an absence. `skipped: true` alone
1022
+ // says a gate did not run; these say WHICH policy declined it and WHY, so a reader of the
1023
+ // ledger never has to infer participation from a details string. The green-skip branch that
1024
+ // made this row indistinguishable from a pass is gone (src/gates/review.ts).
1025
+ ...(g.meta?.verdict === "skipped"
1026
+ ? { verdict: "skipped", policy: g.meta.policy, reason: g.meta.reason }
1027
+ : {}),
1028
+ // T4 (OBS-265): a test verdict says WHICH suite spoke. A round runs the selected subset as a
1029
+ // screen and the full suite as the verdict, so without these two the journal would carry a
1030
+ // `test` row whose scope no consumer could recover.
1031
+ ...(Array.isArray(g.meta?.selectedTests) ? { selectedTests: g.meta.selectedTests } : {}),
1032
+ ...(g.meta?.fullSuite === true ? { fullSuite: true } : {}),
1033
+ // A finding's path is its own evidence path. Do not pass task scope here: a declaration says
1034
+ // where work is allowed, not where this verdict found the defect.
1035
+ ...(blocking ? { findings: structuredFindings(g.gate, g.details) } : {}),
1036
+ });
1037
+ };
1038
+ // R3 (OBS-186): judge ‖ review are launched together and publish in COMPLETION order
1039
+ // (run-gates.ts) — a race. Three oracles assert the opposite: a scripted run's journal is
1040
+ // byte-identical run to run (tests/run/narration.test.ts, tests/run/notify-identity.test.ts), and
1041
+ // a round's gate-result order matches its phase-start order (tests/run/daemon.test.ts). Retiring
1042
+ // complexityThreshold is what REACHES this, not what introduces it: those fixtures used to skip
1043
+ // review and journal ONE verdict row per round, so the pair's order was never exercised — and the
1044
+ // operator's config has run `complexityThreshold: 0` since 2026-07-31, so production rounds have
1045
+ // journaled both siblings all along.
1046
+ //
1047
+ // Ordering is the LEDGER's job, not the pipeline's. run-gates still reports each completion the
1048
+ // instant it happens; the daemon writes its ledger in GATE_NAMES order. Only the LATER gate is
1049
+ // ever held, and only while an earlier sibling is still in flight — a review that finishes first
1050
+ // waits for acceptance, never the reverse. That keeps T4's durability where it pays (the first
1051
+ // verdict to land is still published immediately) and bounds the exposure to one row for the
1052
+ // remainder of one already-running gate. A gate that THROWS kills the round before merge, so a
1053
+ // row held behind it is lost with the round it belonged to — not a verdict that could have merged.
1054
+ const parallelPending = new Set();
1055
+ let heldParallel;
1056
+ const notePhaseStart = (e) => {
1057
+ if (e.parentAt !== undefined)
1058
+ parallelPending.add(e.gate);
1059
+ };
1060
+ const inParallelOrder = (gate, publish) => {
1061
+ parallelPending.delete(gate);
1062
+ const rank = GATE_NAMES.indexOf(gate);
1063
+ if ([...parallelPending].some((p) => GATE_NAMES.indexOf(p) < rank)) {
1064
+ heldParallel = publish;
1065
+ return;
1066
+ }
1067
+ publish();
1068
+ const held = heldParallel;
1069
+ heldParallel = undefined;
1070
+ held?.();
1071
+ };
501
1072
  // OBS-189: the operator upheld the reviewer — the findings ARE the brief for this funded attempt.
502
- let feedback = rs?.upheldFeedback
503
- ? `The operator UPHELD the reviewer's findings — address them without discarding landed work.\nreview: ${rs.upheldFeedback}`
1073
+ // OBS-254: RE-DERIVED from the journal here, at prompt-build time, rather than trusted to survive
1074
+ // in resume state. The journal already holds the upheld review's bytes; no reset of attempt or
1075
+ // channel state can take them away, on any path, including `resume --retry-failed`.
1076
+ const upheldFeedback = upheldFeedbackByTask(journal.read()).get(t.id) ?? rs?.upheldFeedback;
1077
+ let feedback = upheldFeedback
1078
+ ? `The operator UPHELD the reviewer's findings — address them without discarding landed work.\nreview: ${upheldFeedback}`
504
1079
  : "";
505
1080
  let ladderIdx = 0;
506
1081
  let modeFallbackNoted = false; // v1.2: journal the interactive→print fallback once per task, not per attempt
@@ -509,6 +1084,12 @@ export async function runDaemon(repoRoot, opts = {}) {
509
1084
  let tokens; // SPEND-02: accumulated across attempts — parked spend is still spend
510
1085
  let metered = 0; // SPEND-02: attempts that returned a usage record; distinguishes unmetered from measured-zero
511
1086
  let tipMoves = 0; // OBS-15: one re-gate allowance per task, never reset by a worker retry
1087
+ // T4 (OBS-265): a task whose test gate has failed once stops selecting tests for good — the
1088
+ // selector already proved it cannot speak for this diff, so every later round runs full. SEEDED
1089
+ // FROM THE JOURNAL, not from process memory: a park, a `resume`, or a daemon restart must not
1090
+ // hand the selector a clean slate it did not earn, and the journal is the only state that
1091
+ // survives all three (same read as upheldFeedbackByTask above).
1092
+ let testGateFailed = journal.read().some((e) => e.event === "gate-result" && e.taskId === t.id && e.data.gate === "test" && e.data.pass === false);
512
1093
  let retryMode = "fresh";
513
1094
  let lastContextTokens; // v1.23 reset signal, including stalled/quota attempts
514
1095
  // v1.29: only a gate-failed attempt can seed same-session retry. The next attempt consumes this
@@ -568,7 +1149,14 @@ export async function runDaemon(repoRoot, opts = {}) {
568
1149
  });
569
1150
  await driver.notify(`tickmarkr ${runId}: ${t.id} consult verdict: ${v.action}`, { tier: "attention" });
570
1151
  if (v.action === "retry") {
571
- feedback = renderRetryGuidance(v) || feedback;
1152
+ // The guidance is ADDED to the brief, never swapped for it: the failure bytes the journal
1153
+ // already holds are the one thing the next attempt cannot rediscover for free. The
1154
+ // fingerprint cap's ban on an identical retry is NOT enforced here — a verdict is only one
1155
+ // of several ways this task reaches a re-dispatch, so the ban is enforced at the dispatch
1156
+ // seam every one of them passes through (see enforceRetryBan).
1157
+ const guidance = renderRetryGuidance(v);
1158
+ if (guidance && !feedback.includes(guidance))
1159
+ feedback = feedback ? `${feedback}\n\n${guidance}` : guidance;
572
1160
  return true;
573
1161
  }
574
1162
  if (v.action === "reroute") {
@@ -657,7 +1245,36 @@ export async function runDaemon(repoRoot, opts = {}) {
657
1245
  };
658
1246
  const gateAuthor = rs?.lastAssignment ?? assignment;
659
1247
  const satisfiedIndex = GATE_NAMES.indexOf(satisfiedGate);
660
- const remainingGates = t.gates.filter((gate) => GATE_NAMES.indexOf(gate) > satisfiedIndex);
1248
+ // The serial pipeline could have at most one blocking result, so "everything after the
1249
+ // approved gate" was enough. v1.85 can record both verdict siblings red in one round, and a
1250
+ // selected test screen can be green without a complete suite. Approval waives exactly its
1251
+ // named gate: every other red from that round is re-run, and test is forced unless the prior
1252
+ // journal proves a full suite completed.
1253
+ const priorEvents = journal.read();
1254
+ let priorRoundStart = -1;
1255
+ for (let i = priorEvents.length - 1; i >= 0; i--) {
1256
+ const e = priorEvents[i];
1257
+ if (e.event === "phase-start" && e.taskId === t.id && e.data.phase === "gates") {
1258
+ priorRoundStart = i;
1259
+ break;
1260
+ }
1261
+ }
1262
+ const priorResults = new Map();
1263
+ for (const e of priorEvents.slice(priorRoundStart + 1)) {
1264
+ if (e.event !== "gate-result" || e.taskId !== t.id || typeof e.data.gate !== "string"
1265
+ || !GATE_NAMES.includes(e.data.gate))
1266
+ continue;
1267
+ priorResults.set(e.data.gate, e);
1268
+ }
1269
+ const remainingGates = t.gates.filter((gate) => {
1270
+ if (gate === satisfiedGate)
1271
+ return false;
1272
+ const prior = priorResults.get(gate)?.data;
1273
+ const followsApproved = GATE_NAMES.indexOf(gate) > satisfiedIndex;
1274
+ const otherRed = prior?.pass === false;
1275
+ const needsFullSuite = gate === "test" && prior?.fullSuite !== true;
1276
+ return followsApproved || otherRed || needsFullSuite;
1277
+ });
661
1278
  const resumedTask = { ...t, gates: remainingGates };
662
1279
  gateLoop: while (true) {
663
1280
  const gated = await gitHead(wt);
@@ -665,6 +1282,8 @@ export async function runDaemon(repoRoot, opts = {}) {
665
1282
  const { results } = await runGates(resumedTask, {
666
1283
  worktree: wt, baseRef: taskBase, result: priorResult, author: gateAuthor,
667
1284
  commands, baseline, channels, adapters, cfg, artifactDir: journal.dir,
1285
+ // a recheck re-verifies a human's release: it never selects tests down, it runs the suite.
1286
+ pipeline: "v185",
668
1287
  via: cfg.visibility.llm === "pane"
669
1288
  ? {
670
1289
  driver: trackedDriver,
@@ -677,25 +1296,25 @@ export async function runDaemon(repoRoot, opts = {}) {
677
1296
  excludeReviewers: badReviewers,
678
1297
  onGate: async (e) => {
679
1298
  if (e.phase === "start") {
680
- journal.phaseStart(t.id, phaseForGate(e.gate), { gate: e.gate, index: e.index, total: e.total });
1299
+ notePhaseStart(e);
1300
+ journal.phaseStart(t.id, phaseForGate(e.gate), { gate: e.gate, index: e.index, total: e.total, ...(e.parentAt === undefined ? {} : { parallel: true }) });
681
1301
  return;
682
1302
  }
683
1303
  const g = e.result;
684
- journal.append("gate-result", t.id, {
685
- gate: g.gate, pass: g.pass, details: g.details,
686
- ...(g.meta?.skipped === true ? { skipped: true } : {}),
1304
+ inParallelOrder(g.gate, () => {
1305
+ journalGateResult(g);
1306
+ noteReviewRetry(g);
1307
+ if (g.gate === "review" && !g.pass && /unparseable/.test(g.details)
1308
+ && typeof g.meta?.reviewer === "string") {
1309
+ badReviewers.push(g.meta.reviewer);
1310
+ }
687
1311
  });
688
- noteReviewRetry(g);
689
- if (g.gate === "review" && !g.pass && /unparseable/.test(g.details)
690
- && typeof g.meta?.reviewer === "string") {
691
- badReviewers.push(g.meta.reviewer);
692
- }
693
1312
  },
694
1313
  });
695
1314
  const approvedCommits = await commitsAheadOf(taskBase, wt);
696
1315
  graph = addEvidence(graph, t.id, { commits: approvedCommits, gateResults: results });
697
1316
  saveGraph(repoRoot, graph);
698
- if (!results.every((g) => g.pass)) {
1317
+ if (!results.every(gateSatisfied)) {
699
1318
  gateFails++;
700
1319
  await park(t, "post-approval gate failed", "gate-fail", gateAuthor, rs?.attempts ?? 0, startMs, gateFails, consults, tokens, metered, retryMode);
701
1320
  return;
@@ -772,15 +1391,58 @@ export async function runDaemon(repoRoot, opts = {}) {
772
1391
  // over-threshold context still forces fresh, and the escalation ladder bounds the chain.
773
1392
  const priorSession = retrySession;
774
1393
  retrySession = undefined;
1394
+ // v1.85 T3: the fingerprint cap banned an IDENTICAL retry — this task+gate already produced the
1395
+ // same failure bytes twice, so re-running the same channel on the same brief is a paid
1396
+ // re-measurement of an answer the journal already holds. Enforced HERE, at the one seam every
1397
+ // re-dispatch passes through, and NOT at the consult verdict that set it: a terminal verdict
1398
+ // falling through to the cap's own `retry` rung, a review-fix round and the ladder itself all
1399
+ // reach a dispatch without ever consulting again, and worker-launch below would then expire a
1400
+ // ban nothing had honoured. Bound to the channel the cap fired on, so a move that already went
1401
+ // elsewhere is not refused for a ban that was never about it; with no untried channel left the
1402
+ // task parks naming the ban rather than buying a third round.
1403
+ const banned = activeRetryBan(journal.read(), t.id, channelKey(assignment));
1404
+ if (banned) {
1405
+ const next = failover("escalate");
1406
+ journal.append("retry-same-banned", t.id, {
1407
+ gate: banned, from: channelKey(assignment), to: next ? channelKey(next) : null,
1408
+ });
1409
+ if (!next) {
1410
+ await park(t, `identical ${banned} failure twice this engagement — an identical retry is banned and no untried channel is left`, "gate-fail", assignment, attempt, startMs, gateFails, consults, tokens, metered, retryMode);
1411
+ return;
1412
+ }
1413
+ assignment = next;
1414
+ if (!tried.includes(channelKey(next)))
1415
+ tried.push(channelKey(next));
1416
+ }
775
1417
  const retryAdapter = adapters.find((a) => a.id === assignment.adapter);
776
- retryMode = priorSession
777
- && priorSession.channel === channelKey(assignment)
778
- && (priorSession.contextTokens !== undefined
779
- ? priorSession.contextTokens < cfg.contextWarnTokens
780
- : retryAdapter?.resumeUnknownContext === true)
781
- && retryAdapter?.resumeCommand
782
- ? "resume"
783
- : "fresh";
1418
+ // v1.85 T3: a repair attempt is dispatched FRESH by construction — the whole point is that the
1419
+ // brief, not a surviving session, carries the findings and the diff. It therefore outranks the
1420
+ // session-resume choice, and the mode is journaled so the ledger can price repairs against
1421
+ // fresh re-dispatches. Read from the JOURNAL, before this dispatch's own event lands: a run that
1422
+ // stopped between funding the repair and sending it resumes still carrying the findings.
1423
+ const journaledSoFar = journal.read(); // read BEFORE this dispatch's own event lands
1424
+ let repairFindings = pendingRepairFindings(journaledSoFar, t.id);
1425
+ // OBS-254, one layer below the upheld brief: the ordinary gate-fail brief was loop-local, so any
1426
+ // path that rebuilt this task's state (a resume, `--retry-failed`, a fresh daemon) dispatched a
1427
+ // retry that had forgotten why it was retrying. Re-derived from the journal here, at prompt-build
1428
+ // time, and MERGED rather than substituted — a retry never discards what the journal already
1429
+ // holds. Row-wise, because the live brief may already quote one of them (a delivery-readiness
1430
+ // failure the loop just wrote) and repeating it helps no worker.
1431
+ const journaledRows = journaledFailureBrief(journaledSoFar, t.id).filter((row) => !feedback.includes(row));
1432
+ if (journaledRows.length > 0) {
1433
+ const brief = journaledRows.join("\n\n");
1434
+ feedback = feedback ? `${brief}\n\n${feedback}` : brief;
1435
+ }
1436
+ retryMode = repairFindings
1437
+ ? "repair"
1438
+ : priorSession
1439
+ && priorSession.channel === channelKey(assignment)
1440
+ && (priorSession.contextTokens !== undefined
1441
+ ? priorSession.contextTokens < cfg.contextWarnTokens
1442
+ : retryAdapter?.resumeUnknownContext === true)
1443
+ && retryAdapter?.resumeCommand
1444
+ ? "resume"
1445
+ : "fresh";
784
1446
  // v1.23 T3: over-threshold context still forces fresh at the retry boundary; never interrupt a
785
1447
  // running attempt. Unknown/below emits no reset event.
786
1448
  if (attempt > 0 && lastContextTokens !== undefined && lastContextTokens >= cfg.contextWarnTokens) {
@@ -808,6 +1470,15 @@ export async function runDaemon(repoRoot, opts = {}) {
808
1470
  carriedCommits = await cherryPickCommits(wt, commitsToCarry);
809
1471
  journal.append("worktree-recreation", t.id, { attempted: commitsToCarry, carried: carriedCommits });
810
1472
  }
1473
+ // T2 review (material): harvest eligibility is "does this WORKTREE carry unverified work",
1474
+ // measured against taskBase — the same base the fast-kill's delta probe and the gates
1475
+ // themselves use. It was measured against this attempt's post-carry HEAD, which excluded
1476
+ // every commit cherry-picked forward: attempt 0 commits and walls, attempt 1 receives that
1477
+ // commit and goes silent, and the silent retry — whose worktree already held the whole
1478
+ // deliverable — was NOT harvested, took the stall consult, and could be redispatched to
1479
+ // re-produce it. The routing branches that "starting HEAD" was protecting no longer need it:
1480
+ // quota, dead-channel and provider-death all classify the PRE-HARVEST outcome below and fire
1481
+ // BEFORE the synthesis, so carried-only work reaches gates without bypassing any failover.
811
1482
  const priorNamed = [...new Set([...commitsToCarry, ...carriedCommits])];
812
1483
  const presentCommits = new Set(carriedCommits);
813
1484
  for (const h of commitsToCarry) {
@@ -836,6 +1507,36 @@ export async function runDaemon(repoRoot, opts = {}) {
836
1507
  });
837
1508
  await driver.notify(`tickmarkr ${runId}: ${t.id} lost ${lostCommits.length} of ${commitsToCarry.length} landed commit(s) recreating its worktree — it will re-do that work`, { tier: "attention" });
838
1509
  }
1510
+ // v1.85 T3: "fully carried commits" is a repair PRECONDITION, and it is re-validated here —
1511
+ // the eligibility test ran one attempt ago, but the carry that decides it happens above, on
1512
+ // this dispatch. A tree that lost part of the prior attempt's work cannot be repaired: a
1513
+ // fix-only contract ("do NOT re-implement that work") over a diff whose implementation is
1514
+ // missing would have the worker patch an incomplete tree and forbid it from rebuilding the
1515
+ // rest. The fresh ladder owns this dispatch instead. `retryMode` is corrected before
1516
+ // worker-launch records it, so the ledger's launch event names what the worker actually got.
1517
+ if (repairFindings !== undefined && lostCommits.length > 0) {
1518
+ journal.append("repair-cancelled", t.id, { reason: "carry incomplete", attempted: commitsToCarry, lost: lostCommits });
1519
+ repairFindings = undefined;
1520
+ retryMode = "fresh";
1521
+ }
1522
+ // v1.85 T3: a repair dispatch replaces the bare gate-fail brief with a fix-only contract that
1523
+ // carries the failing findings VERBATIM and the diff CONTENT of the work already in this
1524
+ // worktree. The measured loss it removes: 62 of 68 re-dispatches were fresh, each re-buying
1525
+ // ~20m of onboarding to rediscover a diff and a finding the journal already held.
1526
+ // The diff is measured HERE, from the worktree the worker will actually open, after the carry —
1527
+ // never from the pre-recreation tree, so what the brief quotes is what the worker has.
1528
+ if (repairFindings) {
1529
+ const raw = await shGit(`git diff ${shq(taskBase)}..HEAD`, wt);
1530
+ const cap = cfg.gates.diffCap ?? DEFAULT_DIFF_CAP; // same fallback the measuring gates use
1531
+ const diff = raw.stdout.length > cap
1532
+ ? `${raw.stdout.slice(0, cap)}\n… diff truncated at gates.diffCap (${cap} bytes)`
1533
+ : raw.stdout;
1534
+ const brief = repairBrief(repairFindings, diff, taskBase);
1535
+ // anything the live brief holds beyond the journaled findings (a consult's guidance) is kept:
1536
+ // a repair adds the diff and the fix-only contract, it never subtracts what was already known.
1537
+ feedback = feedback && !brief.includes(feedback) ? `${brief}\n\n${feedback}` : brief;
1538
+ journal.append("repair-dispatch", t.id, { diffBytes: diff.length, capped: raw.stdout.length > cap });
1539
+ }
839
1540
  if (feedback || priorNamed.length > 0) {
840
1541
  feedback = augmentRetryBrief(feedback, { attempted: commitsToCarry, carried: carriedCommits, present: presentCommits });
841
1542
  }
@@ -892,6 +1593,90 @@ export async function runDaemon(repoRoot, opts = {}) {
892
1593
  workerCmd,
893
1594
  exitMarkerCmd,
894
1595
  ].join("\n"));
1596
+ // T2 (OBS-264): the liveness triad, shared by BOTH wait loops (a headless worker stalls on
1597
+ // finished work exactly as a visible one does — and rode the whole window before this). It
1598
+ // sits ABOVE the fast-kill and the nudge because the population it governs is the opposite
1599
+ // one: the kill condemns a pane holding NOTHING, while every one of the 18 observed stalls
1600
+ // held 2-33 commits that the redispatch then re-bought. Concluding is not killing — the
1601
+ // post-loop tail harvests this attempt exactly as a window expiry does, and the carried
1602
+ // worktree goes to gates. The probe runs only once the tracker is ALREADY silent, so a
1603
+ // working worker never pays for it, and an unreadable snapshot RESETS the observation:
1604
+ // unmeasurable CPU is never evidence a worker stopped.
1605
+ // v1.85 T3: the prompt has now actually reached a worker. The journal-derived retry decisions (a
1606
+ // funded repair's findings, an identical-retry ban) expire HERE and nowhere earlier: everything
1607
+ // between task-dispatch and this line — worktree recreation, setup, prompt write, slot allocation,
1608
+ // the launch itself — can still die with no worker having seen the brief, and `--retry-failed`
1609
+ // must then re-send that same brief rather than a fresh prompt on a possibly banned channel.
1610
+ const noteLaunched = () => journal.append("worker-launch", t.id, { attempt, retryMode });
1611
+ let cpuFlat;
1612
+ let cpuAccountant;
1613
+ let cpuGapCount = 0;
1614
+ let unmeasurableNoted = false;
1615
+ // One line per attempt, whichever way the CPU leg turns out to be unmeasurable. A triad that
1616
+ // can never conclude is this feature silently ABSENT — on a host whose `ps` the probe cannot
1617
+ // read, every stall would ride its whole window out again with nothing saying why. Named once
1618
+ // per attempt, not per slice: the condition is structural, and a per-slice line would bury it.
1619
+ const noteUnmeasurable = (reason) => {
1620
+ if (unmeasurableNoted)
1621
+ return;
1622
+ unmeasurableNoted = true;
1623
+ journal.append("worker-harvest-unmeasurable", t.id, { slot: slot.name, attempt, reason });
1624
+ };
1625
+ const harvestConcludes = async (silentMs) => {
1626
+ if (silentMs < harvestSilentMs) {
1627
+ await cpuAccountant?.stop();
1628
+ cpuAccountant = undefined;
1629
+ cpuFlat = undefined;
1630
+ cpuGapCount = 0;
1631
+ return false;
1632
+ }
1633
+ // The CPU leg needs a marker in the worker's own argv, and every launch path puts this
1634
+ // attempt's dispatch script there EXCEPT interactiveSeed: runInteractiveSeed launches the
1635
+ // TUI directly, by a command the ADAPTER owns (seed.launch(model)) which tickmarkr cannot
1636
+ // make attempt-unique and must deliver verbatim. A marker that matches nothing reads as
1637
+ // zero CPU — precisely the false "flat" that would harvest a worker mid-turn — so a seeded
1638
+ // attempt has no measurable CPU leg and the triad never concludes it. The other half of
1639
+ // OBS-264 is untouched there: when its window does expire with commits on the worktree, the
1640
+ // no-trailer tail gates them instead of buying a fresh worker to re-produce them.
1641
+ if (hasSeed) {
1642
+ noteUnmeasurable("interactive-seed launch is not in the probed process tree");
1643
+ return false;
1644
+ }
1645
+ if (cpuAccountant === undefined) {
1646
+ cpuAccountant = new WorkerTreeCpuAccountant(dispatchScript, wt);
1647
+ await cpuAccountant.start();
1648
+ }
1649
+ const observation = cpuAccountant.read();
1650
+ if (observation.gaps !== cpuGapCount) {
1651
+ cpuGapCount = observation.gaps;
1652
+ cpuFlat = undefined;
1653
+ noteUnmeasurable("one or more worker process snapshots could not be read");
1654
+ }
1655
+ const cpu = observation.cpu;
1656
+ if (cpu === undefined) {
1657
+ // Unmeasurable CPU is never evidence a worker stopped: RESET the observation rather than
1658
+ // conclude on it, and name the gap — a probe whose snapshot never parses is the same
1659
+ // structural hole as the seeded launch, and must not be the one that stays silent.
1660
+ cpuFlat = undefined;
1661
+ noteUnmeasurable("the worker process snapshot could not be read");
1662
+ return false;
1663
+ }
1664
+ const now = Date.now();
1665
+ if (cpu.ms !== cpuFlat?.ms) {
1666
+ cpuFlat = { ms: cpu.ms, since: now };
1667
+ return false;
1668
+ }
1669
+ if (now - cpuFlat.since < harvestCpuFlatWindowMs(cpu.resolutionMs))
1670
+ return false;
1671
+ const carried = await commitsAheadOf(taskBase, wt);
1672
+ if (carried.length === 0)
1673
+ return false; // nothing landed: not this branch's population
1674
+ journal.append("worker-harvest", t.id, {
1675
+ slot: slot.name, attempt, commits: carried.length,
1676
+ silentMs, cpuMs: cpu.ms, cpuResolutionMs: cpu.resolutionMs,
1677
+ });
1678
+ return true;
1679
+ };
895
1680
  // SPEND-01: this attempt's dispatch wall-clock — the usage collect cursor. Captured once here, the
896
1681
  // single site, so a test can reason about it; keep Date.now() out of profile.ts (still pure) and
897
1682
  // out of adapter module scope (the cursor is a parameter, threaded from the daemon).
@@ -931,6 +1716,10 @@ export async function runDaemon(repoRoot, opts = {}) {
931
1716
  let output;
932
1717
  let exitCode;
933
1718
  let timedOut = false;
1719
+ // T2 review: print mode's "the exit marker appeared". Kept apart from `finished` (the
1720
+ // trailer) but still needed by the keepPanes decision below, whose contract is about a
1721
+ // subprocess tree that REACHED its exit marker, not about what the worker claimed.
1722
+ let processExited = false;
934
1723
  let earlyLaunchDead = false;
935
1724
  let settleParsed;
936
1725
  let seedResult;
@@ -966,26 +1755,369 @@ export async function runDaemon(repoRoot, opts = {}) {
966
1755
  await park(t, "escalation ladder exhausted", "ladder-exhausted", assignment, attempt + 1, startMs, gateFails, consults, tokens, metered, retryMode);
967
1756
  return false;
968
1757
  };
969
- if (interactive) {
970
- // v1.2 interactive: the TUI doesn't exit on completion — the trailer is the finish line.
971
- // The exit wrapper still fires if the TUI dies (crash/quit): fast-fail instead of burning the timeout.
972
- finished = false;
973
- exitCode = null;
974
- if (adapter.interactiveSeed) {
975
- // v1.69 T6: launch the real TUI without a prompt, wait for readiness, inject one seed turn,
976
- // then fall through to the normal trailer harvest. A failed seed is recorded as a finished
977
- // failure rather than allowed to race the trailer wait.
978
- try {
979
- seedResult = await runInteractiveSeed({ driver, slot, adapter, assignment, promptFile, taskTimeoutMinutes });
1758
+ try {
1759
+ if (interactive) {
1760
+ // v1.2 interactive: the TUI doesn't exit on completion — the trailer is the finish line.
1761
+ // The exit wrapper still fires if the TUI dies (crash/quit): fast-fail instead of burning the timeout.
1762
+ finished = false;
1763
+ exitCode = null;
1764
+ if (adapter.interactiveSeed) {
1765
+ // v1.69 T6: launch the real TUI without a prompt, wait for readiness, inject one seed turn,
1766
+ // then fall through to the normal trailer harvest. A failed seed is recorded as a finished
1767
+ // failure rather than allowed to race the trailer wait.
1768
+ try {
1769
+ seedResult = await runInteractiveSeed({ driver, slot, adapter, assignment, promptFile, taskTimeoutMinutes });
1770
+ }
1771
+ catch (error) {
1772
+ if (!(error instanceof DeliveryReadinessError))
1773
+ throw error;
1774
+ if (await handleDeliveryReadiness(error))
1775
+ continue attempts;
1776
+ return;
1777
+ }
1778
+ noteLaunched();
1779
+ output = seedResult.output;
980
1780
  }
981
- catch (error) {
982
- if (!(error instanceof DeliveryReadinessError))
983
- throw error;
984
- if (await handleDeliveryReadiness(error))
985
- continue attempts;
986
- return;
1781
+ else {
1782
+ try {
1783
+ await driver.run(slot, paneDispatchCommand(dispatchScript));
1784
+ }
1785
+ catch (error) {
1786
+ if (!(error instanceof DeliveryReadinessError))
1787
+ throw error;
1788
+ if (await handleDeliveryReadiness(error))
1789
+ continue attempts;
1790
+ return;
1791
+ }
1792
+ noteLaunched();
1793
+ output = await driver.read(slot, PANE_READ_ROWS);
1794
+ }
1795
+ if (seedResult?.seedFailed) {
1796
+ finished = false;
1797
+ }
1798
+ else {
1799
+ // v1.22 T5 / OBS-19: auto-answer a fingerprint-matched trust dialog exactly once per slot.
1800
+ // Any other blocked/idle dialog pages the operator (unlatched since T1 — see below).
1801
+ let trustAnswered = false;
1802
+ // OBS-201: one liveness nudge per attempt; the grace deadline is its OWN timer, never the
1803
+ // stall window (the nudge's pane echo is absorbed before it starts, or the echo itself
1804
+ // would reset the window and make the early conclusion unreachable).
1805
+ let nudged = false;
1806
+ let nudgeFailed = false; // T1: an undeliverable nudge — the operator, not the daemon, is the actor
1807
+ let nudgeDeadline;
1808
+ // T1: page cadence, NOT a latch — a second page fires on a status change or once
1809
+ // pageRepeatMs elapses, so an operator who missed the first one is paged again.
1810
+ let lastPagedStatus;
1811
+ let lastPagedAt = 0;
1812
+ // T1 review: worker-status is journaled on CHANGE, not per slice — every journal append
1813
+ // is narrated to the run's live surface (cli/commands/run.ts) and feeds activity.ts's
1814
+ // `now:` cell, so a per-slice append wrote a status line per worker per ~30s slice and
1815
+ // pinned `now` to worker-status. On-change keeps post-hoc analysis at a fraction of the
1816
+ // volume while still recording which gate held.
1817
+ let lastStatus;
1818
+ finished = false;
1819
+ exitCode = null;
1820
+ // OBS-54: reaping keys on new pane output, not dispatch wall clock. Poll at least twice per
1821
+ // stall window (and at the existing 30s cadence for normal windows) so an active worker resets it.
1822
+ const stallWindowMs = taskTimeoutMinutes * 60_000;
1823
+ // v1.76: only monotonic work (seed submission, transcript growth, or context growth) resets
1824
+ // the stall clock. Raw pane differences are terminal chrome until proven otherwise.
1825
+ let everHadOutput = output.length > 0;
1826
+ const stallProgress = new StallProgressTracker();
1827
+ stallProgress.observe({ paneText: output, seedSubmitted: true, contextTokens });
1828
+ let lastProgressAt = Date.now();
1829
+ // T1: in-loop detector state — consecutive quota-banner slices. The dead-channel fast-kill
1830
+ // reads the tracker's RAW row-growth clock (lastRowGrowthAt), not the re-arm-suppressed
1831
+ // lastProgressAt — see the kill below.
1832
+ let quotaStreak = 0;
1833
+ let rowSaturationHeld = false; // journaled once per attempt when the kill stands down
1834
+ while (Date.now() - lastProgressAt < stallWindowMs) {
1835
+ const sliceStart = Date.now();
1836
+ const remaining = stallWindowMs - (sliceStart - lastProgressAt);
1837
+ let slice = Math.min(BLOCKED_POLL_MS, Math.max(100, Math.min(stallWindowMs / 2, remaining)));
1838
+ if (!everHadOutput) {
1839
+ const earlyLeft = earlyLaunchLivenessMs - (sliceStart - attemptStart);
1840
+ if (earlyLeft > 0)
1841
+ slice = Math.min(slice, earlyLeft);
1842
+ }
1843
+ // T2 (OBS-264): never sleep PAST the instant the harvest probe becomes eligible, and past
1844
+ // it let the probe own the cadence (HARVEST_POLL_MS) — a 30s trailer slice would
1845
+ // otherwise cost a minute per pair of CPU samples. At the shipped 5m gate this leaves
1846
+ // every slice before the gate exactly as long as it already was.
1847
+ slice = Math.min(slice, harvestSliceMs(sliceStart - lastProgressAt));
1848
+ if (await driver.waitOutput(slot, `(${trailerPattern(nonce)})|TICKMARKR_EXIT_${nonce}:\\d`, slice, { regex: true })) {
1849
+ // verify before accepting: a worker that merely DISPLAYS a marker (e.g. editing tickmarkr's
1850
+ // own source, where "TICKMARKR_EXIT:" is a string literal) must not end the wait. Only a
1851
+ // parseable trailer or a digit-suffixed exit marker in the harvest is completion.
1852
+ output = await driver.read(slot, PANE_READ_ROWS); // TUI transcripts carry chrome — read deeper than print's 500
1853
+ finished = new RegExp(trailerPattern(nonce)).test(output);
1854
+ const exit = exitRe.exec(output);
1855
+ if (finished || exit) {
1856
+ exitCode = exit ? Number(exit[1]) : null; // null ⇔ the TUI is still alive
1857
+ await sampleContext(); // final poll-seam sample before leaving the wait
1858
+ break;
1859
+ }
1860
+ }
1861
+ const paneText = await driver.read(slot, PANE_READ_ROWS);
1862
+ if (paneText.length > 0)
1863
+ everHadOutput = true;
1864
+ // OBS-117 (v1.71 T6): zero raw output by the early-launch deadline is a dead channel now.
1865
+ if (!everHadOutput && Date.now() - attemptStart >= earlyLaunchLivenessMs) {
1866
+ earlyLaunchDead = true;
1867
+ output = paneText;
1868
+ break;
1869
+ }
1870
+ // v1.23 T2: piggyback on this poll slice — same cadence as blocked/idle checks, no new timer.
1871
+ await sampleContext();
1872
+ if (stallProgress.observe({ paneText, contextTokens })) {
1873
+ lastProgressAt = Date.now();
1874
+ // T1 review fix: progress AFTER a delivered nudge means the worker answered it —
1875
+ // disarm the grace deadline. Without this the expiry below fires at the next
1876
+ // quiet patch ≥ the grace (4m) measured from the rolling lastProgressAt, so the
1877
+ // exact population the nudge rescued (workers prone to long silences, e.g. a 6m
1878
+ // test run) was force-concluded as if it had ignored the nudge. The answered
1879
+ // worker returns to the full rolling window, and the consult sees the truth.
1880
+ if (nudged && nudgeDeadline !== undefined) {
1881
+ nudgeDeadline = undefined;
1882
+ journal.append("worker-nudge-answered", t.id, { slot: slot.name, attempt });
1883
+ }
1884
+ }
1885
+ const sliceNow = Date.now();
1886
+ // T1 (OBS-263): quota banners are classified IN-LOOP — two consecutive matching slices
1887
+ // plus >=3m tracker silence — then the post-loop quota failover runs NOW, not after the
1888
+ // window. The match reads the chrome-filtered tail: the bottom of a rendered TUI frame
1889
+ // is fixed composer/welcome chrome (codex pins a "usage limit resets available" line
1890
+ // there — it matched every frame of the wedged-MCP fixture), so the known chrome is
1891
+ // filtered by identity, never by novelty — a banner already on screen at launch
1892
+ // classifies exactly like one printed mid-attempt (T1 review: a novelty baseline
1893
+ // exculpated the launch-throttle case forever).
1894
+ if (QUOTA_RE.test(stallSnapshotBannerRows(paneText)))
1895
+ quotaStreak++;
1896
+ else
1897
+ quotaStreak = 0;
1898
+ if (quotaStreak >= 2 && sliceNow - lastProgressAt >= quotaBannerSilentMs) {
1899
+ // no `output =` here: the post-loop no-trailer tail re-reads the pane anyway, so an
1900
+ // assignment would only split the classification read from the verdict read.
1901
+ journal.append("quota-banner", t.id, { slot: slot.name, attempt, silentMs: sliceNow - lastProgressAt });
1902
+ break;
1903
+ }
1904
+ // T1 (OBS-262): the `paged` latch is deleted — status is sampled EVERY slice (and
1905
+ // journaled on change — see lastStatus above), so post-hoc analysis can see which
1906
+ // gate held. page on "idle" too: herdr's blocked-scrape is strict and proved flaky
1907
+ // for TUI dialogs (live check: cursor's trust dialog scraped as idle).
1908
+ // "unknown"/"working" never page.
1909
+ const st = await driver.status(slot);
1910
+ if (st !== lastStatus) {
1911
+ lastStatus = st;
1912
+ journal.append("worker-status", t.id, { slot: slot.name, status: st, attempt });
1913
+ }
1914
+ if (st === "blocked" || st === "idle") {
1915
+ // T5: once-per-slot auto-answer when the adapter declares a trust dialog and the pane
1916
+ // text matches. tickmarkr created the worktree from the operator's own repo — safe by construction.
1917
+ if (!trustAnswered && adapter.trustDialog && driver.sendKey) {
1918
+ try {
1919
+ const paneText = await driver.read(slot, 80);
1920
+ if (matchesTrustDialog(paneText, adapter.trustDialog)) {
1921
+ trustAnswered = true;
1922
+ // v1.25 T1: audit trail for live runs — prove the dialog appeared and was answered.
1923
+ // Latch + sendKey + no-page continue stay byte-identical; this append is additive only.
1924
+ journal.append("trust-auto-answer", t.id, { slot: slot.name, adapter: adapter.id });
1925
+ await driver.sendKey(slot, adapter.trustDialog.key);
1926
+ const spent = Date.now() - sliceStart;
1927
+ if (spent < slice)
1928
+ await new Promise((r) => setTimeout(r, Math.min(slice - spent, 1_000)));
1929
+ continue; // do not page — keep waiting for the trailer
1930
+ }
1931
+ }
1932
+ catch {
1933
+ /* read/send failed — fall through to page the operator */
1934
+ }
1935
+ }
1936
+ }
1937
+ // T1 (OBS-262): the daemon ACTS on a silent worker before paging anyone. Gate: monotonic
1938
+ // tracker silent ≥ the nudge threshold — the herdr status reading (idle/unknown/working)
1939
+ // no longer holds the gate hostage; only `blocked` stays page-only (nudging a dialog
1940
+ // prompt can't help). One nudge per attempt, then the grace timer owns the conclusion.
1941
+ const nudgeable = st !== "blocked" && !!driver.nudge && NUDGEABLE_ADAPTERS.has(adapter.id);
1942
+ // T1 review (answer-then-die): `nudgeable` is a per-slice property of the adapter and
1943
+ // status — it is NOT "a daemon action is pending". An action is pending only while the
1944
+ // nudge can still fire (un-nudged) or its grace window is armed; once the worker
1945
+ // ANSWERS, the disarm above clears nudgeDeadline while `nudged` stays latched, and the
1946
+ // daemon has nothing left to do — the pane falls back under the fast-kill and page
1947
+ // watchdogs like any other, instead of riding the whole rolling window untended.
1948
+ const nudgePending = nudgeable && (!nudged || nudgeDeadline !== undefined);
1949
+ // T2 (OBS-264): the liveness triad CONCLUDES the wait on finished work — commits ahead
1950
+ // of the task base (this attempt's own AND any carried forward — see the eligibility
1951
+ // comment at the dispatch site), a flat worker-tree CPU delta, and >= harvestSilentMs
1952
+ // of monotonic tracker silence (defined once, above, and run identically by the print loop).
1953
+ // T2 review (material): it carries the SAME nudge hold as the fast-kill below, by the
1954
+ // same clause and for the same reason. The CPU leg cannot tell "idle because finished"
1955
+ // from "idle because holding an unsubmitted turn in its input box" — both read flat CPU
1956
+ // under a silent tracker — and the nudge is the one signal that can. Under the shipped
1957
+ // constants (harvest 5m < nudge 10m) an unheld harvest concluded every committed
1958
+ // claude-code worker before the rescue could fire, leaving T1's nudge dead code for
1959
+ // exactly the committed-and-stalled population OBS-264 is about. Holding concludes at
1960
+ // ~14m (nudge + grace) rather than ~36m — nearly all of the OBS-264 win, and a worker
1961
+ // that only needed a submit answers with a full trailer instead of partial work. An
1962
+ // ANSWERED or twice-undeliverable nudge leaves nothing pending, so the triad governs
1963
+ // again; the hold is on a pending daemon ACTION, never on the adapter being nudgeable.
1964
+ if ((!nudgePending || nudgeFailed) && await harvestConcludes(sliceNow - lastProgressAt))
1965
+ break;
1966
+ // T1 (R1 dead-channel fast-kill): no trailer, no worktree delta, and no output growth
1967
+ // for the fast-kill window — the channel is dead, so conclude NOW
1968
+ // (journaled) and let the existing no-trailer tail classify and route the attempt.
1969
+ // The tracker is the growth signal on purpose: raw pane bytes grow on cosmetic repaint
1970
+ // (an elapsed "9s"→"10s" lengthens the read and would hide a frozen pane). But the
1971
+ // tracker SHARES the read window's ceiling: its row signal saturates once a sample
1972
+ // FILLS a PANE_READ_ROWS read on raw lines (blanks/chrome included — see stall.ts),
1973
+ // and past that point a flat tracker means "unmeasurable", not "dead". For unmetered
1974
+ // adapters (codex, cursor-agent, grok, opencode — no contextUsage, so the token signal
1975
+ // never fires) rows are the ONLY
1976
+ // signal, and those are exactly the non-nudgeable adapters this kill governs — a live
1977
+ // worker past the ceiling would be concluded dead mid-work. So the kill STANDS DOWN on
1978
+ // a saturated row signal (journaled once per attempt); the rolling window still owns
1979
+ // that pane, exactly as pre-T1. Token growth counts as life either way, so a metered
1980
+ // worker thinking through a long tool run survives.
1981
+ // The triad has NO status exemption: a pane that herdr reports as blocked, idle, working
1982
+ // or unknown dies alike once it holds no trailer, no delta and no growth — waiting the
1983
+ // rolling window out on a status reading is exactly the blindness T1 removes. A matched
1984
+ // trust dialog is auto-answered above and `continue`s before ever reaching here.
1985
+ // The NUDGE gets first crack at a nudgeable pane: the fast-kill holds while the daemon
1986
+ // still has an action of its own pending (un-nudged, or inside the grace window) —
1987
+ // under the shipped constants (kill 5m < nudge 10m) a delta-less pane would otherwise
1988
+ // die before the rescue could ever fire. An ANSWERED nudge leaves nothing pending, so
1989
+ // the hold lifts and the triad governs again. Once a nudge has failed to deliver TWICE
1990
+ // (one in-slice retry filters a driver flake — see below), the delta clause drops too:
1991
+ // a delta is past work, and a pane no signal can reach must not buy the rest of the
1992
+ // window with it (a `working` pane can't be paged either, so this is the only bound
1993
+ // that path has).
1994
+ if (stallProgress.rowSignalSaturated && !rowSaturationHeld) {
1995
+ rowSaturationHeld = true;
1996
+ journal.append("worker-dead-held", t.id, { slot: slot.name, attempt, reason: "row-signal-saturated" });
1997
+ }
1998
+ // T1 review fix: the kill's "no output growth" leg clocks off the RAW growth signals,
1999
+ // never lastProgressAt alone — the flat-token rule (stall.ts) deliberately suppresses
2000
+ // the re-arm report on row growth once tokens stick, and contextTokens is sticky across
2001
+ // read misses, so a metered non-nudgeable adapter (pi) streaming rows under a stale
2002
+ // counter presented a frozen lastProgressAt and was killed mid-work. lastRowGrowthAt is
2003
+ // recorded on every high-water advance, suppressed or not; token growth already rides
2004
+ // lastProgressAt. Either one advancing is output growth.
2005
+ const lastOutputGrowthAt = Math.max(stallProgress.lastRowGrowthAt ?? 0, lastProgressAt);
2006
+ if (!stallProgress.rowSignalSaturated
2007
+ && (!nudgePending || nudgeFailed)
2008
+ && sliceNow - lastOutputGrowthAt >= deadChannelFastKillMs
2009
+ && (nudgeFailed || !(await worktreeHasDelta(wt, taskBase)))) {
2010
+ journal.append("worker-dead", t.id, { slot: slot.name, attempt, silentMs: sliceNow - lastOutputGrowthAt });
2011
+ break;
2012
+ }
2013
+ if (nudgeable && !nudged && sliceNow - lastProgressAt >= nudgeAfterSilentMs) {
2014
+ nudged = true;
2015
+ // T1 review: a false return is a driver-delivery outcome (missing pin, readiness
2016
+ // stable-frame timeout, read-back hiccup), not proof of an unreachable channel — so
2017
+ // one failure is a flake class, retried once in-slice after a short settle. Only a
2018
+ // failed RETRY condemns the channel. Both failures happen inside this slice, so the
2019
+ // latch stays immediate and exactly one failure is journaled per attempt.
2020
+ let delivered = await driver.nudge(slot, WORKER_NUDGE_MESSAGE);
2021
+ if (!delivered) {
2022
+ await new Promise((r) => setTimeout(r, NUDGE_REDELIVER_MS));
2023
+ delivered = await driver.nudge(slot, WORKER_NUDGE_MESSAGE);
2024
+ }
2025
+ if (delivered) {
2026
+ // absorb the nudge's own echo BEFORE arming the grace timer — post-nudge progress
2027
+ // is measured against this baseline, not against the echo.
2028
+ const echo = await driver.read(slot, PANE_READ_ROWS);
2029
+ stallProgress.observe({ paneText: echo, contextTokens });
2030
+ nudgeDeadline = Date.now() + workerNudgeGraceMs;
2031
+ journal.append("worker-nudge", t.id, { slot: slot.name, attempt });
2032
+ }
2033
+ else {
2034
+ // T1: an undeliverable nudge is itself a condemnation — the channel is unreachable,
2035
+ // so the fast-kill above stops requiring a clean worktree for THIS pane (see there).
2036
+ // A blocked/idle pane is still paged below; a `working`/`unknown` one cannot be, and
2037
+ // for it the conclusion's own escalation is what notifies. Nothing here waits the
2038
+ // rolling window out.
2039
+ nudgeFailed = true;
2040
+ journal.append("worker-nudge-failed", t.id, { slot: slot.name, attempt });
2041
+ }
2042
+ }
2043
+ if (nudged && nudgeDeadline !== undefined && sliceNow >= nudgeDeadline && sliceNow - lastProgressAt >= workerNudgeGraceMs) {
2044
+ // grace spent, still no post-nudge progress: re-harvest once (the trailer may have
2045
+ // landed between polls), then conclude the wait as a stall NOW — the consult sees the
2046
+ // un-answered nudge instead of the remainder of the window.
2047
+ nudgeDeadline = undefined;
2048
+ output = await driver.read(slot, PANE_READ_ROWS);
2049
+ finished = new RegExp(trailerPattern(nonce)).test(output);
2050
+ const exit = exitRe.exec(output);
2051
+ if (finished || exit) {
2052
+ exitCode = exit ? Number(exit[1]) : null;
2053
+ await sampleContext();
2054
+ break;
2055
+ }
2056
+ journal.append("worker-nudge-expired", t.id, { slot: slot.name, attempt, graceMs: workerNudgeGraceMs });
2057
+ lastProgressAt = Date.now() - stallWindowMs; // the existing harvest/classify tail runs unmodified
2058
+ continue; // conclude via the loop condition — never page over an acted-on nudge
2059
+ }
2060
+ // Unlatched page (T1): the page DECISION fires and is journaled every slice the
2061
+ // operator is the right actor — i.e. the nudge path doesn't own it (non-allowlisted
2062
+ // adapter, no nudge surface, or a blocked dialog) or the nudge was attempted and
2063
+ // FAILED. A nudgeable worker below the silence threshold waits for its nudge; a
2064
+ // DELIVERED nudge's pending grace suppresses the page — the daemon already acted. An
2065
+ // ANSWERED nudge has no action pending (the disarm cleared the deadline), so a pane
2066
+ // that then reads blocked/idle is the operator's again.
2067
+ // Delivery is unlatched too: the operator is notified again on a status change or
2068
+ // once pageRepeatMs elapses, so a missed first page is not the last one.
2069
+ const pageable = (st === "blocked" || st === "idle")
2070
+ && (!nudgePending || nudgeFailed);
2071
+ if (pageable) {
2072
+ journal.append("operator-page", t.id, { slot: slot.name, attempt, status: st });
2073
+ if (st !== lastPagedStatus || sliceNow - lastPagedAt >= pageRepeatMs) {
2074
+ lastPagedStatus = st;
2075
+ lastPagedAt = sliceNow;
2076
+ const why = st === "blocked" ? "is blocked on a prompt — approve in its pane" : "looks idle without finishing — check its pane";
2077
+ await driver.notify(`tickmarkr ${runId}: ${slot.name} ${why}`, { tier: "attention" });
2078
+ }
2079
+ }
2080
+ // a dead pane or a false-positive marker display returns fast — sleep the unspent slice, never hot-spin
2081
+ const spent = Date.now() - sliceStart;
2082
+ if (spent < slice)
2083
+ await new Promise((r) => setTimeout(r, Math.min(slice - spent, 1_000)));
2084
+ }
2085
+ if (!finished && exitCode === null) {
2086
+ // timed out (or only ever saw false positives): harvest whatever the pane holds now
2087
+ timedOut = Date.now() - lastProgressAt >= stallWindowMs;
2088
+ output = await driver.read(slot, PANE_READ_ROWS);
2089
+ finished = new RegExp(trailerPattern(nonce)).test(output);
2090
+ const exit = exitRe.exec(output);
2091
+ exitCode = exit ? Number(exit[1]) : null;
2092
+ }
2093
+ if (finished) {
2094
+ await driver.waitAgentStatus(slot, "idle", 5_000); // settle, then re-harvest the final render
2095
+ output = await driver.read(slot, PANE_READ_ROWS);
2096
+ }
2097
+ // T5 / OBS-111: an interactive harvest can race the TUI's final paint. When the pane
2098
+ // contains the nonce token but the JSON hasn't balanced yet, settle and re-read through
2099
+ // the existing pane-read seam once or twice before recording a malformed-trailer cause.
2100
+ if (interactive) {
2101
+ const stallWindowMs = taskTimeoutMinutes * 60_000;
2102
+ const settleDeadline = attemptStart + stallWindowMs;
2103
+ const settleDelayMs = 1_000;
2104
+ const maxSettleRetries = 2;
2105
+ let settleTries = 0;
2106
+ settleParsed = adapter.parse(output, nonce);
2107
+ while (settleParsed.summary === UNPARSEABLE_TRAILER_SUMMARY && settleTries < maxSettleRetries) {
2108
+ const remaining = settleDeadline - Date.now();
2109
+ if (remaining <= 0)
2110
+ break;
2111
+ await new Promise((r) => setTimeout(r, Math.min(settleDelayMs, remaining)));
2112
+ output = await driver.read(slot, PANE_READ_ROWS);
2113
+ settleParsed = adapter.parse(output, nonce);
2114
+ settleTries++;
2115
+ }
2116
+ if (settleParsed.summary !== UNPARSEABLE_TRAILER_SUMMARY) {
2117
+ finished = settleParsed.summary !== NO_TRAILER_SUMMARY;
2118
+ }
2119
+ }
987
2120
  }
988
- output = seedResult.output;
989
2121
  }
990
2122
  else {
991
2123
  try {
@@ -998,229 +2130,68 @@ export async function runDaemon(repoRoot, opts = {}) {
998
2130
  continue attempts;
999
2131
  return;
1000
2132
  }
1001
- output = await driver.read(slot, 1000);
1002
- }
1003
- if (seedResult?.seedFailed) {
1004
- finished = false;
1005
- }
1006
- else {
1007
- let paged = false;
1008
- // v1.22 T5 / OBS-19: auto-answer a fingerprint-matched trust dialog exactly once per slot.
1009
- // Any other blocked/idle dialog still pages the operator (paged latch below).
1010
- let trustAnswered = false;
1011
- // OBS-201: one liveness nudge per attempt; the grace deadline is its OWN timer, never the
1012
- // stall window (the nudge's pane echo is absorbed before it starts, or the echo itself
1013
- // would reset the window and make the early conclusion unreachable).
1014
- let nudged = false;
1015
- let nudgeDeadline;
1016
- finished = false;
1017
- exitCode = null;
1018
- // OBS-54: reaping keys on new pane output, not dispatch wall clock. Poll at least twice per
1019
- // stall window (and at the existing 30s cadence for normal windows) so an active worker resets it.
2133
+ noteLaunched();
2134
+ // OBS-54: headless workers have the same output-inactivity budget as visible panes.
2135
+ // v1.76: same monotonic-progress measure as the interactive site; harvest stays raw.
1020
2136
  const stallWindowMs = taskTimeoutMinutes * 60_000;
1021
- // v1.76: only monotonic work (seed submission, transcript growth, or context growth) resets
1022
- // the stall clock. Raw pane differences are terminal chrome until proven otherwise.
1023
- let everHadOutput = output.length > 0;
2137
+ const initialPane = await driver.read(slot, 500);
2138
+ let everHadOutput = initialPane.length > 0;
1024
2139
  const stallProgress = new StallProgressTracker();
1025
- stallProgress.observe({ paneText: output, seedSubmitted: true, contextTokens });
2140
+ stallProgress.observe({ paneText: initialPane, seedSubmitted: true, contextTokens });
1026
2141
  let lastProgressAt = Date.now();
2142
+ finished = false;
2143
+ // T2 review (material): the exit marker proves the PROCESS EXITED, never that the worker
2144
+ // emitted a trailer — the two were the same flag here, so a headless worker that committed
2145
+ // and exited cleanly without one entered the tail as finished:true, skipping the harvest
2146
+ // synthesis entirely and reaching gates with the worker's own ok:false and no
2147
+ // worker-result-harvested row. The interactive site has always kept them apart (`finished`
2148
+ // there is the trailer regex; the exit marker only sets exitCode), and the cause taxonomy
2149
+ // already names this shape "clean-exit-no-trailer" — unreachable in print mode until now.
1027
2150
  while (Date.now() - lastProgressAt < stallWindowMs) {
1028
- const sliceStart = Date.now();
1029
- const remaining = stallWindowMs - (sliceStart - lastProgressAt);
2151
+ const remaining = stallWindowMs - (Date.now() - lastProgressAt);
1030
2152
  let slice = Math.min(BLOCKED_POLL_MS, Math.max(100, Math.min(stallWindowMs / 2, remaining)));
1031
2153
  if (!everHadOutput) {
1032
- const earlyLeft = earlyLaunchLivenessMs - (sliceStart - attemptStart);
2154
+ const earlyLeft = earlyLaunchLivenessMs - (Date.now() - attemptStart);
1033
2155
  if (earlyLeft > 0)
1034
2156
  slice = Math.min(slice, earlyLeft);
1035
2157
  }
1036
- if (await driver.waitOutput(slot, `(${trailerPattern(nonce)})|TICKMARKR_EXIT_${nonce}:\\d`, slice, { regex: true })) {
1037
- // verify before accepting: a worker that merely DISPLAYS a marker (e.g. editing tickmarkr's
1038
- // own source, where "TICKMARKR_EXIT:" is a string literal) must not end the wait. Only a
1039
- // parseable trailer or a digit-suffixed exit marker in the harvest is completion.
1040
- output = await driver.read(slot, 1000); // TUI transcripts carry chrome — read deeper than print's 500
1041
- finished = new RegExp(trailerPattern(nonce)).test(output);
1042
- const exit = exitRe.exec(output);
1043
- if (finished || exit) {
1044
- exitCode = exit ? Number(exit[1]) : null; // null ⇔ the TUI is still alive
1045
- await sampleContext(); // final poll-seam sample before leaving the wait
1046
- break;
1047
- }
2158
+ // T2 (OBS-264): same probe cadence the interactive loop uses — see harvestSliceMs.
2159
+ slice = Math.min(slice, harvestSliceMs(Date.now() - lastProgressAt));
2160
+ if (await driver.waitOutput(slot, `TICKMARKR_EXIT_${nonce}:\\d`, slice, { regex: true })) {
2161
+ processExited = true;
2162
+ break;
1048
2163
  }
1049
- const paneText = await driver.read(slot, 1000);
2164
+ const paneText = await driver.read(slot, 500);
1050
2165
  if (paneText.length > 0)
1051
2166
  everHadOutput = true;
1052
- // OBS-117 (v1.71 T6): zero raw output by the early-launch deadline is a dead channel now.
1053
2167
  if (!everHadOutput && Date.now() - attemptStart >= earlyLaunchLivenessMs) {
1054
2168
  earlyLaunchDead = true;
1055
- output = paneText;
1056
2169
  break;
1057
2170
  }
1058
- // v1.23 T2: piggyback on this poll slice — same cadence as blocked/idle checks, no new timer.
1059
2171
  await sampleContext();
1060
2172
  if (stallProgress.observe({ paneText, contextTokens }))
1061
2173
  lastProgressAt = Date.now();
1062
- // page on "idle" too: herdr's blocked-scrape is strict and proved flaky for TUI dialogs
1063
- // (live check: cursor's trust dialog scraped as idle). "unknown" never pages — that's just
1064
- // a pane the scraper can't read (subprocess, dead pane); the task timeout covers those.
1065
- const st = paged ? "" : await driver.status(slot);
1066
- if (!paged && (st === "blocked" || st === "idle")) {
1067
- // T5: once-per-slot auto-answer when the adapter declares a trust dialog and the pane
1068
- // text matches. tickmarkr created the worktree from the operator's own repo — safe by construction.
1069
- if (!trustAnswered && adapter.trustDialog && driver.sendKey) {
1070
- try {
1071
- const paneText = await driver.read(slot, 80);
1072
- if (matchesTrustDialog(paneText, adapter.trustDialog)) {
1073
- trustAnswered = true;
1074
- // v1.25 T1: audit trail for live runs — prove the dialog appeared and was answered.
1075
- // Latch + sendKey + no-page continue stay byte-identical; this append is additive only.
1076
- journal.append("trust-auto-answer", t.id, { slot: slot.name, adapter: adapter.id });
1077
- await driver.sendKey(slot, adapter.trustDialog.key);
1078
- const spent = Date.now() - sliceStart;
1079
- if (spent < slice)
1080
- await new Promise((r) => setTimeout(r, Math.min(slice - spent, 1_000)));
1081
- continue; // do not page — keep waiting for the trailer
1082
- }
1083
- }
1084
- catch {
1085
- /* read/send failed — fall through to page the operator */
1086
- }
1087
- }
1088
- // OBS-201: the daemon ACTS on an idle pane before paging anyone. Gate: idle AND the
1089
- // monotonic tracker silent ≥ the nudge threshold (a worker inside a tool run reads as
1090
- // `working` and never gets here; a briefly-settling TUI hasn't been silent long enough).
1091
- const nudgeable = st === "idle" && !!driver.nudge && NUDGEABLE_ADAPTERS.has(adapter.id);
1092
- if (nudgeable && !nudged) {
1093
- if (Date.now() - lastProgressAt >= nudgeAfterSilentMs) {
1094
- nudged = true;
1095
- if (await driver.nudge(slot, WORKER_NUDGE_MESSAGE)) {
1096
- // absorb the nudge's own echo BEFORE arming the grace timer — post-nudge progress
1097
- // is measured against this baseline, not against the echo.
1098
- const echo = await driver.read(slot, 1000);
1099
- stallProgress.observe({ paneText: echo, contextTokens });
1100
- nudgeDeadline = Date.now() + workerNudgeGraceMs;
1101
- journal.append("worker-nudge", t.id, { slot: slot.name, attempt });
1102
- }
1103
- else {
1104
- journal.append("worker-nudge-failed", t.id, { slot: slot.name, attempt });
1105
- // nudge undeliverable — fall back to today's behavior: page once, window backstop
1106
- paged = true;
1107
- await driver.notify(`tickmarkr ${runId}: ${slot.name} looks idle without finishing — check its pane`, { tier: "attention" });
1108
- }
1109
- }
1110
- // idle but not yet silent past the threshold: neither nudge nor page this slice
1111
- }
1112
- else if (nudgeable && nudged && nudgeDeadline !== undefined) {
1113
- if (Date.now() >= nudgeDeadline && Date.now() - lastProgressAt >= workerNudgeGraceMs) {
1114
- // grace spent, still idle, no post-nudge progress: re-harvest once (the trailer may
1115
- // have landed between polls), then conclude the wait as a stall NOW — the consult
1116
- // sees the un-answered nudge instead of the remainder of the window.
1117
- nudgeDeadline = undefined;
1118
- output = await driver.read(slot, 1000);
1119
- finished = new RegExp(trailerPattern(nonce)).test(output);
1120
- const exit = exitRe.exec(output);
1121
- if (finished || exit) {
1122
- exitCode = exit ? Number(exit[1]) : null;
1123
- await sampleContext();
1124
- break;
1125
- }
1126
- journal.append("worker-nudge-expired", t.id, { slot: slot.name, attempt, graceMs: workerNudgeGraceMs });
1127
- lastProgressAt = Date.now() - stallWindowMs; // the existing harvest/classify tail runs unmodified
1128
- }
1129
- }
1130
- else {
1131
- paged = true; // page once — the visible pane is the operator's to unblock; task timeout is the backstop
1132
- const why = st === "blocked" ? "is blocked on a prompt — approve in its pane" : "looks idle without finishing — check its pane";
1133
- await driver.notify(`tickmarkr ${runId}: ${slot.name} ${why}`, { tier: "attention" });
1134
- }
1135
- }
1136
- // a dead pane or a false-positive marker display returns fast — sleep the unspent slice, never hot-spin
1137
- const spent = Date.now() - sliceStart;
1138
- if (spent < slice)
1139
- await new Promise((r) => setTimeout(r, Math.min(slice - spent, 1_000)));
1140
- }
1141
- if (!finished && exitCode === null) {
1142
- // timed out (or only ever saw false positives): harvest whatever the pane holds now
1143
- timedOut = Date.now() - lastProgressAt >= stallWindowMs;
1144
- output = await driver.read(slot, 1000);
1145
- finished = new RegExp(trailerPattern(nonce)).test(output);
1146
- const exit = exitRe.exec(output);
1147
- exitCode = exit ? Number(exit[1]) : null;
1148
- }
1149
- if (finished) {
1150
- await driver.waitAgentStatus(slot, "idle", 5_000); // settle, then re-harvest the final render
1151
- output = await driver.read(slot, 1000);
1152
- }
1153
- // T5 / OBS-111: an interactive harvest can race the TUI's final paint. When the pane
1154
- // contains the nonce token but the JSON hasn't balanced yet, settle and re-read through
1155
- // the existing pane-read seam once or twice before recording a malformed-trailer cause.
1156
- if (interactive) {
1157
- const stallWindowMs = taskTimeoutMinutes * 60_000;
1158
- const settleDeadline = attemptStart + stallWindowMs;
1159
- const settleDelayMs = 1_000;
1160
- const maxSettleRetries = 2;
1161
- let settleTries = 0;
1162
- settleParsed = adapter.parse(output, nonce);
1163
- while (settleParsed.summary === UNPARSEABLE_TRAILER_SUMMARY && settleTries < maxSettleRetries) {
1164
- const remaining = settleDeadline - Date.now();
1165
- if (remaining <= 0)
1166
- break;
1167
- await new Promise((r) => setTimeout(r, Math.min(settleDelayMs, remaining)));
1168
- output = await driver.read(slot, 1000);
1169
- settleParsed = adapter.parse(output, nonce);
1170
- settleTries++;
1171
- }
1172
- if (settleParsed.summary !== UNPARSEABLE_TRAILER_SUMMARY) {
1173
- finished = settleParsed.summary !== NO_TRAILER_SUMMARY;
1174
- }
2174
+ // T2 (OBS-264): the same liveness triad the interactive loop runs. A headless worker that
2175
+ // committed and went quiet is finished work too, and before this it rode the entire
2176
+ // window out before anything looked at its commits.
2177
+ // No nudge hold here, deliberately: print mode has no nudge surface at all (no pane to
2178
+ // steer, driver.nudge is never consulted on this path), so there is no pending daemon
2179
+ // action for the triad to preempt — the asymmetry with the interactive call site above is
2180
+ // the absence of the thing being held for, not an oversight.
2181
+ if (await harvestConcludes(Date.now() - lastProgressAt))
2182
+ break;
1175
2183
  }
2184
+ output = await driver.read(slot, 500);
2185
+ exitCode = Number(exitRe.exec(output)?.[1] ?? 1);
2186
+ // Completion is the trailer, exactly as in the interactive loop. A process that exited
2187
+ // without one is finished:false with a non-null exitCode — the harvest synthesis then owns
2188
+ // it when the worktree carries work, and classifyWorkerResultCause names it otherwise.
2189
+ finished = new RegExp(trailerPattern(nonce)).test(output);
2190
+ timedOut = !processExited && !finished && Date.now() - lastProgressAt >= stallWindowMs;
1176
2191
  }
1177
2192
  }
1178
- else {
1179
- try {
1180
- await driver.run(slot, paneDispatchCommand(dispatchScript));
1181
- }
1182
- catch (error) {
1183
- if (!(error instanceof DeliveryReadinessError))
1184
- throw error;
1185
- if (await handleDeliveryReadiness(error))
1186
- continue attempts;
1187
- return;
1188
- }
1189
- // OBS-54: headless workers have the same output-inactivity budget as visible panes.
1190
- // v1.76: same monotonic-progress measure as the interactive site; harvest stays raw.
1191
- const stallWindowMs = taskTimeoutMinutes * 60_000;
1192
- const initialPane = await driver.read(slot, 500);
1193
- let everHadOutput = initialPane.length > 0;
1194
- const stallProgress = new StallProgressTracker();
1195
- stallProgress.observe({ paneText: initialPane, seedSubmitted: true, contextTokens });
1196
- let lastProgressAt = Date.now();
1197
- finished = false;
1198
- while (Date.now() - lastProgressAt < stallWindowMs) {
1199
- const remaining = stallWindowMs - (Date.now() - lastProgressAt);
1200
- let slice = Math.min(BLOCKED_POLL_MS, Math.max(100, Math.min(stallWindowMs / 2, remaining)));
1201
- if (!everHadOutput) {
1202
- const earlyLeft = earlyLaunchLivenessMs - (Date.now() - attemptStart);
1203
- if (earlyLeft > 0)
1204
- slice = Math.min(slice, earlyLeft);
1205
- }
1206
- if (await driver.waitOutput(slot, `TICKMARKR_EXIT_${nonce}:\\d`, slice, { regex: true })) {
1207
- finished = true;
1208
- break;
1209
- }
1210
- const paneText = await driver.read(slot, 500);
1211
- if (paneText.length > 0)
1212
- everHadOutput = true;
1213
- if (!everHadOutput && Date.now() - attemptStart >= earlyLaunchLivenessMs) {
1214
- earlyLaunchDead = true;
1215
- break;
1216
- }
1217
- await sampleContext();
1218
- if (stallProgress.observe({ paneText, contextTokens }))
1219
- lastProgressAt = Date.now();
1220
- }
1221
- output = await driver.read(slot, 500);
1222
- exitCode = Number(exitRe.exec(output)?.[1] ?? 1);
1223
- timedOut = !finished && Date.now() - lastProgressAt >= stallWindowMs;
2193
+ finally {
2194
+ await cpuAccountant?.stop();
1224
2195
  }
1225
2196
  // SPEND-01 interactive metering race: the harvest loop breaks on the trailer, but the worker
1226
2197
  // shell may still be running post-trailer bookkeeping (session-store flush, fake usage stamp,
@@ -1232,7 +2203,7 @@ export async function runDaemon(repoRoot, opts = {}) {
1232
2203
  }
1233
2204
  // keepPanes retains visible context, not a timed-out subprocess tree. Close before consult/retry
1234
2205
  // can recreate the worktree; Herdr and subprocesses that reached their exit marker stay unchanged.
1235
- if (keepOpen && (finished || driver.id !== "subprocess"))
2206
+ if (keepOpen && (finished || processExited || driver.id !== "subprocess"))
1236
2207
  keptSlots.push(slot);
1237
2208
  else
1238
2209
  await closeSlot(slot);
@@ -1247,15 +2218,51 @@ export async function runDaemon(repoRoot, opts = {}) {
1247
2218
  tokens = addUsage(tokens, attemptUsage);
1248
2219
  metered++;
1249
2220
  }
1250
- const result = settleParsed ?? adapter.parse(output, nonce);
1251
- const cause = classifyWorkerResultCause({ output, ok: result.ok, finished, exitCode, summary: result.summary, timedOut });
2221
+ let result = settleParsed ?? adapter.parse(output, nonce);
2222
+ const workerFinished = finished;
2223
+ const workerCause = classifyWorkerResultCause({ output, ok: result.ok, finished, exitCode, summary: result.summary, timedOut });
1252
2224
  journal.append("worker-result", t.id, {
1253
- ok: result.ok, summary: result.summary, deviations: result.deviations, finished, exitCode,
1254
- mode: interactive ? "interactive" : "print", ...(cause ? { cause } : {}),
2225
+ ok: result.ok, summary: result.summary, deviations: result.deviations, finished: workerFinished, exitCode,
2226
+ mode: interactive ? "interactive" : "print", ...(workerCause ? { cause: workerCause } : {}),
1255
2227
  });
1256
- if (result.ok && finished)
2228
+ // T2 review (routing precedence): provider-death, quota and dead-channel classification
2229
+ // derive from ONE rule — the worker's OWN outcome, workerFinished and this PRE-HARVEST
2230
+ // parse — never from the synthesized result below. A worker that committed and then walled
2231
+ // (provider outage, quota banner, dead CLI) must still route; the harvest synthesis only
2232
+ // decides whether THIS attempt's worktree goes to gates, and when routing wins instead, the
2233
+ // commits survive via the existing commitsToCarry/cherryPickCommits carry-forward.
2234
+ const preHarvestResult = result;
2235
+ // T2 (OBS-264): recognize committed no-trailer work BEFORE any no-trailer streak, provider,
2236
+ // quota or dead-channel routing. Gates never trusted the trailer, so this successful synthesis
2237
+ // must enter exactly where a worker-claimed ok enters. Preserve the parsed worker truth in the
2238
+ // worker-result row above and name the synthesized gate input in its own harvested event.
2239
+ let harvestedCommits = [];
2240
+ if (!workerFinished) {
2241
+ harvestedCommits = await commitsAheadOf(taskBase, wt);
2242
+ if (harvestedCommits.length > 0) {
2243
+ result = { ok: true, summary: HARVESTED_RESULT_SUMMARY, deviations: [], raw: output };
2244
+ finished = true;
2245
+ journal.append("worker-result-harvested", t.id, {
2246
+ attempt, commits: harvestedCommits, summary: HARVESTED_RESULT_SUMMARY, source: "harvest",
2247
+ });
2248
+ }
2249
+ }
2250
+ // T2 review: provider-death is the THIRD routing branch that must read the pre-harvest
2251
+ // outcome — the synthesis must not null it. A worker that committed and then printed the
2252
+ // outage banner without exiting (workerFinished false, so the harvest fires) still takes
2253
+ // the capped same-channel requeue below; nulling the cause here would skip that branch and
2254
+ // let classifyDeadChannel(preHarvestResult) demote the channel run-wide on a transient blip.
2255
+ const cause = harvestedCommits.length > 0 && workerCause !== "provider-death" ? undefined : workerCause;
2256
+ // T2 review (family): the no-trailer streak is accounted on the SAME ONE rule the routing
2257
+ // branches above use — workerFinished and the PRE-HARVEST parse — never the synthesized
2258
+ // result. A channel that commits but never emits a parseable trailer still burned a
2259
+ // no-trailer window (OBS-57): the synthesis decides whether THIS worktree goes to gates, it
2260
+ // never certifies the channel. Reading the synthesized `finished`/`ok` here reset the streak
2261
+ // on every harvest, so a CLI that produces commits and swallows every trailer was immune to
2262
+ // the two-window demotion and stayed first pick for the rest of the run.
2263
+ if (preHarvestResult.ok && workerFinished)
1257
2264
  noTrailerStreak.set(channelKey(assignment), 0);
1258
- else if (!finished && cause !== "provider-death") {
2265
+ else if (!workerFinished && cause !== "provider-death") {
1259
2266
  const ck = channelKey(assignment);
1260
2267
  const streak = (noTrailerStreak.get(ck) ?? 0) + 1;
1261
2268
  noTrailerStreak.set(ck, streak);
@@ -1275,8 +2282,19 @@ export async function runDaemon(repoRoot, opts = {}) {
1275
2282
  }
1276
2283
  // quota exhaustion → failover within floor; does NOT consume the ladder (spec §4)
1277
2284
  // print: guarded on exit code — exit-0 output that merely MENTIONS "rate limit" must not failover
1278
- // interactive: a harvested trailer beats quota mentions; without one, quota text fails over (spec v1.2 §2)
1279
- const quotaHit = (interactive ? !finished : exitCode !== 0) && QUOTA_RE.test(output);
2285
+ // interactive: a worker-CLAIMED trailer beats quota mentions; without one, quota text fails over
2286
+ // (spec v1.2 §2) — matched on the chrome-filtered tail, the exact discrimination the in-loop
2287
+ // classifier makes, so the two can never disagree. A TUI harvest is the whole retained pane:
2288
+ // an unscoped match failed a worker over for quoting "rate limit" in its own diff (tail
2289
+ // scoping kills that — the mention sits ABOVE the tail), and a raw-tail match fires on fixed
2290
+ // chrome (codex's welcome line — filtered by identity, so a launch-time banner this backstop
2291
+ // exists to catch is never exculpated). Print output keeps the exit-code guard.
2292
+ // T2 review: the gate is `workerFinished`, not the harvest-synthesized `finished` — a
2293
+ // committed-but-quota-walled attempt routes here FIRST (its commits ride the carry-forward
2294
+ // into the next attempt's recreated worktree), it never buys a gate run on throttled work.
2295
+ const quotaHit = interactive
2296
+ ? !workerFinished && QUOTA_RE.test(stallSnapshotBannerRows(output))
2297
+ : exitCode !== 0 && QUOTA_RE.test(output);
1280
2298
  if (quotaHit) {
1281
2299
  const next = failover("quota-failover");
1282
2300
  journal.append("quota-failover", t.id, { from: channelKey(assignment), to: next ? channelKey(next) : null });
@@ -1316,7 +2334,10 @@ export async function runDaemon(repoRoot, opts = {}) {
1316
2334
  // cap above is spent, so a transient blip still recovers in place.
1317
2335
  // OBS-117 (v1.71 T6): a silent launch failure has no CLI signature to parse — the same
1318
2336
  // setup-required typed dead-channel path a late-harvest "command not found" would take.
1319
- const dead = classifyDeadChannel(result) ?? (earlyLaunchDead ? "setup-required" : undefined);
2337
+ // T2 review: classify the PRE-HARVEST parse — classifyDeadChannel bails on any ok:true
2338
+ // result, so reading the synthesized harvest result would swallow auth-required /
2339
+ // setup-required / provider-outage for every committed-but-walled attempt (in both modes).
2340
+ const dead = classifyDeadChannel(preHarvestResult) ?? (earlyLaunchDead ? "setup-required" : undefined);
1320
2341
  if (dead) {
1321
2342
  const from = channelKey(assignment);
1322
2343
  demotedChannels.add(from); // excluded for later attempts AND later tasks in this run
@@ -1382,33 +2403,36 @@ export async function runDaemon(repoRoot, opts = {}) {
1382
2403
  }
1383
2404
  const onGate = async (e) => {
1384
2405
  if (e.phase === "start") {
1385
- journal.phaseStart(t.id, phaseForGate(e.gate), { gate: e.gate, index: e.index, total: e.total });
2406
+ notePhaseStart(e);
2407
+ journal.phaseStart(t.id, phaseForGate(e.gate), { gate: e.gate, index: e.index, total: e.total, ...(e.parentAt === undefined ? {} : { parallel: true }) });
1386
2408
  return;
1387
2409
  }
1388
2410
  const g = e.result;
1389
- // GATE-09 (ROADMAP SC-4): journal every judge retry as an attributable event — which gate flaked,
1390
- // which channel flaked, which channel retried — so `tickmarkr journal`/report can distinguish "judge
1391
- // flaked, retried" from "worker failed" (run-20260711-185020 P43-03 L70-72 billed a judge flake as
1392
- // a worker attempt; 47-01 fixed WHO retries, this closes the audit-trail half). The condition is
1393
- // META-ONLY (D-03): gate === "acceptance" + typeof-shape guards on meta.judgeRetry — never a
1394
- // details-regex. The v1.1 review regex below is grandfathered, not precedent. Appended BEFORE the
1395
- // gate-result so attribution precedes the verdict in the stream. secondUnparseable is derived from
1396
- // the final result's meta.unparseable (set by run-gates when the retry ALSO flaked — double-garbage).
1397
- if (g.gate === "acceptance" && typeof g.meta?.judgeRetry === "object" && g.meta.judgeRetry !== null) {
1398
- const jr = g.meta.judgeRetry;
1399
- if (typeof jr.flaked === "string" && typeof jr.retried === "string") {
1400
- journal.append("judge-retry", t.id, {
1401
- gate: "acceptance", flaked: jr.flaked, retried: jr.retried,
1402
- ...(g.meta.unparseable === true ? { secondUnparseable: true } : {}),
1403
- });
2411
+ inParallelOrder(g.gate, () => {
2412
+ // GATE-09 (ROADMAP SC-4): journal every judge retry as an attributable event — which gate flaked,
2413
+ // which channel flaked, which channel retried — so `tickmarkr journal`/report can distinguish "judge
2414
+ // flaked, retried" from "worker failed" (run-20260711-185020 P43-03 L70-72 billed a judge flake as
2415
+ // a worker attempt; 47-01 fixed WHO retries, this closes the audit-trail half). The condition is
2416
+ // META-ONLY (D-03): gate === "acceptance" + typeof-shape guards on meta.judgeRetry — never a
2417
+ // details-regex. The v1.1 review regex below is grandfathered, not precedent. Appended BEFORE the
2418
+ // gate-result so attribution precedes the verdict in the stream. secondUnparseable is derived from
2419
+ // the final result's meta.unparseable (set by run-gates when the retry ALSO flaked — double-garbage).
2420
+ if (g.gate === "acceptance" && typeof g.meta?.judgeRetry === "object" && g.meta.judgeRetry !== null) {
2421
+ const jr = g.meta.judgeRetry;
2422
+ if (typeof jr.flaked === "string" && typeof jr.retried === "string") {
2423
+ journal.append("judge-retry", t.id, {
2424
+ gate: "acceptance", flaked: jr.flaked, retried: jr.retried,
2425
+ ...(g.meta.unparseable === true ? { secondUnparseable: true } : {}),
2426
+ });
2427
+ }
1404
2428
  }
1405
- }
1406
- journal.append("gate-result", t.id, { gate: g.gate, pass: g.pass, details: g.details, ...(g.meta?.skipped === true ? { skipped: true } : {}) });
1407
- noteReviewRetry(g);
1408
- // v1.1 failover: never re-ask a reviewer channel that produced garbage for this task
1409
- if (g.gate === "review" && !g.pass && /unparseable/.test(g.details) && typeof g.meta?.reviewer === "string") {
1410
- badReviewers.push(g.meta.reviewer);
1411
- }
2429
+ journalGateResult(g);
2430
+ noteReviewRetry(g);
2431
+ // v1.1 failover: never re-ask a reviewer channel that produced garbage for this task
2432
+ if (g.gate === "review" && !g.pass && /unparseable/.test(g.details) && typeof g.meta?.reviewer === "string") {
2433
+ badReviewers.push(g.meta.reviewer);
2434
+ }
2435
+ });
1412
2436
  };
1413
2437
  let results = [];
1414
2438
  let commits = [];
@@ -1418,6 +2442,7 @@ export async function runDaemon(repoRoot, opts = {}) {
1418
2442
  ({ results, commits } = await runGates(t, {
1419
2443
  worktree: wt, baseRef: taskBase, result, author: assignment,
1420
2444
  commands, baseline, channels, adapters, cfg, artifactDir: journal.dir,
2445
+ pipeline: "v185", selectTests: !testGateFailed,
1421
2446
  via: cfg.visibility.llm === "pane"
1422
2447
  ? {
1423
2448
  driver: trackedDriver,
@@ -1439,7 +2464,9 @@ export async function runDaemon(repoRoot, opts = {}) {
1439
2464
  }));
1440
2465
  graph = addEvidence(graph, t.id, { commits, gateResults: results, artifacts: [promptFile] });
1441
2466
  saveGraph(repoRoot, graph);
1442
- if (results.every((g) => g.pass)) {
2467
+ if (results.some((g) => g.gate === "test" && !g.pass))
2468
+ testGateFailed = true;
2469
+ if (results.every(gateSatisfied)) {
1443
2470
  const m = await mergeSerial(taskBranch, t, gated);
1444
2471
  if (m.tipMoved) {
1445
2472
  journal.append("tip-moved", t.id, m.tipMoved);
@@ -1484,21 +2511,113 @@ export async function runDaemon(repoRoot, opts = {}) {
1484
2511
  // v1.53 T3: prefer the CLI's own session id captured from this attempt's output (kimi's resume
1485
2512
  // trailer) over the harness slot name; absent hook or no capture keeps today's slot-name id.
1486
2513
  retrySession = { channel: channelKey(assignment), id: adapter.sessionIdFrom?.(output) ?? sessionId, contextTokens };
1487
- feedback = results.filter((g) => !g.pass).map((g) => `${g.gate}: ${g.details}`).join("\n\n");
2514
+ feedback = results.filter(gateFailed).map((g) => `${g.gate}: ${g.details}`).join("\n\n");
1488
2515
  // OBS-189/G3 (park-economics patch): a request-changes review is a findings brief, not a worker
1489
2516
  // defect — the fix attempt stays on the same channel with the findings as feedback and consumes
1490
2517
  // no escalation-ladder rung. Bounded by the engagement round cap at the top of this loop.
1491
2518
  // Unparseable verdicts (already retried in-gate, OBS-193) and diff-cap trips (the diff cannot
1492
2519
  // shrink by retrying, OBS-48) fall through to the ladder unchanged. Review runs last, so a
1493
2520
  // failed review with every other gate green is exactly "the work landed, the reviewer objects".
1494
- const reviewFail = results.find((g) => g.gate === "review" && !g.pass);
2521
+ const reviewFail = results.find((g) => g.gate === "review" && gateFailed(g));
1495
2522
  const reviewFixRetry = reviewFail !== undefined
1496
2523
  && reviewFail.meta?.unparseable !== true
1497
2524
  && !isDiffCapPark(reviewFail)
1498
- && results.every((g) => g.pass || g.gate === "review");
1499
- const step = reviewFixRetry ? "retry" : r.ladder[Math.min(ladderIdx++, r.ladder.length - 1)];
1500
- journal.append("escalation", t.id, { step, attempt: attempt + 1, ...(reviewFixRetry ? { reviewFix: true } : {}) });
1501
- await driver.notify(`tickmarkr ${runId}: ${t.id} escalation: ${step}`, { tier: "attention" });
2525
+ && results.every((g) => gateSatisfied(g) || g.gate === "review");
2526
+ const failing = results.filter(gateFailed);
2527
+ // Decided before the cap, because the cap's question is whether the NEXT move would re-buy a
2528
+ // measurement already made — and a funded repair is one of the moves that would.
2529
+ const landed = await commitsAheadOf(taskBase, wt);
2530
+ // T4 (OBS-265): judge and review are now ONE round, so a round can report both failing where the
2531
+ // serial walk returned at the judge and review never ran. Eligibility is scored on the battery
2532
+ // the serial contract would have surfaced — review only speaks for a round nothing else failed —
2533
+ // so removing the waiting does not re-price the ladder. The journal below still names every
2534
+ // failing gate, and `feedback` still carries every one of them to the next attempt.
2535
+ const repairBattery = failing.some((g) => g.gate !== "review") ? failing.filter((g) => g.gate !== "review") : failing;
2536
+ const repairable = narrowRepairBattery(repairBattery) && lostCommits.length === 0 && landed.length > 0;
2537
+ const repairsDrawn = repairsSinceApproval(journal.read(), t.id);
2538
+ const repair = repairable && repairsDrawn < MAX_REPAIRS;
2539
+ // v1.85 T3: the fingerprint cap. Two normalized-identical failures of one DETERMINISTIC gate on
2540
+ // one task (volatile tokens — worktree prefixes, line refs, durations, run ids — are not
2541
+ // information) mean the round about to be bought is a re-measurement: ~663m across 5 runs went to
2542
+ // exactly this loop. The threshold is a property of the FAILURE, never of the move that would
2543
+ // follow it, so it is evaluated on EVERY such gate at EVERY ladder position — including a rung
2544
+ // that would change channel, and including a review-fix round. Conditioning it on the next rung
2545
+ // was the first shape of this and let a second identical failure buy an escalate/consult round
2546
+ // the criterion says it may not buy. What the cap does NOT reach is an LLM verdict, which is a
2547
+ // different object with its own tighter bound (see isDeterministicFailure).
2548
+ //
2549
+ // It fires on the CROSSING, not as a latch: the consult it forces and the ban it sets govern the
2550
+ // next move, so re-firing on the third identical failure would only re-buy the round it just paid
2551
+ // for — and the ladder, whose rung this failure still spends, bounds the rest.
2552
+ const repeated = failing.find((g) => isDeterministicFailure(g)
2553
+ && identicalGateFailures(journal.read(), t.id, g.gate, normalizeGateFailure(g.details)) === GATE_FINGERPRINT_CAP);
2554
+ // The rung the cap spent, when it fired — the move below executes THIS instead of drawing a
2555
+ // second one, so a cap costs exactly the rung the failure would have cost anyway.
2556
+ let capStep;
2557
+ if (repeated) {
2558
+ const normalized = normalizeGateFailure(repeated.details);
2559
+ journal.append("gate-fingerprint-cap", t.id, {
2560
+ gate: repeated.gate,
2561
+ occurrences: GATE_FINGERPRINT_CAP,
2562
+ fingerprint: normalized.slice(0, 500),
2563
+ retrySameBanned: true,
2564
+ channel: channelKey(assignment), // the ban is bound to the channel that produced the repeat
2565
+ attempt: attempt + 1,
2566
+ });
2567
+ await driver.notify(`tickmarkr ${runId}: ${t.id} ${repeated.gate} failed identically twice — consulting, identical retry banned`, { tier: "attention" });
2568
+ // The cap takes the ladder's MOVE, never its accounting: this failure still spends the rung it
2569
+ // would have spent, so a task that cannot converge still reaches ladder exhaustion on exactly
2570
+ // the budget it always had and the cap can never hand a stuck task extra rounds.
2571
+ capStep = r.ladder[Math.min(ladderIdx++, r.ladder.length - 1)];
2572
+ journal.append("escalation", t.id, { step: capStep, attempt: attempt + 1, fingerprintCap: true });
2573
+ await driver.notify(`tickmarkr ${runId}: ${t.id} escalation: ${capStep}`, { tier: "attention" });
2574
+ const v = await runConsult("gate-fail-repeat", output, feedback, results);
2575
+ // Rule: a terminal cap consult vetoes a same-channel retry, but cannot veto an `escalate`
2576
+ // rung that already satisfies retry-same-banned by changing channel. Thus terminal+retry
2577
+ // parks through the shared verdict boundary, while terminal+escalate records the consult as
2578
+ // advisory and executes the already-spent rung below. Recoverable verdicts still control the
2579
+ // move directly. The paired fixture asserts both directions of this boundary.
2580
+ const recoverable = v.action === "retry" || v.action === "reroute";
2581
+ if (recoverable || capStep !== "escalate") {
2582
+ if (await applyVerdict(v, attempt + 1, "gate-fail"))
2583
+ continue;
2584
+ return;
2585
+ }
2586
+ journal.append("consult-verdict", t.id, { action: v.action, notes: v.notes, capAdvisory: true });
2587
+ }
2588
+ // v1.85 T3: a narrow battery over fully carried commits earns a REPAIR (decided above) — the
2589
+ // next dispatch carries the findings verbatim and the diff content instead of re-onboarding a
2590
+ // fresh worker. Budget is engagement-scoped and journal-derived, so a resume inherits it rather
2591
+ // than refunding it; the third repair-eligible failure falls back to the fresh ladder.
2592
+ //
2593
+ // `landed` is measured from the worktree rather than read from runGates: runGates returns at
2594
+ // the first failure, so a red test or lint gate never reaches the evidence stage and its
2595
+ // `commits` come back empty — reading them would make the ruling's test/lint case unreachable.
2596
+ //
2597
+ // Budget spent means the FRESH LADDER owns this failure — including a review-only one, whose
2598
+ // same-channel fix retry is exactly the round the budget just declared too expensive to repeat.
2599
+ const repairExhausted = repairable && !repair;
2600
+ if (repair && !capStep) {
2601
+ journal.append("repair-attempt", t.id, {
2602
+ repair: repairsDrawn + 1, of: MAX_REPAIRS, gates: failing.map((g) => g.gate),
2603
+ commits: landed.length,
2604
+ findings: feedback, // the failure bytes this repair must carry, replayable across a resume
2605
+ });
2606
+ }
2607
+ else if (repairExhausted && !capStep) {
2608
+ journal.append("repair-exhausted", t.id, { repairs: repairsDrawn, of: MAX_REPAIRS, gates: failing.map((g) => g.gate) });
2609
+ }
2610
+ const step = capStep ?? (repair || (reviewFixRetry && !repairExhausted)
2611
+ ? "retry"
2612
+ : r.ladder[Math.min(ladderIdx++, r.ladder.length - 1)]);
2613
+ if (!capStep) { // a capped failure already journaled and announced the rung it spent
2614
+ journal.append("escalation", t.id, {
2615
+ step, attempt: attempt + 1,
2616
+ ...(reviewFixRetry && !repairExhausted ? { reviewFix: true } : {}),
2617
+ ...(repair ? { repair: repairsDrawn + 1 } : {}),
2618
+ });
2619
+ await driver.notify(`tickmarkr ${runId}: ${t.id} escalation: ${step}`, { tier: "attention" });
2620
+ }
1502
2621
  if (step === "retry")
1503
2622
  continue;
1504
2623
  if (step === "escalate") {
@@ -1574,24 +2693,7 @@ export async function runDaemon(repoRoot, opts = {}) {
1574
2693
  // OBS-34: post-merge integration-tip verify — strict exit codes, no baseline forgiveness.
1575
2694
  const lastMergedTask = [...journal.read()].reverse().find((e) => e.event === "merge" && e.taskId)?.taskId;
1576
2695
  if (summary.done.length > 0 && Object.keys(commands).length > 0) {
1577
- const tipResults = await verifyIntegrationTip(intWt, commands, journal.dir);
1578
- let tipFailed = false;
1579
- for (const r of tipResults) {
1580
- if (r.pass) {
1581
- journal.append("tip-verify", undefined, { gate: r.gate, cmd: r.cmd, pass: true, exitCode: r.exitCode, details: r.details });
1582
- }
1583
- else {
1584
- journal.append("tip-verify-failed", undefined, {
1585
- gate: r.gate,
1586
- cmd: r.cmd,
1587
- exitCode: r.exitCode,
1588
- fingerprints: r.fingerprints,
1589
- artifact: r.artifact,
1590
- lastMergedTask,
1591
- });
1592
- tipFailed = true;
1593
- }
1594
- }
2696
+ const tipFailed = await verifyIntegrationTipCached(intWt, commands, journal, { lastMergedTask });
1595
2697
  summary.tipVerify = tipFailed ? "failed" : "passed";
1596
2698
  if (tipFailed && lastMergedTask)
1597
2699
  summary.lastMergedTask = lastMergedTask;