tickmarkr 1.84.0 → 1.86.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (87) hide show
  1. package/README.md +4 -2
  2. package/dist/adapters/catalog-remote.d.ts +64 -0
  3. package/dist/adapters/catalog-remote.js +287 -0
  4. package/dist/adapters/catalog.d.ts +96 -0
  5. package/dist/adapters/catalog.js +176 -0
  6. package/dist/adapters/claude-code.d.ts +1 -0
  7. package/dist/adapters/claude-code.js +59 -1
  8. package/dist/adapters/fake.js +42 -4
  9. package/dist/adapters/model-lints.d.ts +25 -5
  10. package/dist/adapters/model-lints.js +184 -50
  11. package/dist/adapters/model-windows.d.ts +31 -0
  12. package/dist/adapters/model-windows.js +69 -0
  13. package/dist/adapters/prompt.d.ts +5 -1
  14. package/dist/adapters/prompt.js +13 -4
  15. package/dist/adapters/registry.d.ts +25 -26
  16. package/dist/adapters/registry.js +173 -110
  17. package/dist/adapters/types.d.ts +3 -0
  18. package/dist/adapters/types.js +36 -3
  19. package/dist/brand.d.ts +5 -1
  20. package/dist/brand.js +18 -2
  21. package/dist/cli/commands/doctor.d.ts +3 -0
  22. package/dist/cli/commands/doctor.js +43 -21
  23. package/dist/cli/commands/fleet.d.ts +7 -0
  24. package/dist/cli/commands/fleet.js +94 -74
  25. package/dist/cli/commands/init.js +118 -5
  26. package/dist/cli/commands/status.js +202 -46
  27. package/dist/compile/collateral.d.ts +86 -2
  28. package/dist/compile/collateral.js +294 -3
  29. package/dist/compile/gsd.d.ts +2 -1
  30. package/dist/compile/gsd.js +68 -2
  31. package/dist/compile/native.d.ts +14 -0
  32. package/dist/compile/native.js +161 -12
  33. package/dist/config/config.d.ts +82 -5
  34. package/dist/config/config.js +253 -66
  35. package/dist/config/fleet-overlay.d.ts +25 -20
  36. package/dist/config/fleet-overlay.js +195 -77
  37. package/dist/config/fleet-why.d.ts +23 -0
  38. package/dist/config/fleet-why.js +42 -0
  39. package/dist/drivers/herdr.d.ts +21 -3
  40. package/dist/drivers/herdr.js +344 -110
  41. package/dist/gates/acceptance.js +7 -2
  42. package/dist/gates/baseline.d.ts +1 -0
  43. package/dist/gates/baseline.js +91 -13
  44. package/dist/gates/llm.d.ts +0 -1
  45. package/dist/gates/llm.js +5 -30
  46. package/dist/gates/review.d.ts +9 -1
  47. package/dist/gates/review.js +105 -10
  48. package/dist/gates/run-gates.d.ts +9 -0
  49. package/dist/gates/run-gates.js +285 -41
  50. package/dist/gates/verdict-cause.d.ts +4 -0
  51. package/dist/gates/verdict-cause.js +63 -0
  52. package/dist/graph/schema.d.ts +6 -0
  53. package/dist/graph/schema.js +8 -5
  54. package/dist/route/router.d.ts +0 -5
  55. package/dist/route/router.js +16 -20
  56. package/dist/run/consult.d.ts +6 -0
  57. package/dist/run/consult.js +35 -25
  58. package/dist/run/daemon.d.ts +48 -2
  59. package/dist/run/daemon.js +1488 -330
  60. package/dist/run/journal.d.ts +56 -3
  61. package/dist/run/journal.js +358 -4
  62. package/dist/run/stall.d.ts +35 -1
  63. package/dist/run/stall.js +118 -8
  64. package/dist/tui/cockpit/capture.d.ts +12 -0
  65. package/dist/tui/cockpit/capture.js +37 -1
  66. package/dist/tui/cockpit/components.d.ts +2 -0
  67. package/dist/tui/cockpit/components.js +8 -8
  68. package/dist/tui/cockpit/derive.d.ts +29 -2
  69. package/dist/tui/cockpit/derive.js +219 -23
  70. package/dist/tui/cockpit/run-cockpit.js +128 -27
  71. package/dist/tui/cockpit/theme.d.ts +32 -26
  72. package/dist/tui/cockpit/theme.js +11 -5
  73. package/dist/tui/ink/components.d.ts +0 -15
  74. package/dist/tui/ink/components.js +0 -17
  75. package/dist/tui/ink/fleet-app.d.ts +4 -1
  76. package/dist/tui/ink/fleet-app.js +134 -13
  77. package/fixtures/sample.native.md +1 -1
  78. package/package.json +1 -1
  79. package/skills/tickmarkr-overseer/SKILL.md +354 -34
  80. package/skills/tickmarkr-overseer/scripts/watch-artifacts.sh +70 -0
  81. package/skills/tickmarkr-overseer/scripts/watch-panes.sh +1 -1
  82. package/dist/tui/ink/studio-app.d.ts +0 -59
  83. package/dist/tui/ink/studio-app.js +0 -320
  84. package/dist/tui/save.d.ts +0 -38
  85. package/dist/tui/save.js +0 -96
  86. package/dist/tui/staging.d.ts +0 -29
  87. package/dist/tui/staging.js +0 -78
@@ -1,6 +1,6 @@
1
- import { randomBytes } from "node:crypto";
1
+ import { createHash, randomBytes } from "node:crypto";
2
2
  import { shq } from "../adapters/types.js";
3
- import { existsSync, mkdirSync, mkdtempSync, readFileSync, rmSync, writeFileSync } from "node:fs";
3
+ import { appendFileSync, closeSync, existsSync, mkdirSync, mkdtempSync, openSync, readFileSync, readSync, rmSync, statSync, writeFileSync } from "node:fs";
4
4
  import { tmpdir } from "node:os";
5
5
  import { join } from "node:path";
6
6
  import { stringify } from "yaml";
@@ -8,7 +8,7 @@ import { classifyDeadChannel, NO_TRAILER_SUMMARY, trailerPattern, UNPARSEABLE_TR
8
8
  import { allAdapters, discoverChannels, getAdapter, probeAll, readDoctor } from "../adapters/registry.js";
9
9
  import { addUsage, channelKey, matchesTrustDialog, QUOTA_RE } from "../adapters/types.js";
10
10
  import { bannerShell, paneDispatchCommand } from "../brand.js";
11
- import { globalConfigDir, loadConfigWithMode, readOverlayFile, repoOverlayPath, } from "../config/config.js";
11
+ import { DEFAULT_DIFF_CAP, globalConfigDir, loadConfigWithMode, readOverlayFile, repoOverlayPath, } from "../config/config.js";
12
12
  import { DeliveryReadinessError } from "../drivers/herdr.js";
13
13
  import { herdrSealShellPrefix, SubprocessDriver } from "../drivers/subprocess.js";
14
14
  import { formatOwnedName } from "../drivers/types.js";
@@ -20,13 +20,13 @@ import { augmentRetryBrief, consult, renderRetryGuidance } from "./consult.js";
20
20
  import { runEnvironment } from "./environment.js";
21
21
  import { cleanupRunWorktrees, gitHead, linkNodeModules, npmDependencyInstallCommand, npmDependencyManifestChanged, sh, shGit, WORKTREE_LAYOUT_CONTRACT, worktreePath } from "./git.js";
22
22
  import { runInteractiveSeed } from "./interactive-seed.js";
23
- import { classifyTaskFailure, classifyWorkerResultCause, engagementComparable, Journal, loadRoutingProfile, newRunId, phaseForGate, recordedTaskFailureKind, reviewRoundsSinceApproval } from "./journal.js";
23
+ import { activeRetryBan, classifyTaskFailure, classifyWorkerResultCause, engagementComparable, GATE_FINGERPRINT_CAP, identicalGateFailures, journaledFailureBrief, Journal, loadRoutingProfile, newRunId, normalizeGateFailure, pendingRepairFindings, phaseForGate, recordedTaskFailureKind, repairsSinceApproval, reviewRoundsSinceApproval, structuredFindings, upheldFeedbackByTask } from "./journal.js";
24
24
  import { isDiffCapPark } from "../gates/review.js";
25
25
  import { acquireRunLock, releaseRunLock } from "./lock.js";
26
26
  import { ensureIntegration, integrationBranch, integrationHead, mergeTask, verifyIntegrationTip } from "./merge.js";
27
27
  import { nextChannel, route } from "../route/router.js";
28
28
  import { desiredPanes } from "./reconcile.js";
29
- import { StallProgressTracker } from "./stall.js";
29
+ import { NUDGEABLE_ADAPTERS, PANE_READ_ROWS, StallProgressTracker, stallSnapshotBannerRows } from "./stall.js";
30
30
  const MODE_RANK = { "staff-led": 0, "risk-based": 1, "partner-led": 2 };
31
31
  // An override (flag/spec) re-resolves through loadConfigWithMode itself, via a synthesized repo overlay
32
32
  // carrying routing.mode — floors, explore, lints, and provenance all come from config.ts's preset
@@ -69,6 +69,107 @@ export function formatSummary(s) {
69
69
  return `done: ${s.done.length}, failed: ${s.failed.length}, human: ${s.human.length}, blocked: ${s.blocked.length}, pending: ${s.pending.length}\nintegration branch: ${s.branch}${tip}`;
70
70
  }
71
71
  const MAX_ATTEMPTS = 10; // ponytail: hard cap so a pathological ladder can never loop forever
72
+ // v1.85 T3 (retry economics): two repairs per engagement, then the fresh ladder. A repair re-uses the
73
+ // findings and the landed diff instead of re-buying onboarding; when two of them have not closed the
74
+ // battery, the cheaper next move is the ladder's channel change, not a third fix-only pass.
75
+ const MAX_REPAIRS = 2;
76
+ /** A named oracle decided this acceptance failure — deterministic, unlike an LLM judge verdict. */
77
+ const isOracleFailure = (g) => g.details.startsWith("oracle failed:");
78
+ /**
79
+ * R3 (OBS-186): a gate that DECLINED to run is not a gate that failed. The review gate's skip branch
80
+ * no longer forges `pass: true` to buy passage, so the merge decision has to read the same predicate
81
+ * the run surfaces already read (src/run/activity.ts): pass, or an honest declared skip. Without this
82
+ * the honesty change would silently park every judge-only task at merge — an unrun gate blocking work
83
+ * it was never asked to review. `skipped` is set only by a gate that says so about ITSELF; a red
84
+ * verdict from a review that actually ran still fails here, exactly as before.
85
+ *
86
+ * ONE pair of predicates, every fold. `!g.pass` was correct only while the sole `pass:false` producer
87
+ * was a gate that actually failed; the moment a decline can be recorded red, every `!g.pass` in this
88
+ * file — the retry feedback brief, the review-fix eligibility test, the failing-battery list the
89
+ * ladder and the fingerprint cap are scored on, the structured findings attached to a blocking
90
+ * verdict — reads an unrun gate as a defect. `gateFailed` is the seam they now share, and the journal
91
+ * write below is the seam every OUT-of-file fold shares.
92
+ */
93
+ const gateSatisfied = (g) => g.pass || g.meta?.skipped === true;
94
+ const gateFailed = (g) => !gateSatisfied(g);
95
+ // v1.85 T3: the gates whose failure IS a deterministic measurement — a machine re-ran a command over a
96
+ // tree and printed the same bytes. Those are the failures the fingerprint cap governs (the ruling names
97
+ // it a "deterministic-gate" cap): a third identical answer to a question already answered twice is the
98
+ // ~663m-across-5-runs loop, whatever rung the ladder happens to stand on. An LLM verdict is a different
99
+ // object — two reviewers, or a judge asked twice, can restate one another without the question being
100
+ // closed — and each already carries a tighter bound of its own: REVIEW_ROUND_CAP parks review at the
101
+ // OPERATOR in two rounds, and a judge verdict rides the ladder and the attempt cap. The boundary is a
102
+ // property of the GATE, never of the ladder rung or of the move that would follow the failure.
103
+ const DETERMINISTIC_GATES = new Set(["build", "test", "lint", "evidence", "scope"]);
104
+ const isDeterministicFailure = (g) => DETERMINISTIC_GATES.has(g.gate) || (g.gate === "acceptance" && isOracleFailure(g));
105
+ /**
106
+ * Is the failing battery narrow enough that a fix-only pass can close it? The ruling's three cases:
107
+ * review-only, a single deterministic test/lint gate, or acceptance decided by a named oracle.
108
+ * Unparseable verdicts and diff-cap trips are excluded exactly as they are from the review-fix retry —
109
+ * neither names anything a worker can fix.
110
+ */
111
+ function narrowRepairBattery(failing) {
112
+ if (failing.length !== 1)
113
+ return false;
114
+ const g = failing[0];
115
+ if (g.gate === "review")
116
+ return g.meta?.unparseable !== true && !isDiffCapPark(g);
117
+ if (g.gate === "test" || g.gate === "lint")
118
+ return true;
119
+ if (g.gate === "acceptance")
120
+ return isOracleFailure(g) && g.meta?.unparseable !== true;
121
+ return false;
122
+ }
123
+ /**
124
+ * T4 (OBS-265): the journal with the review objections a round did NOT hinge on removed. Judge and
125
+ * review are now launched together, so a round can journal a failed review that the serial walk would
126
+ * never have asked for — it returned at the judge. Those verdicts stay on the record (they are real,
127
+ * and the retry brief carries them), but they must not spend the OPERATOR-facing review round budget:
128
+ * otherwise concurrency alone parks a task rounds early for objections the old pipeline never bought.
129
+ * A round is the gate-result span opened by each `gates` phase-start, per task.
130
+ */
131
+ export function decisiveReviewRounds(events) {
132
+ const open = new Map();
133
+ const spent = new Set();
134
+ const close = (taskId) => {
135
+ const round = open.get(taskId) ?? [];
136
+ if (round.some((e) => e.data.gate !== "review" && e.data.pass === false)) {
137
+ for (const e of round)
138
+ if (e.data.gate === "review" && e.data.pass === false)
139
+ spent.add(e);
140
+ }
141
+ open.delete(taskId);
142
+ };
143
+ for (const e of events) {
144
+ if (!e.taskId)
145
+ continue;
146
+ if (e.event === "phase-start" && e.data.phase === "gates")
147
+ close(e.taskId);
148
+ else if (e.event === "gate-result")
149
+ open.set(e.taskId, [...(open.get(e.taskId) ?? []), e]);
150
+ }
151
+ for (const taskId of open.keys())
152
+ close(taskId); // Map iteration tolerates deleting the current key
153
+ return events.filter((e) => !spent.has(e));
154
+ }
155
+ /** The fix-only contract: the findings verbatim, then the diff content of the work already landed. */
156
+ function repairBrief(findings, diff, baseRef) {
157
+ const repair = { findings, diff };
158
+ return [
159
+ "## Repair attempt — fix ONLY what these findings name",
160
+ "The commits from your prior attempt are already in this worktree and their diff is reproduced"
161
+ + " below. Do NOT re-implement that work, do not start over, and do not revert it: make the"
162
+ + " smallest change that resolves every finding, then commit.",
163
+ "",
164
+ "### Failing gate findings (verbatim)",
165
+ repair.findings,
166
+ "",
167
+ `### The work under review (git diff ${baseRef.slice(0, 12)}..HEAD)`,
168
+ "```diff",
169
+ repair.diff,
170
+ "```",
171
+ ].join("\n");
172
+ }
72
173
  // v1.70 T5: request-changes review rounds a single task may draw before it parks for a human decision
73
174
  // instead of cycling. Well below MAX_ATTEMPTS so review non-convergence is caught long before the
74
175
  // global cap. ponytail: literal constant; lift to cfg.review.roundCap only if a second knob-turner appears.
@@ -92,19 +193,23 @@ export function setEarlyLaunchLivenessMsForTests(ms) {
92
193
  export function resetEarlyLaunchLivenessMsForTests() {
93
194
  earlyLaunchLivenessMs = EARLY_LAUNCH_LIVENESS_MS;
94
195
  }
95
- // OBS-201: the liveness nudge — the daemon's ACTIVE response to an idle worker holding no trailer,
96
- // replacing page-a-human-then-burn-the-window (289 of 692 worker-minutes in one measured day).
97
- // Gate: herdr classifies the pane idle AND the monotonic tracker has seen nothing for
98
- // NUDGE_AFTER_SILENT_MS (a worker grinding inside a tool run is `working` and never reaches the
99
- // gate). One nudge per attempt; if the grace passes still idle with no progress, the wait concludes
100
- // as a stall NOW and the consult sees the un-answered nudge instead of an hour of silence.
101
- // Allowlist: claude-code first (steering path proven, OBS-122); widen per adapter only with a
102
- // captured occupied-frame fixture (OBS-181 scar). The message builder takes no nonce — the
196
+ // OBS-201 + T1 (OBS-262): the liveness nudge — the daemon's ACTIVE response to a worker holding no
197
+ // trailer, replacing page-a-human-then-burn-the-window (289 of 692 worker-minutes in one measured
198
+ // day). Gate: NUDGE_AFTER_SILENT_MS of monotonic-tracker silence, regardless of the herdr status
199
+ // reading (unknown/working no longer suppress it — a wedged TUI often scrapes as either); a
200
+ // `blocked` pane still pages instead, since nudging a dialog prompt can't help. One nudge per
201
+ // attempt; if the grace passes with no progress, the wait concludes as a stall NOW and the consult
202
+ // sees the un-answered nudge instead of an hour of silence. Scope allowlist lives in stall.ts
203
+ // (claude-code only; widening is a fixture-capture chore). The message builder takes no nonce — the
103
204
  // self-reference guard holds by construction, an echoed bare token can never match the wait regex.
104
- export const NUDGEABLE_ADAPTERS = new Set(["claude-code"]);
205
+ export { NUDGEABLE_ADAPTERS } from "./stall.js";
105
206
  export const WORKER_NUDGE_MESSAGE = "tickmarkr liveness check: if the task is complete, print your TICKMARKR_RESULT completion trailer exactly as specified in your prompt now. If not, state your next concrete action and continue working.";
106
- const NUDGE_AFTER_SILENT_MS = 3 * 60_000;
207
+ const NUDGE_AFTER_SILENT_MS = 10 * 60_000; // T1 (OBS-262): >=10m tracker silence — was 3m behind an unreachable status gate
107
208
  const WORKER_NUDGE_GRACE_MS = 4 * 60_000;
209
+ // T1 review: a false return from driver.nudge is a DELIVERY outcome (missing pin, readiness
210
+ // stable-frame timeout, read-back hiccup), not proof of an unreachable channel — so one failure
211
+ // is retried once in-slice after this settle, and only a failed retry latches nudgeFailed.
212
+ const NUDGE_REDELIVER_MS = 2_000;
108
213
  let nudgeAfterSilentMs = NUDGE_AFTER_SILENT_MS;
109
214
  let workerNudgeGraceMs = WORKER_NUDGE_GRACE_MS;
110
215
  /** Test seam — shrink the nudge gate and grace without minute-long sleeps. */
@@ -116,6 +221,375 @@ export function resetNudgeTimingForTests() {
116
221
  nudgeAfterSilentMs = NUDGE_AFTER_SILENT_MS;
117
222
  workerNudgeGraceMs = WORKER_NUDGE_GRACE_MS;
118
223
  }
224
+ // T1 (OBS-263): in-loop quota-banner classification — the banner IS output, so the empty-output
225
+ // rules can never catch it and the post-loop QUOTA_RE check only runs after the full window. Two
226
+ // consecutive matching slices plus this much monotonic-tracker silence classify (a worker whose
227
+ // diff merely quotes "rate limit" keeps working undisturbed).
228
+ const QUOTA_BANNER_SILENT_MS = 3 * 60_000;
229
+ let quotaBannerSilentMs = QUOTA_BANNER_SILENT_MS;
230
+ /** Test seam — shrink the quota-banner silence gate without minute-long sleeps. */
231
+ export function setQuotaBannerSilentMsForTests(ms) {
232
+ quotaBannerSilentMs = ms;
233
+ }
234
+ export function resetQuotaBannerSilentMsForTests() {
235
+ quotaBannerSilentMs = QUOTA_BANNER_SILENT_MS;
236
+ }
237
+ // T1 (OBS-262): the operator page is UNLATCHED — every eligible slice journals a page, and the
238
+ // notification is delivered again on a status change or once this cadence elapses. The cadence is
239
+ // an operator-spam guard only; it can no longer turn a stall into a single page forever. It sits
240
+ // BELOW the dead-channel fast-kill window on purpose (T1 review): at 5m == 5m the second delivery
241
+ // raced the kill on the same slice boundary, so an idle non-nudgeable pane holding no delta — the
242
+ // exact class the repeat exists for — got exactly one page in production.
243
+ const PAGE_REPEAT_MS = 2 * 60_000;
244
+ let pageRepeatMs = PAGE_REPEAT_MS;
245
+ /** Test seam — shrink the repeat-page cadence without minute-long sleeps. */
246
+ export function setPageRepeatMsForTests(ms) {
247
+ pageRepeatMs = ms;
248
+ }
249
+ export function resetPageRepeatMsForTests() {
250
+ pageRepeatMs = PAGE_REPEAT_MS;
251
+ }
252
+ // T1 (R1 dead-channel fast-kill): a worker with no trailer, no worktree delta, and no output
253
+ // growth for this long is dead — conclude immediately instead of burning the rolling window.
254
+ // Per-task timeoutMinutes stays the escape valve for slow-but-live workers.
255
+ const DEAD_CHANNEL_FAST_KILL_MS = 5 * 60_000;
256
+ let deadChannelFastKillMs = DEAD_CHANNEL_FAST_KILL_MS;
257
+ /** Test seam — shrink the fast-kill window without minute-long sleeps. */
258
+ export function setDeadChannelFastKillMsForTests(ms) {
259
+ deadChannelFastKillMs = ms;
260
+ }
261
+ export function resetDeadChannelFastKillMsForTests() {
262
+ deadChannelFastKillMs = DEAD_CHANNEL_FAST_KILL_MS;
263
+ }
264
+ // T2 (OBS-264): finished work is harvested, never redone. 18 of 18 observed stalls carried 2-33
265
+ // commits, and the redispatch then re-bought verification of work that had already landed. The
266
+ // liveness triad — commits made by this attempt, a FLAT worker-tree CPU delta, and this much
267
+ // monotonic-tracker silence — CONCLUDES the wait. Conclude, never kill: the pane is harvested by
268
+ // the same tail a window expiry uses, and the carried worktree goes straight to gates. Set at the
269
+ // fast-kill's window on purpose (the OBS-264 arithmetic is "a ~36m stall + ~15m redo becomes a
270
+ // ~5m gate pass"); a worker that is merely thinking still burns CPU and is never concluded here.
271
+ const HARVEST_SILENT_MS = 5 * 60_000;
272
+ let harvestSilentMs = HARVEST_SILENT_MS;
273
+ /** Test seam — shrink the harvest silence gate without minute-long sleeps. */
274
+ export function setHarvestSilentMsForTests(ms) {
275
+ harvestSilentMs = ms;
276
+ }
277
+ export function resetHarvestSilentMsForTests() {
278
+ harvestSilentMs = HARVEST_SILENT_MS;
279
+ }
280
+ // A CPU delta needs two samples separated in WALL CLOCK, and the CPU clock is QUANTIZED: darwin's
281
+ // `ps` prints hundredths ("0:00.03"), linux's prints whole seconds ("00:00:01"). Equality across a
282
+ // window shorter than the quantum is not evidence of anything — a worker throttled to a low duty
283
+ // cycle accrues less than one tick per sample and reads flat while genuinely working. So the flat
284
+ // observation must span the LARGER of a floor and this many ticks of the clock actually in use:
285
+ // crossing 30 ticks means the tree burned <1 tick in 30, i.e. under ~3% of one core. On a
286
+ // hundredths host that is a 3s window; on a whole-second host it is 30s — still nothing against the
287
+ // ~15m redispatch it replaces. Resolution is read off the sampled rows, never assumed.
288
+ const HARVEST_CPU_FLAT_MS = 3_000;
289
+ const HARVEST_CPU_FLAT_TICKS = 30;
290
+ // Once the flat window opens, retain descendants often enough to observe brief tool processes that
291
+ // can start and exit between the daemon's ordinary wait slices. This sampler exists only during an
292
+ // eligible silence window; it is stopped on progress or as soon as the worker wait concludes.
293
+ const HARVEST_CPU_ACCOUNTING_POLL_MS = 100;
294
+ // T2 review (material): that 100ms cadence forks a shell plus `ps` ten times a second, and on a host
295
+ // where `ps` is unsupported or denied (the managed-sandbox class) EVERY sample fails — tens of
296
+ // thousands of processes per silent attempt, multiplied by daemon concurrency, for a probe that can
297
+ // never conclude anything. Persistent failure is structural, not transient, so the sampler STOPS
298
+ // after this many consecutive unreadable snapshots. It stays stopped for the silence window it was
299
+ // started for: read() then reports no CPU, the triad refuses to conclude and journals the gap, and a
300
+ // later window (after real progress clears the accountant) starts a fresh one that pays the same
301
+ // bounded probe again.
302
+ const HARVEST_CPU_UNMEASURABLE_SAMPLE_CAP = 20;
303
+ let harvestCpuFlatMs;
304
+ export function harvestCpuFlatWindowMs(resolutionMs) {
305
+ return harvestCpuFlatMs ?? Math.max(HARVEST_CPU_FLAT_MS, resolutionMs * HARVEST_CPU_FLAT_TICKS);
306
+ }
307
+ /** Test seam — pin the flat window so a probe case need not sit through a real one. */
308
+ export function setHarvestCpuFlatMsForTests(ms) {
309
+ harvestCpuFlatMs = ms;
310
+ }
311
+ export function resetHarvestCpuFlatMsForTests() {
312
+ harvestCpuFlatMs = undefined;
313
+ }
314
+ // Once the silence gate is met the CPU probe owns the poll cadence: the trailer-wait slice is 30s,
315
+ // so two samples would otherwise cost a minute of wall clock apiece. Below the gate the only rule
316
+ // is not to sleep PAST it — at the shipped 5m gate that changes no slice a worker sees today.
317
+ const HARVEST_POLL_MS = 2_000;
318
+ function harvestSliceMs(silentMs) {
319
+ return Math.max(100, silentMs >= harvestSilentMs ? HARVEST_POLL_MS : harvestSilentMs - silentMs);
320
+ }
321
+ // The synthesized result a carried no-trailer harvest hands to the gates. Distinct from anything a
322
+ // worker can claim: it never comes from adapter.parse, and it is journaled under its own event.
323
+ export const HARVESTED_RESULT_SUMMARY = "harvested: the worktree carries committed work; the worker emitted no TICKMARKR_RESULT trailer";
324
+ /** T4 (OBS-266): identity of the command SET a tip verify ran — a changed command is a different verify. */
325
+ export function commandsHash(commands) {
326
+ return createHash("sha256").update(JSON.stringify(Object.entries(commands).sort())).digest("hex").slice(0, 12);
327
+ }
328
+ /**
329
+ * T4 (OBS-266): the journal's LAST verification cycle — the (tip, cmdHash) pair the most recent run
330
+ * of the verify commands spoke for, the gates it got a pass from, and whether anything failed in it.
331
+ *
332
+ * The LAST one, never a history of every pair ever green. "The last GREEN verified SHA" is what the
333
+ * spec licenses a skip against, and only the last cycle is a statement about the state the run is in
334
+ * now: after A→B→A the tip really moved, and after commands A→B→A the last thing that ran on this
335
+ * SHA was command set B — both re-verify. A cycle is the contiguous run of events sharing one pair,
336
+ * so a cycle cut short by a killed process is missing gates and can never satisfy the caller. A
337
+ * legacy event (no tip/cmdHash) is unattributable and breaks the chain outright.
338
+ */
339
+ function lastVerifyCycle(events) {
340
+ let cur;
341
+ let afterRunEnd = false;
342
+ for (const e of events) {
343
+ // New journals delimit every attempt explicitly. run-end is the legacy delimiter: it starts a
344
+ // new cycle only when another verify event follows, while preserving the just-closed cycle as
345
+ // the cache candidate for an otherwise unmoved next run-end.
346
+ if (e.event === "tip-verify-start") {
347
+ const { tip, cmdHash } = e.data;
348
+ cur = typeof tip === "string" && typeof cmdHash === "string"
349
+ ? { tip, cmdHash, gates: new Set(), failed: false }
350
+ : undefined;
351
+ afterRunEnd = false;
352
+ continue;
353
+ }
354
+ if (e.event === "run-end") {
355
+ afterRunEnd = true;
356
+ continue;
357
+ }
358
+ if (e.event !== "tip-verify" && e.event !== "tip-verify-failed")
359
+ continue;
360
+ const { tip, gate, cmdHash } = e.data;
361
+ if (typeof tip !== "string" || typeof gate !== "string" || typeof cmdHash !== "string") {
362
+ cur = undefined;
363
+ afterRunEnd = false;
364
+ continue;
365
+ }
366
+ if (!cur || afterRunEnd || cur.tip !== tip || cur.cmdHash !== cmdHash) {
367
+ cur = { tip, cmdHash, gates: new Set(), failed: false };
368
+ }
369
+ afterRunEnd = false;
370
+ if (e.event === "tip-verify-failed")
371
+ cur.failed = true;
372
+ else
373
+ cur.gates.add(gate);
374
+ }
375
+ return cur;
376
+ }
377
+ /**
378
+ * OBS-34's strict tip verify, but it stops re-paying for an unmoved tip (~334m corpus-wide; 69.5m in
379
+ * one park-heavy run whose 18 resume cycles merged nothing new). The verify journals the SHA it
380
+ * verified and the hash of the command set, so a later run-end can recognize the same verified state:
381
+ * head equals the LAST green verified SHA, commands unchanged, tree clean → journal
382
+ * `tip-verify-cached` and skip. ANY doubt — moved head, changed commands, dirty tree, a gate missing
383
+ * from that cycle's green set, a failure recorded in it, a cycle older than the last one — runs the
384
+ * full verify. The tip-verify-before-green law is untouched: a cached green is a verified green OF
385
+ * THAT EXACT COMMIT, established by the most recent real run of the same commands.
386
+ * Returns whether the tip is failing.
387
+ */
388
+ export async function verifyIntegrationTipCached(intWt, commands, journal, opts = {}) {
389
+ const cmdHash = commandsHash(commands);
390
+ const tip = await gitHead(intWt);
391
+ const porcelain = await shGit("git status --porcelain", intWt);
392
+ const clean = porcelain.code === 0 && porcelain.stdout.trim() === "";
393
+ const last = lastVerifyCycle(journal.read());
394
+ const cached = last !== undefined && !last.failed && last.tip === tip && last.cmdHash === cmdHash
395
+ && Object.keys(commands).every((g) => last.gates.has(g));
396
+ // A pair can be verified red and then green without either SHA or command hash changing (for
397
+ // example, an external service or ignored fixture recovers). Delimit attempts explicitly so that
398
+ // the earlier red cannot remain latched into the later complete green cycle.
399
+ journal.append("tip-verify-start", undefined, { tip, cmdHash, gates: Object.keys(commands), cached: clean && cached });
400
+ if (clean && cached) {
401
+ journal.append("tip-verify-cached", undefined, { tip, cmdHash, gates: Object.keys(commands) });
402
+ // The skip must not read as a red. Every surface derives the tip's verdict from this cycle's
403
+ // `tip-verify` events (cockpit derive.ts tipVerificationPassed: a run-end claiming "passed" with
404
+ // ZERO events is fail-closed to FALSE), so a carried-forward green still journals its per-gate
405
+ // pass — `cached: true` keeps it honest about not having re-run the command.
406
+ for (const gate of Object.keys(commands)) {
407
+ journal.append("tip-verify", undefined, { gate, cmd: commands[gate], pass: true, exitCode: 0, cached: true, tip, cmdHash });
408
+ }
409
+ return false;
410
+ }
411
+ let tipFailed = false;
412
+ for (const r of await verifyIntegrationTip(intWt, commands, journal.dir)) {
413
+ if (r.pass) {
414
+ journal.append("tip-verify", undefined, { gate: r.gate, cmd: r.cmd, pass: true, exitCode: r.exitCode, details: r.details, tip, cmdHash });
415
+ }
416
+ else {
417
+ journal.append("tip-verify-failed", undefined, {
418
+ gate: r.gate,
419
+ cmd: r.cmd,
420
+ exitCode: r.exitCode,
421
+ fingerprints: r.fingerprints,
422
+ artifact: r.artifact,
423
+ lastMergedTask: opts.lastMergedTask,
424
+ tip,
425
+ cmdHash,
426
+ });
427
+ tipFailed = true;
428
+ }
429
+ }
430
+ return tipFailed;
431
+ }
432
+ // `ps` CPU time: "[[dd-]hh:]mm:ss[.frac]" (darwin prints "0:00.03", linux "00:00:01", both print
433
+ // "1-02:03:04" past a day). Anything else is a header or a row this parser must not guess at.
434
+ // `frac` reports whether THIS host prints sub-second digits — the quantum the flat window is sized
435
+ // against, measured rather than assumed (a darwin sample is 10ms, a linux one 1000ms).
436
+ function parsePsCpu(raw) {
437
+ const m = /^(?:(\d+)-)?(?:(\d+):)?(\d+):(\d+(?:\.\d+)?)$/.exec(raw);
438
+ if (!m)
439
+ return undefined;
440
+ const ms = ((Number(m[1] ?? 0) * 24 + Number(m[2] ?? 0)) * 60 + Number(m[3])) * 60_000 + Math.round(Number(m[4]) * 1000);
441
+ return { ms, frac: m[4].includes(".") };
442
+ }
443
+ let linuxClockTickMs;
444
+ function linuxProcessCpuMs(pid, cwd) {
445
+ if (!existsSync("/proc/self/stat"))
446
+ return Promise.resolve(undefined);
447
+ // shGit, not sh: the accountant samples this path at a 100ms cadence, and a LOGIN shell would
448
+ // re-run the operator's profile (nvm/pyenv/direnv side effects included) on every sample.
449
+ linuxClockTickMs ??= shGit("getconf CLK_TCK", cwd, 15_000).then((r) => {
450
+ const ticks = r.code === 0 ? Number(r.stdout.trim()) : Number.NaN;
451
+ return Number.isFinite(ticks) && ticks > 0 ? 1_000 / ticks : undefined;
452
+ });
453
+ return linuxClockTickMs.then((resolutionMs) => {
454
+ if (resolutionMs === undefined)
455
+ return undefined;
456
+ try {
457
+ // `/proc/<pid>/stat` fields 14-17 are user/system jiffies for the process and its waited-for
458
+ // children. The child totals retain tools that start and exit wholly between live-tree polls.
459
+ // Split after the LAST ')' because comm may contain spaces or parentheses; field 3 is rest[0].
460
+ const stat = readFileSync(`/proc/${pid}/stat`, "utf8");
461
+ const fields = stat.slice(stat.lastIndexOf(")") + 2).trim().split(/\s+/);
462
+ const ticks = Number(fields[11]) + Number(fields[12]) + Number(fields[13]) + Number(fields[14]);
463
+ return Number.isFinite(ticks) ? { ms: ticks * resolutionMs, resolutionMs } : undefined;
464
+ }
465
+ catch {
466
+ return undefined; // process exited between ps ancestry capture and the precise CPU read
467
+ }
468
+ });
469
+ }
470
+ // T2 (OBS-264): the triad's CPU leg. Every non-seeded process of an attempt descends from that
471
+ // attempt's own dispatch script, whose path is unique — print, argv-interactive and resume launches
472
+ // all start there. (interactive-seed is intentionally fail-open below because its adapter-owned
473
+ // launch bypasses this script.) ONE `ps` snapshot finds the root and all current descendants:
474
+ // the agent CLI is a CHILD of the script's shell, so the root's own TIME never moves while the CLI
475
+ // thinks. `resolutionMs` is the sampled clock's quantum, which sizes the caller's flat window.
476
+ // Returns 0 when nothing matches: a worker whose process tree is gone is the strongest possible
477
+ // "not working". Returns undefined when the snapshot itself failed or parsed to nothing —
478
+ // unmeasurable CPU is never evidence a worker stopped, and the caller refuses to conclude on it.
479
+ async function workerTreeCpuSnapshot(marker, cwd) {
480
+ // shGit, not sh: same login-shell cost as the CLK_TCK probe above — `ps` needs no profile.
481
+ const snapshot = await shGit("ps -Awwo pid=,ppid=,time=,command=", cwd, 15_000);
482
+ if (snapshot.code !== 0)
483
+ return undefined;
484
+ const rows = [];
485
+ for (const line of snapshot.stdout.split("\n")) {
486
+ const m = /^\s*(\d+)\s+(\d+)\s+(\S+)\s+(.*)$/.exec(line);
487
+ if (!m)
488
+ continue;
489
+ const cpu = parsePsCpu(m[3]);
490
+ if (cpu !== undefined)
491
+ rows.push({ pid: m[1], ppid: m[2], cpuMs: cpu.ms, frac: cpu.frac, cmd: m[4] });
492
+ }
493
+ if (rows.length === 0)
494
+ return undefined;
495
+ const tree = new Set(rows.filter((p) => p.cmd.includes(marker)).map((p) => p.pid));
496
+ // `ps` output is not topologically ordered — relax the parent→child closure until it stops growing.
497
+ for (let grew = true; grew;) {
498
+ grew = false;
499
+ for (const p of rows) {
500
+ if (!tree.has(p.pid) && tree.has(p.ppid)) {
501
+ tree.add(p.pid);
502
+ grew = true;
503
+ }
504
+ }
505
+ }
506
+ const precise = new Map();
507
+ let preciseResolutionMs;
508
+ for (const p of rows) {
509
+ if (!tree.has(p.pid))
510
+ continue;
511
+ const cpu = await linuxProcessCpuMs(p.pid, cwd);
512
+ precise.set(p.pid, cpu?.ms ?? p.cpuMs);
513
+ if (cpu !== undefined)
514
+ preciseResolutionMs = cpu.resolutionMs;
515
+ }
516
+ // Even an empty worker tree needs the host's actual measurement quantum: on Linux the /proc
517
+ // jiffy clock remains available after the worker exits, while `ps time` only prints whole seconds.
518
+ if (preciseResolutionMs === undefined && existsSync("/proc/self/stat")) {
519
+ preciseResolutionMs = (await linuxProcessCpuMs(String(process.pid), cwd))?.resolutionMs;
520
+ }
521
+ return {
522
+ processes: precise,
523
+ resolutionMs: preciseResolutionMs ?? (rows.some((p) => p.frac) ? 10 : 1_000),
524
+ };
525
+ }
526
+ export async function workerTreeCpuMs(marker, cwd) {
527
+ const snapshot = await workerTreeCpuSnapshot(marker, cwd);
528
+ if (snapshot === undefined)
529
+ return undefined;
530
+ return {
531
+ ms: [...snapshot.processes.values()].reduce((sum, cpuMs) => sum + cpuMs, 0),
532
+ resolutionMs: snapshot.resolutionMs,
533
+ };
534
+ }
535
+ // Sparse live-tree totals forget a tool's CPU as soon as that tool exits. This attempt-local
536
+ // accountant instead adds each observed process's CPU DELTA to a monotonic total and replaces only
537
+ // the live-PID cursor on each sample. When a PID disappears, its contribution stays in `totalMs`;
538
+ // if that PID is later reused, its fresh total is added from zero because it left `live` in between.
539
+ class WorkerTreeCpuAccountant {
540
+ marker;
541
+ cwd;
542
+ active = false;
543
+ loop;
544
+ live = new Map();
545
+ totalMs = 0;
546
+ gaps = 0;
547
+ consecutiveGaps = 0;
548
+ latest;
549
+ constructor(marker, cwd) {
550
+ this.marker = marker;
551
+ this.cwd = cwd;
552
+ }
553
+ async sample() {
554
+ const snapshot = await workerTreeCpuSnapshot(this.marker, this.cwd);
555
+ if (snapshot === undefined) {
556
+ this.gaps++;
557
+ this.live.clear();
558
+ this.latest = undefined;
559
+ // Stop forking `ps` at 10Hz once the host has proved it cannot answer — see the cap's comment.
560
+ if (++this.consecutiveGaps >= HARVEST_CPU_UNMEASURABLE_SAMPLE_CAP)
561
+ this.active = false;
562
+ return;
563
+ }
564
+ this.consecutiveGaps = 0;
565
+ for (const [pid, cpuMs] of snapshot.processes) {
566
+ const prior = this.live.get(pid);
567
+ this.totalMs += prior === undefined || cpuMs < prior ? cpuMs : cpuMs - prior;
568
+ }
569
+ this.live = snapshot.processes;
570
+ this.latest = { ms: this.totalMs, resolutionMs: snapshot.resolutionMs };
571
+ }
572
+ async start() {
573
+ if (this.active)
574
+ return;
575
+ this.active = true;
576
+ await this.sample();
577
+ this.loop = (async () => {
578
+ while (this.active) {
579
+ await new Promise((resolve) => setTimeout(resolve, HARVEST_CPU_ACCOUNTING_POLL_MS));
580
+ if (this.active)
581
+ await this.sample();
582
+ }
583
+ })();
584
+ }
585
+ read() {
586
+ return { cpu: this.latest, gaps: this.gaps };
587
+ }
588
+ async stop() {
589
+ this.active = false;
590
+ await this.loop;
591
+ }
592
+ }
119
593
  async function commitsAheadOf(base, wt) {
120
594
  const head = await gitHead(wt);
121
595
  if (head === base)
@@ -125,6 +599,26 @@ async function commitsAheadOf(base, wt) {
125
599
  return [];
126
600
  return r.stdout.trim().split("\n").filter(Boolean);
127
601
  }
602
+ // T1 (R1 fast-kill): worktree delta = commits ahead of the task base OR any uncommitted change.
603
+ // The fast-kill may only run once the ABSENCE of work has been positively established, so every
604
+ // probe fails OPEN toward "delta": a throwing rev-parse, a non-zero `git log`, and a non-zero
605
+ // `git status` all report delta. commitsAheadOf() is deliberately NOT reused here — it converts a
606
+ // failed log into an empty list, which reads as "no commits" and would kill a live worker.
607
+ async function worktreeHasDelta(wt, base) {
608
+ try {
609
+ const head = await gitHead(wt);
610
+ if (head !== base) {
611
+ const log = await shGit(`git log --reverse --format=%H ${shq(base)}..${shq(head)}`, wt);
612
+ if (log.code !== 0 || log.stdout.trim().length > 0)
613
+ return true;
614
+ }
615
+ const r = await shGit("GIT_OPTIONAL_LOCKS=0 git status --porcelain", wt);
616
+ return r.code !== 0 || r.stdout.trim().length > 0;
617
+ }
618
+ catch {
619
+ return true; // an unreadable worktree is never evidence that the worker did nothing
620
+ }
621
+ }
128
622
  async function cherryPickCommits(wt, commits) {
129
623
  const carried = [];
130
624
  for (const hash of commits) {
@@ -137,6 +631,71 @@ async function cherryPickCommits(wt, commits) {
137
631
  }
138
632
  return carried;
139
633
  }
634
+ // T7 (v1.86): a first run-end append that fails AFTER partial bytes landed leaves a torn tail at
635
+ // EOF with no newline; a blind retry would glue the run-end line onto those bytes and readJsonl's
636
+ // torn-line tolerance would drop the retry too — no terminal record despite a successful write.
637
+ // Terminating the torn fragment keeps it on disk (dropped as a malformed line, never truncated) so
638
+ // the retried run-end lands on a line of its own.
639
+ const terminateTornJournalTail = (journalPath) => {
640
+ if (!existsSync(journalPath))
641
+ return;
642
+ const size = statSync(journalPath).size;
643
+ if (size === 0)
644
+ return;
645
+ const fd = openSync(journalPath, "r");
646
+ try {
647
+ const tail = Buffer.alloc(1);
648
+ if (readSync(fd, tail, 0, 1, size - 1) === 1 && tail[0] !== 0x0a)
649
+ appendFileSync(journalPath, "\n");
650
+ }
651
+ finally {
652
+ closeSync(fd);
653
+ }
654
+ };
655
+ // T7 (v1.86): the fatal handler must never eat the error it reports. Both journal calls are guarded,
656
+ // so a sink failure is reported ALONGSIDE the original error (console.error — the dispatcher's
657
+ // operator-visible line stays the one-line form), never instead of it: a read failure degrades the
658
+ // duplicate-run-end check to "unknown" and fails toward recording; the append is retried ONCE; and
659
+ // a persistently unwritable sink reports a crash naming the journal path and carrying NO terminal
660
+ // record, rather than fabricating an ended run on evidence the harness could not write. (OBS-313:
661
+ // with the sink dead the crash CAUSE is unrecordable — recorded as an observation, not papered over.)
662
+ function recordFatalRunEnd(journal, runId, branch, err) {
663
+ const original = err instanceof Error ? err.message : String(err);
664
+ try {
665
+ if (journal.read().some((e) => e.event === "run-end"))
666
+ return; // already terminal — nothing to add
667
+ }
668
+ catch (readErr) {
669
+ console.error(`tickmarkr ${runId}: journal read failed while recording the fatal run-end (${readErr instanceof Error ? readErr.message : String(readErr)}) — original error: ${original}`);
670
+ }
671
+ const record = {
672
+ runId,
673
+ branch,
674
+ done: [],
675
+ failed: [],
676
+ human: [],
677
+ blocked: [],
678
+ pending: [],
679
+ phase: "setup",
680
+ fatal: true,
681
+ error: original,
682
+ };
683
+ try {
684
+ journal.append("run-end", undefined, record);
685
+ return;
686
+ }
687
+ catch {
688
+ // one retry below — the first failure is reported only if the retry also fails
689
+ }
690
+ const journalPath = join(journal.dir, "journal.jsonl");
691
+ try {
692
+ terminateTornJournalTail(journalPath);
693
+ journal.append("run-end", undefined, record);
694
+ }
695
+ catch (retryErr) {
696
+ console.error(`tickmarkr ${runId}: run crashed — no terminal record written; journal sink unwritable at ${journalPath} (${retryErr instanceof Error ? retryErr.message : String(retryErr)}) — original error: ${original}`);
697
+ }
698
+ }
140
699
  export async function runDaemon(repoRoot, opts = {}) {
141
700
  // v1.51 T2 / OBS-89 (v1.60): retired --quality env seam. Mode resolution owns premium routing;
142
701
  // route() no longer reads the retired env at all, so the old entrypoint scrub is gone with it.
@@ -295,7 +854,12 @@ export async function runDaemon(repoRoot, opts = {}) {
295
854
  for (const [id, st] of journal.replayStatuses()) {
296
855
  if (opts.retryFailed && st === "failed" && recordedTaskFailureKind(replayEvents, id) === "dispatch") {
297
856
  graph = setStatus(graph, id, "pending");
298
- resume.delete(id);
857
+ // OBS-254: clear ATTEMPT AND CHANNEL STATE ONLY. Deleting the whole entry also deleted
858
+ // upheldFeedback — the operator's funded brief — and the next dispatch advertised an empty
859
+ // "fix these specifically" heading. A dispatch that died before worker-result then re-ran the
860
+ // worker with the uphold's findings silently gone.
861
+ const prior = resume.get(id);
862
+ resume.set(id, { attempts: 0, tried: [], ...(prior?.upheldFeedback ? { upheldFeedback: prior.upheldFeedback } : {}) });
299
863
  continue;
300
864
  }
301
865
  // operator release: a graph.json edit back to "pending" beats a replayed human/failed park (locked decision 12)
@@ -485,7 +1049,7 @@ export async function runDaemon(repoRoot, opts = {}) {
485
1049
  // reviewer-exclusion list (badReviewers), never a second parallel counter. OBS-189: scoped to the
486
1050
  // current engagement — an operator approval (uphold or accept) resets the round budget, so an upheld
487
1051
  // task can dispatch its funded attempt instead of re-parking against the whole journal's history.
488
- const reviewRoundsDrawn = () => reviewRoundsSinceApproval(journal.read(), t.id);
1052
+ const reviewRoundsDrawn = () => reviewRoundsSinceApproval(decisiveReviewRounds(journal.read()), t.id);
489
1053
  // OBS-193: journal the in-gate review retry (mirrors judge-retry) and exclude the flaked seat from
490
1054
  // later attempts' reviewer picks. One helper, called from both onGate sites (satisfied-gate + main).
491
1055
  const noteReviewRetry = (g) => {
@@ -498,9 +1062,85 @@ export async function runDaemon(repoRoot, opts = {}) {
498
1062
  badReviewers.push(rr.flaked);
499
1063
  }
500
1064
  };
1065
+ // v1.85 T3 (ruling R4): every BLOCKING review/judge result lands its findings in the journal
1066
+ // structured — class + canonical path + stable symbol — so a retry, a consult or an auto-uphold
1067
+ // decision reads identity instead of re-parsing prose, and line-number churn is not a new finding.
1068
+ // One helper, both onGate sites (satisfied-gate resume + main attempt loop).
1069
+ const journalGateResult = (g) => {
1070
+ const blocking = gateFailed(g) && (g.gate === "review" || g.gate === "acceptance");
1071
+ // R3 (OBS-186): a gate that DECLINED has no verdict to state, and this row is the ONE seam every
1072
+ // fold outside this file shares. Writing `pass: false` for a decline is what turned a skip into
1073
+ // a failure at all of them at once — the engagement round budget (reviewRoundsSinceApproval,
1074
+ // journal.ts), the operator's failed-gate list (cli/commands/approve.ts), the record's
1075
+ // gate-failure total (cli/commands/report.ts), the cockpit's gate rows (tui/cockpit/derive.ts).
1076
+ // Each keys on `pass === false`; none of them is reachable from this task's file scope, and
1077
+ // patching five copies of the same question would be the wrong fix even if they were. So the
1078
+ // ledger simply does not claim a verdict it does not have.
1079
+ // The legacy baseline declines (a build command the repo never configured) have always written
1080
+ // `pass: true` beside `skipped: true` and every consumer already reads them right, so their row
1081
+ // is untouched: only a decline that would otherwise be recorded RED changes shape here.
1082
+ const unverdicted = g.meta?.skipped === true && !g.pass;
1083
+ journal.append("gate-result", t.id, {
1084
+ gate: g.gate, ...(unverdicted ? {} : { pass: g.pass }), details: g.details,
1085
+ ...(g.meta?.skipped === true ? { skipped: true } : {}),
1086
+ // R3 (OBS-186): a declined review is journal truth, not an absence. `skipped: true` alone
1087
+ // says a gate did not run; these say WHICH policy declined it and WHY, so a reader of the
1088
+ // ledger never has to infer participation from a details string. The green-skip branch that
1089
+ // made this row indistinguishable from a pass is gone (src/gates/review.ts).
1090
+ ...(g.meta?.verdict === "skipped"
1091
+ ? { verdict: "skipped", policy: g.meta.policy, reason: g.meta.reason }
1092
+ : {}),
1093
+ // T4 (OBS-265): a test verdict says WHICH suite spoke. A round runs the selected subset as a
1094
+ // screen and the full suite as the verdict, so without these two the journal would carry a
1095
+ // `test` row whose scope no consumer could recover.
1096
+ ...(Array.isArray(g.meta?.selectedTests) ? { selectedTests: g.meta.selectedTests } : {}),
1097
+ ...(g.meta?.fullSuite === true ? { fullSuite: true } : {}),
1098
+ // A finding's path is its own evidence path. Do not pass task scope here: a declaration says
1099
+ // where work is allowed, not where this verdict found the defect.
1100
+ ...(blocking ? { findings: structuredFindings(g.gate, g.details) } : {}),
1101
+ });
1102
+ };
1103
+ // R3 (OBS-186): judge ‖ review are launched together and publish in COMPLETION order
1104
+ // (run-gates.ts) — a race. Three oracles assert the opposite: a scripted run's journal is
1105
+ // byte-identical run to run (tests/run/narration.test.ts, tests/run/notify-identity.test.ts), and
1106
+ // a round's gate-result order matches its phase-start order (tests/run/daemon.test.ts). Retiring
1107
+ // complexityThreshold is what REACHES this, not what introduces it: those fixtures used to skip
1108
+ // review and journal ONE verdict row per round, so the pair's order was never exercised — and the
1109
+ // operator's config has run `complexityThreshold: 0` since 2026-07-31, so production rounds have
1110
+ // journaled both siblings all along.
1111
+ //
1112
+ // Ordering is the LEDGER's job, not the pipeline's. run-gates still reports each completion the
1113
+ // instant it happens; the daemon writes its ledger in GATE_NAMES order. Only the LATER gate is
1114
+ // ever held, and only while an earlier sibling is still in flight — a review that finishes first
1115
+ // waits for acceptance, never the reverse. That keeps T4's durability where it pays (the first
1116
+ // verdict to land is still published immediately) and bounds the exposure to one row for the
1117
+ // remainder of one already-running gate. A gate that THROWS kills the round before merge, so a
1118
+ // row held behind it is lost with the round it belonged to — not a verdict that could have merged.
1119
+ const parallelPending = new Set();
1120
+ let heldParallel;
1121
+ const notePhaseStart = (e) => {
1122
+ if (e.parentAt !== undefined)
1123
+ parallelPending.add(e.gate);
1124
+ };
1125
+ const inParallelOrder = (gate, publish) => {
1126
+ parallelPending.delete(gate);
1127
+ const rank = GATE_NAMES.indexOf(gate);
1128
+ if ([...parallelPending].some((p) => GATE_NAMES.indexOf(p) < rank)) {
1129
+ heldParallel = publish;
1130
+ return;
1131
+ }
1132
+ publish();
1133
+ const held = heldParallel;
1134
+ heldParallel = undefined;
1135
+ held?.();
1136
+ };
501
1137
  // OBS-189: the operator upheld the reviewer — the findings ARE the brief for this funded attempt.
502
- let feedback = rs?.upheldFeedback
503
- ? `The operator UPHELD the reviewer's findings — address them without discarding landed work.\nreview: ${rs.upheldFeedback}`
1138
+ // OBS-254: RE-DERIVED from the journal here, at prompt-build time, rather than trusted to survive
1139
+ // in resume state. The journal already holds the upheld review's bytes; no reset of attempt or
1140
+ // channel state can take them away, on any path, including `resume --retry-failed`.
1141
+ const upheldFeedback = upheldFeedbackByTask(journal.read()).get(t.id) ?? rs?.upheldFeedback;
1142
+ let feedback = upheldFeedback
1143
+ ? `The operator UPHELD the reviewer's findings — address them without discarding landed work.\nreview: ${upheldFeedback}`
504
1144
  : "";
505
1145
  let ladderIdx = 0;
506
1146
  let modeFallbackNoted = false; // v1.2: journal the interactive→print fallback once per task, not per attempt
@@ -509,6 +1149,12 @@ export async function runDaemon(repoRoot, opts = {}) {
509
1149
  let tokens; // SPEND-02: accumulated across attempts — parked spend is still spend
510
1150
  let metered = 0; // SPEND-02: attempts that returned a usage record; distinguishes unmetered from measured-zero
511
1151
  let tipMoves = 0; // OBS-15: one re-gate allowance per task, never reset by a worker retry
1152
+ // T4 (OBS-265): a task whose test gate has failed once stops selecting tests for good — the
1153
+ // selector already proved it cannot speak for this diff, so every later round runs full. SEEDED
1154
+ // FROM THE JOURNAL, not from process memory: a park, a `resume`, or a daemon restart must not
1155
+ // hand the selector a clean slate it did not earn, and the journal is the only state that
1156
+ // survives all three (same read as upheldFeedbackByTask above).
1157
+ let testGateFailed = journal.read().some((e) => e.event === "gate-result" && e.taskId === t.id && e.data.gate === "test" && e.data.pass === false);
512
1158
  let retryMode = "fresh";
513
1159
  let lastContextTokens; // v1.23 reset signal, including stalled/quota attempts
514
1160
  // v1.29: only a gate-failed attempt can seed same-session retry. The next attempt consumes this
@@ -568,7 +1214,14 @@ export async function runDaemon(repoRoot, opts = {}) {
568
1214
  });
569
1215
  await driver.notify(`tickmarkr ${runId}: ${t.id} consult verdict: ${v.action}`, { tier: "attention" });
570
1216
  if (v.action === "retry") {
571
- feedback = renderRetryGuidance(v) || feedback;
1217
+ // The guidance is ADDED to the brief, never swapped for it: the failure bytes the journal
1218
+ // already holds are the one thing the next attempt cannot rediscover for free. The
1219
+ // fingerprint cap's ban on an identical retry is NOT enforced here — a verdict is only one
1220
+ // of several ways this task reaches a re-dispatch, so the ban is enforced at the dispatch
1221
+ // seam every one of them passes through (see enforceRetryBan).
1222
+ const guidance = renderRetryGuidance(v);
1223
+ if (guidance && !feedback.includes(guidance))
1224
+ feedback = feedback ? `${feedback}\n\n${guidance}` : guidance;
572
1225
  return true;
573
1226
  }
574
1227
  if (v.action === "reroute") {
@@ -657,7 +1310,36 @@ export async function runDaemon(repoRoot, opts = {}) {
657
1310
  };
658
1311
  const gateAuthor = rs?.lastAssignment ?? assignment;
659
1312
  const satisfiedIndex = GATE_NAMES.indexOf(satisfiedGate);
660
- const remainingGates = t.gates.filter((gate) => GATE_NAMES.indexOf(gate) > satisfiedIndex);
1313
+ // The serial pipeline could have at most one blocking result, so "everything after the
1314
+ // approved gate" was enough. v1.85 can record both verdict siblings red in one round, and a
1315
+ // selected test screen can be green without a complete suite. Approval waives exactly its
1316
+ // named gate: every other red from that round is re-run, and test is forced unless the prior
1317
+ // journal proves a full suite completed.
1318
+ const priorEvents = journal.read();
1319
+ let priorRoundStart = -1;
1320
+ for (let i = priorEvents.length - 1; i >= 0; i--) {
1321
+ const e = priorEvents[i];
1322
+ if (e.event === "phase-start" && e.taskId === t.id && e.data.phase === "gates") {
1323
+ priorRoundStart = i;
1324
+ break;
1325
+ }
1326
+ }
1327
+ const priorResults = new Map();
1328
+ for (const e of priorEvents.slice(priorRoundStart + 1)) {
1329
+ if (e.event !== "gate-result" || e.taskId !== t.id || typeof e.data.gate !== "string"
1330
+ || !GATE_NAMES.includes(e.data.gate))
1331
+ continue;
1332
+ priorResults.set(e.data.gate, e);
1333
+ }
1334
+ const remainingGates = t.gates.filter((gate) => {
1335
+ if (gate === satisfiedGate)
1336
+ return false;
1337
+ const prior = priorResults.get(gate)?.data;
1338
+ const followsApproved = GATE_NAMES.indexOf(gate) > satisfiedIndex;
1339
+ const otherRed = prior?.pass === false;
1340
+ const needsFullSuite = gate === "test" && prior?.fullSuite !== true;
1341
+ return followsApproved || otherRed || needsFullSuite;
1342
+ });
661
1343
  const resumedTask = { ...t, gates: remainingGates };
662
1344
  gateLoop: while (true) {
663
1345
  const gated = await gitHead(wt);
@@ -665,6 +1347,8 @@ export async function runDaemon(repoRoot, opts = {}) {
665
1347
  const { results } = await runGates(resumedTask, {
666
1348
  worktree: wt, baseRef: taskBase, result: priorResult, author: gateAuthor,
667
1349
  commands, baseline, channels, adapters, cfg, artifactDir: journal.dir,
1350
+ // a recheck re-verifies a human's release: it never selects tests down, it runs the suite.
1351
+ pipeline: "v185",
668
1352
  via: cfg.visibility.llm === "pane"
669
1353
  ? {
670
1354
  driver: trackedDriver,
@@ -677,25 +1361,25 @@ export async function runDaemon(repoRoot, opts = {}) {
677
1361
  excludeReviewers: badReviewers,
678
1362
  onGate: async (e) => {
679
1363
  if (e.phase === "start") {
680
- journal.phaseStart(t.id, phaseForGate(e.gate), { gate: e.gate, index: e.index, total: e.total });
1364
+ notePhaseStart(e);
1365
+ journal.phaseStart(t.id, phaseForGate(e.gate), { gate: e.gate, index: e.index, total: e.total, ...(e.parentAt === undefined ? {} : { parallel: true }) });
681
1366
  return;
682
1367
  }
683
1368
  const g = e.result;
684
- journal.append("gate-result", t.id, {
685
- gate: g.gate, pass: g.pass, details: g.details,
686
- ...(g.meta?.skipped === true ? { skipped: true } : {}),
1369
+ inParallelOrder(g.gate, () => {
1370
+ journalGateResult(g);
1371
+ noteReviewRetry(g);
1372
+ if (g.gate === "review" && !g.pass && /unparseable/.test(g.details)
1373
+ && typeof g.meta?.reviewer === "string") {
1374
+ badReviewers.push(g.meta.reviewer);
1375
+ }
687
1376
  });
688
- noteReviewRetry(g);
689
- if (g.gate === "review" && !g.pass && /unparseable/.test(g.details)
690
- && typeof g.meta?.reviewer === "string") {
691
- badReviewers.push(g.meta.reviewer);
692
- }
693
1377
  },
694
1378
  });
695
1379
  const approvedCommits = await commitsAheadOf(taskBase, wt);
696
1380
  graph = addEvidence(graph, t.id, { commits: approvedCommits, gateResults: results });
697
1381
  saveGraph(repoRoot, graph);
698
- if (!results.every((g) => g.pass)) {
1382
+ if (!results.every(gateSatisfied)) {
699
1383
  gateFails++;
700
1384
  await park(t, "post-approval gate failed", "gate-fail", gateAuthor, rs?.attempts ?? 0, startMs, gateFails, consults, tokens, metered, retryMode);
701
1385
  return;
@@ -772,15 +1456,58 @@ export async function runDaemon(repoRoot, opts = {}) {
772
1456
  // over-threshold context still forces fresh, and the escalation ladder bounds the chain.
773
1457
  const priorSession = retrySession;
774
1458
  retrySession = undefined;
1459
+ // v1.85 T3: the fingerprint cap banned an IDENTICAL retry — this task+gate already produced the
1460
+ // same failure bytes twice, so re-running the same channel on the same brief is a paid
1461
+ // re-measurement of an answer the journal already holds. Enforced HERE, at the one seam every
1462
+ // re-dispatch passes through, and NOT at the consult verdict that set it: a terminal verdict
1463
+ // falling through to the cap's own `retry` rung, a review-fix round and the ladder itself all
1464
+ // reach a dispatch without ever consulting again, and worker-launch below would then expire a
1465
+ // ban nothing had honoured. Bound to the channel the cap fired on, so a move that already went
1466
+ // elsewhere is not refused for a ban that was never about it; with no untried channel left the
1467
+ // task parks naming the ban rather than buying a third round.
1468
+ const banned = activeRetryBan(journal.read(), t.id, channelKey(assignment));
1469
+ if (banned) {
1470
+ const next = failover("escalate");
1471
+ journal.append("retry-same-banned", t.id, {
1472
+ gate: banned, from: channelKey(assignment), to: next ? channelKey(next) : null,
1473
+ });
1474
+ if (!next) {
1475
+ await park(t, `identical ${banned} failure twice this engagement — an identical retry is banned and no untried channel is left`, "gate-fail", assignment, attempt, startMs, gateFails, consults, tokens, metered, retryMode);
1476
+ return;
1477
+ }
1478
+ assignment = next;
1479
+ if (!tried.includes(channelKey(next)))
1480
+ tried.push(channelKey(next));
1481
+ }
775
1482
  const retryAdapter = adapters.find((a) => a.id === assignment.adapter);
776
- retryMode = priorSession
777
- && priorSession.channel === channelKey(assignment)
778
- && (priorSession.contextTokens !== undefined
779
- ? priorSession.contextTokens < cfg.contextWarnTokens
780
- : retryAdapter?.resumeUnknownContext === true)
781
- && retryAdapter?.resumeCommand
782
- ? "resume"
783
- : "fresh";
1483
+ // v1.85 T3: a repair attempt is dispatched FRESH by construction — the whole point is that the
1484
+ // brief, not a surviving session, carries the findings and the diff. It therefore outranks the
1485
+ // session-resume choice, and the mode is journaled so the ledger can price repairs against
1486
+ // fresh re-dispatches. Read from the JOURNAL, before this dispatch's own event lands: a run that
1487
+ // stopped between funding the repair and sending it resumes still carrying the findings.
1488
+ const journaledSoFar = journal.read(); // read BEFORE this dispatch's own event lands
1489
+ let repairFindings = pendingRepairFindings(journaledSoFar, t.id);
1490
+ // OBS-254, one layer below the upheld brief: the ordinary gate-fail brief was loop-local, so any
1491
+ // path that rebuilt this task's state (a resume, `--retry-failed`, a fresh daemon) dispatched a
1492
+ // retry that had forgotten why it was retrying. Re-derived from the journal here, at prompt-build
1493
+ // time, and MERGED rather than substituted — a retry never discards what the journal already
1494
+ // holds. Row-wise, because the live brief may already quote one of them (a delivery-readiness
1495
+ // failure the loop just wrote) and repeating it helps no worker.
1496
+ const journaledRows = journaledFailureBrief(journaledSoFar, t.id).filter((row) => !feedback.includes(row));
1497
+ if (journaledRows.length > 0) {
1498
+ const brief = journaledRows.join("\n\n");
1499
+ feedback = feedback ? `${brief}\n\n${feedback}` : brief;
1500
+ }
1501
+ retryMode = repairFindings
1502
+ ? "repair"
1503
+ : priorSession
1504
+ && priorSession.channel === channelKey(assignment)
1505
+ && (priorSession.contextTokens !== undefined
1506
+ ? priorSession.contextTokens < cfg.contextWarnTokens
1507
+ : retryAdapter?.resumeUnknownContext === true)
1508
+ && retryAdapter?.resumeCommand
1509
+ ? "resume"
1510
+ : "fresh";
784
1511
  // v1.23 T3: over-threshold context still forces fresh at the retry boundary; never interrupt a
785
1512
  // running attempt. Unknown/below emits no reset event.
786
1513
  if (attempt > 0 && lastContextTokens !== undefined && lastContextTokens >= cfg.contextWarnTokens) {
@@ -808,6 +1535,15 @@ export async function runDaemon(repoRoot, opts = {}) {
808
1535
  carriedCommits = await cherryPickCommits(wt, commitsToCarry);
809
1536
  journal.append("worktree-recreation", t.id, { attempted: commitsToCarry, carried: carriedCommits });
810
1537
  }
1538
+ // T2 review (material): harvest eligibility is "does this WORKTREE carry unverified work",
1539
+ // measured against taskBase — the same base the fast-kill's delta probe and the gates
1540
+ // themselves use. It was measured against this attempt's post-carry HEAD, which excluded
1541
+ // every commit cherry-picked forward: attempt 0 commits and walls, attempt 1 receives that
1542
+ // commit and goes silent, and the silent retry — whose worktree already held the whole
1543
+ // deliverable — was NOT harvested, took the stall consult, and could be redispatched to
1544
+ // re-produce it. The routing branches that "starting HEAD" was protecting no longer need it:
1545
+ // quota, dead-channel and provider-death all classify the PRE-HARVEST outcome below and fire
1546
+ // BEFORE the synthesis, so carried-only work reaches gates without bypassing any failover.
811
1547
  const priorNamed = [...new Set([...commitsToCarry, ...carriedCommits])];
812
1548
  const presentCommits = new Set(carriedCommits);
813
1549
  for (const h of commitsToCarry) {
@@ -836,6 +1572,36 @@ export async function runDaemon(repoRoot, opts = {}) {
836
1572
  });
837
1573
  await driver.notify(`tickmarkr ${runId}: ${t.id} lost ${lostCommits.length} of ${commitsToCarry.length} landed commit(s) recreating its worktree — it will re-do that work`, { tier: "attention" });
838
1574
  }
1575
+ // v1.85 T3: "fully carried commits" is a repair PRECONDITION, and it is re-validated here —
1576
+ // the eligibility test ran one attempt ago, but the carry that decides it happens above, on
1577
+ // this dispatch. A tree that lost part of the prior attempt's work cannot be repaired: a
1578
+ // fix-only contract ("do NOT re-implement that work") over a diff whose implementation is
1579
+ // missing would have the worker patch an incomplete tree and forbid it from rebuilding the
1580
+ // rest. The fresh ladder owns this dispatch instead. `retryMode` is corrected before
1581
+ // worker-launch records it, so the ledger's launch event names what the worker actually got.
1582
+ if (repairFindings !== undefined && lostCommits.length > 0) {
1583
+ journal.append("repair-cancelled", t.id, { reason: "carry incomplete", attempted: commitsToCarry, lost: lostCommits });
1584
+ repairFindings = undefined;
1585
+ retryMode = "fresh";
1586
+ }
1587
+ // v1.85 T3: a repair dispatch replaces the bare gate-fail brief with a fix-only contract that
1588
+ // carries the failing findings VERBATIM and the diff CONTENT of the work already in this
1589
+ // worktree. The measured loss it removes: 62 of 68 re-dispatches were fresh, each re-buying
1590
+ // ~20m of onboarding to rediscover a diff and a finding the journal already held.
1591
+ // The diff is measured HERE, from the worktree the worker will actually open, after the carry —
1592
+ // never from the pre-recreation tree, so what the brief quotes is what the worker has.
1593
+ if (repairFindings) {
1594
+ const raw = await shGit(`git diff ${shq(taskBase)}..HEAD`, wt);
1595
+ const cap = cfg.gates.diffCap ?? DEFAULT_DIFF_CAP; // same fallback the measuring gates use
1596
+ const diff = raw.stdout.length > cap
1597
+ ? `${raw.stdout.slice(0, cap)}\n… diff truncated at gates.diffCap (${cap} bytes)`
1598
+ : raw.stdout;
1599
+ const brief = repairBrief(repairFindings, diff, taskBase);
1600
+ // anything the live brief holds beyond the journaled findings (a consult's guidance) is kept:
1601
+ // a repair adds the diff and the fix-only contract, it never subtracts what was already known.
1602
+ feedback = feedback && !brief.includes(feedback) ? `${brief}\n\n${feedback}` : brief;
1603
+ journal.append("repair-dispatch", t.id, { diffBytes: diff.length, capped: raw.stdout.length > cap });
1604
+ }
839
1605
  if (feedback || priorNamed.length > 0) {
840
1606
  feedback = augmentRetryBrief(feedback, { attempted: commitsToCarry, carried: carriedCommits, present: presentCommits });
841
1607
  }
@@ -892,6 +1658,90 @@ export async function runDaemon(repoRoot, opts = {}) {
892
1658
  workerCmd,
893
1659
  exitMarkerCmd,
894
1660
  ].join("\n"));
1661
+ // T2 (OBS-264): the liveness triad, shared by BOTH wait loops (a headless worker stalls on
1662
+ // finished work exactly as a visible one does — and rode the whole window before this). It
1663
+ // sits ABOVE the fast-kill and the nudge because the population it governs is the opposite
1664
+ // one: the kill condemns a pane holding NOTHING, while every one of the 18 observed stalls
1665
+ // held 2-33 commits that the redispatch then re-bought. Concluding is not killing — the
1666
+ // post-loop tail harvests this attempt exactly as a window expiry does, and the carried
1667
+ // worktree goes to gates. The probe runs only once the tracker is ALREADY silent, so a
1668
+ // working worker never pays for it, and an unreadable snapshot RESETS the observation:
1669
+ // unmeasurable CPU is never evidence a worker stopped.
1670
+ // v1.85 T3: the prompt has now actually reached a worker. The journal-derived retry decisions (a
1671
+ // funded repair's findings, an identical-retry ban) expire HERE and nowhere earlier: everything
1672
+ // between task-dispatch and this line — worktree recreation, setup, prompt write, slot allocation,
1673
+ // the launch itself — can still die with no worker having seen the brief, and `--retry-failed`
1674
+ // must then re-send that same brief rather than a fresh prompt on a possibly banned channel.
1675
+ const noteLaunched = () => journal.append("worker-launch", t.id, { attempt, retryMode });
1676
+ let cpuFlat;
1677
+ let cpuAccountant;
1678
+ let cpuGapCount = 0;
1679
+ let unmeasurableNoted = false;
1680
+ // One line per attempt, whichever way the CPU leg turns out to be unmeasurable. A triad that
1681
+ // can never conclude is this feature silently ABSENT — on a host whose `ps` the probe cannot
1682
+ // read, every stall would ride its whole window out again with nothing saying why. Named once
1683
+ // per attempt, not per slice: the condition is structural, and a per-slice line would bury it.
1684
+ const noteUnmeasurable = (reason) => {
1685
+ if (unmeasurableNoted)
1686
+ return;
1687
+ unmeasurableNoted = true;
1688
+ journal.append("worker-harvest-unmeasurable", t.id, { slot: slot.name, attempt, reason });
1689
+ };
1690
+ const harvestConcludes = async (silentMs) => {
1691
+ if (silentMs < harvestSilentMs) {
1692
+ await cpuAccountant?.stop();
1693
+ cpuAccountant = undefined;
1694
+ cpuFlat = undefined;
1695
+ cpuGapCount = 0;
1696
+ return false;
1697
+ }
1698
+ // The CPU leg needs a marker in the worker's own argv, and every launch path puts this
1699
+ // attempt's dispatch script there EXCEPT interactiveSeed: runInteractiveSeed launches the
1700
+ // TUI directly, by a command the ADAPTER owns (seed.launch(model)) which tickmarkr cannot
1701
+ // make attempt-unique and must deliver verbatim. A marker that matches nothing reads as
1702
+ // zero CPU — precisely the false "flat" that would harvest a worker mid-turn — so a seeded
1703
+ // attempt has no measurable CPU leg and the triad never concludes it. The other half of
1704
+ // OBS-264 is untouched there: when its window does expire with commits on the worktree, the
1705
+ // no-trailer tail gates them instead of buying a fresh worker to re-produce them.
1706
+ if (hasSeed) {
1707
+ noteUnmeasurable("interactive-seed launch is not in the probed process tree");
1708
+ return false;
1709
+ }
1710
+ if (cpuAccountant === undefined) {
1711
+ cpuAccountant = new WorkerTreeCpuAccountant(dispatchScript, wt);
1712
+ await cpuAccountant.start();
1713
+ }
1714
+ const observation = cpuAccountant.read();
1715
+ if (observation.gaps !== cpuGapCount) {
1716
+ cpuGapCount = observation.gaps;
1717
+ cpuFlat = undefined;
1718
+ noteUnmeasurable("one or more worker process snapshots could not be read");
1719
+ }
1720
+ const cpu = observation.cpu;
1721
+ if (cpu === undefined) {
1722
+ // Unmeasurable CPU is never evidence a worker stopped: RESET the observation rather than
1723
+ // conclude on it, and name the gap — a probe whose snapshot never parses is the same
1724
+ // structural hole as the seeded launch, and must not be the one that stays silent.
1725
+ cpuFlat = undefined;
1726
+ noteUnmeasurable("the worker process snapshot could not be read");
1727
+ return false;
1728
+ }
1729
+ const now = Date.now();
1730
+ if (cpu.ms !== cpuFlat?.ms) {
1731
+ cpuFlat = { ms: cpu.ms, since: now };
1732
+ return false;
1733
+ }
1734
+ if (now - cpuFlat.since < harvestCpuFlatWindowMs(cpu.resolutionMs))
1735
+ return false;
1736
+ const carried = await commitsAheadOf(taskBase, wt);
1737
+ if (carried.length === 0)
1738
+ return false; // nothing landed: not this branch's population
1739
+ journal.append("worker-harvest", t.id, {
1740
+ slot: slot.name, attempt, commits: carried.length,
1741
+ silentMs, cpuMs: cpu.ms, cpuResolutionMs: cpu.resolutionMs,
1742
+ });
1743
+ return true;
1744
+ };
895
1745
  // SPEND-01: this attempt's dispatch wall-clock — the usage collect cursor. Captured once here, the
896
1746
  // single site, so a test can reason about it; keep Date.now() out of profile.ts (still pure) and
897
1747
  // out of adapter module scope (the cursor is a parameter, threaded from the daemon).
@@ -931,6 +1781,10 @@ export async function runDaemon(repoRoot, opts = {}) {
931
1781
  let output;
932
1782
  let exitCode;
933
1783
  let timedOut = false;
1784
+ // T2 review: print mode's "the exit marker appeared". Kept apart from `finished` (the
1785
+ // trailer) but still needed by the keepPanes decision below, whose contract is about a
1786
+ // subprocess tree that REACHED its exit marker, not about what the worker claimed.
1787
+ let processExited = false;
934
1788
  let earlyLaunchDead = false;
935
1789
  let settleParsed;
936
1790
  let seedResult;
@@ -966,26 +1820,369 @@ export async function runDaemon(repoRoot, opts = {}) {
966
1820
  await park(t, "escalation ladder exhausted", "ladder-exhausted", assignment, attempt + 1, startMs, gateFails, consults, tokens, metered, retryMode);
967
1821
  return false;
968
1822
  };
969
- if (interactive) {
970
- // v1.2 interactive: the TUI doesn't exit on completion — the trailer is the finish line.
971
- // The exit wrapper still fires if the TUI dies (crash/quit): fast-fail instead of burning the timeout.
972
- finished = false;
973
- exitCode = null;
974
- if (adapter.interactiveSeed) {
975
- // v1.69 T6: launch the real TUI without a prompt, wait for readiness, inject one seed turn,
976
- // then fall through to the normal trailer harvest. A failed seed is recorded as a finished
977
- // failure rather than allowed to race the trailer wait.
978
- try {
979
- seedResult = await runInteractiveSeed({ driver, slot, adapter, assignment, promptFile, taskTimeoutMinutes });
1823
+ try {
1824
+ if (interactive) {
1825
+ // v1.2 interactive: the TUI doesn't exit on completion — the trailer is the finish line.
1826
+ // The exit wrapper still fires if the TUI dies (crash/quit): fast-fail instead of burning the timeout.
1827
+ finished = false;
1828
+ exitCode = null;
1829
+ if (adapter.interactiveSeed) {
1830
+ // v1.69 T6: launch the real TUI without a prompt, wait for readiness, inject one seed turn,
1831
+ // then fall through to the normal trailer harvest. A failed seed is recorded as a finished
1832
+ // failure rather than allowed to race the trailer wait.
1833
+ try {
1834
+ seedResult = await runInteractiveSeed({ driver, slot, adapter, assignment, promptFile, taskTimeoutMinutes });
1835
+ }
1836
+ catch (error) {
1837
+ if (!(error instanceof DeliveryReadinessError))
1838
+ throw error;
1839
+ if (await handleDeliveryReadiness(error))
1840
+ continue attempts;
1841
+ return;
1842
+ }
1843
+ noteLaunched();
1844
+ output = seedResult.output;
980
1845
  }
981
- catch (error) {
982
- if (!(error instanceof DeliveryReadinessError))
983
- throw error;
984
- if (await handleDeliveryReadiness(error))
985
- continue attempts;
986
- return;
1846
+ else {
1847
+ try {
1848
+ await driver.run(slot, paneDispatchCommand(dispatchScript));
1849
+ }
1850
+ catch (error) {
1851
+ if (!(error instanceof DeliveryReadinessError))
1852
+ throw error;
1853
+ if (await handleDeliveryReadiness(error))
1854
+ continue attempts;
1855
+ return;
1856
+ }
1857
+ noteLaunched();
1858
+ output = await driver.read(slot, PANE_READ_ROWS);
1859
+ }
1860
+ if (seedResult?.seedFailed) {
1861
+ finished = false;
1862
+ }
1863
+ else {
1864
+ // v1.22 T5 / OBS-19: auto-answer a fingerprint-matched trust dialog exactly once per slot.
1865
+ // Any other blocked/idle dialog pages the operator (unlatched since T1 — see below).
1866
+ let trustAnswered = false;
1867
+ // OBS-201: one liveness nudge per attempt; the grace deadline is its OWN timer, never the
1868
+ // stall window (the nudge's pane echo is absorbed before it starts, or the echo itself
1869
+ // would reset the window and make the early conclusion unreachable).
1870
+ let nudged = false;
1871
+ let nudgeFailed = false; // T1: an undeliverable nudge — the operator, not the daemon, is the actor
1872
+ let nudgeDeadline;
1873
+ // T1: page cadence, NOT a latch — a second page fires on a status change or once
1874
+ // pageRepeatMs elapses, so an operator who missed the first one is paged again.
1875
+ let lastPagedStatus;
1876
+ let lastPagedAt = 0;
1877
+ // T1 review: worker-status is journaled on CHANGE, not per slice — every journal append
1878
+ // is narrated to the run's live surface (cli/commands/run.ts) and feeds activity.ts's
1879
+ // `now:` cell, so a per-slice append wrote a status line per worker per ~30s slice and
1880
+ // pinned `now` to worker-status. On-change keeps post-hoc analysis at a fraction of the
1881
+ // volume while still recording which gate held.
1882
+ let lastStatus;
1883
+ finished = false;
1884
+ exitCode = null;
1885
+ // OBS-54: reaping keys on new pane output, not dispatch wall clock. Poll at least twice per
1886
+ // stall window (and at the existing 30s cadence for normal windows) so an active worker resets it.
1887
+ const stallWindowMs = taskTimeoutMinutes * 60_000;
1888
+ // v1.76: only monotonic work (seed submission, transcript growth, or context growth) resets
1889
+ // the stall clock. Raw pane differences are terminal chrome until proven otherwise.
1890
+ let everHadOutput = output.length > 0;
1891
+ const stallProgress = new StallProgressTracker();
1892
+ stallProgress.observe({ paneText: output, seedSubmitted: true, contextTokens });
1893
+ let lastProgressAt = Date.now();
1894
+ // T1: in-loop detector state — consecutive quota-banner slices. The dead-channel fast-kill
1895
+ // reads the tracker's RAW row-growth clock (lastRowGrowthAt), not the re-arm-suppressed
1896
+ // lastProgressAt — see the kill below.
1897
+ let quotaStreak = 0;
1898
+ let rowSaturationHeld = false; // journaled once per attempt when the kill stands down
1899
+ while (Date.now() - lastProgressAt < stallWindowMs) {
1900
+ const sliceStart = Date.now();
1901
+ const remaining = stallWindowMs - (sliceStart - lastProgressAt);
1902
+ let slice = Math.min(BLOCKED_POLL_MS, Math.max(100, Math.min(stallWindowMs / 2, remaining)));
1903
+ if (!everHadOutput) {
1904
+ const earlyLeft = earlyLaunchLivenessMs - (sliceStart - attemptStart);
1905
+ if (earlyLeft > 0)
1906
+ slice = Math.min(slice, earlyLeft);
1907
+ }
1908
+ // T2 (OBS-264): never sleep PAST the instant the harvest probe becomes eligible, and past
1909
+ // it let the probe own the cadence (HARVEST_POLL_MS) — a 30s trailer slice would
1910
+ // otherwise cost a minute per pair of CPU samples. At the shipped 5m gate this leaves
1911
+ // every slice before the gate exactly as long as it already was.
1912
+ slice = Math.min(slice, harvestSliceMs(sliceStart - lastProgressAt));
1913
+ if (await driver.waitOutput(slot, `(${trailerPattern(nonce)})|TICKMARKR_EXIT_${nonce}:\\d`, slice, { regex: true })) {
1914
+ // verify before accepting: a worker that merely DISPLAYS a marker (e.g. editing tickmarkr's
1915
+ // own source, where "TICKMARKR_EXIT:" is a string literal) must not end the wait. Only a
1916
+ // parseable trailer or a digit-suffixed exit marker in the harvest is completion.
1917
+ output = await driver.read(slot, PANE_READ_ROWS); // TUI transcripts carry chrome — read deeper than print's 500
1918
+ finished = new RegExp(trailerPattern(nonce)).test(output);
1919
+ const exit = exitRe.exec(output);
1920
+ if (finished || exit) {
1921
+ exitCode = exit ? Number(exit[1]) : null; // null ⇔ the TUI is still alive
1922
+ await sampleContext(); // final poll-seam sample before leaving the wait
1923
+ break;
1924
+ }
1925
+ }
1926
+ const paneText = await driver.read(slot, PANE_READ_ROWS);
1927
+ if (paneText.length > 0)
1928
+ everHadOutput = true;
1929
+ // OBS-117 (v1.71 T6): zero raw output by the early-launch deadline is a dead channel now.
1930
+ if (!everHadOutput && Date.now() - attemptStart >= earlyLaunchLivenessMs) {
1931
+ earlyLaunchDead = true;
1932
+ output = paneText;
1933
+ break;
1934
+ }
1935
+ // v1.23 T2: piggyback on this poll slice — same cadence as blocked/idle checks, no new timer.
1936
+ await sampleContext();
1937
+ if (stallProgress.observe({ paneText, contextTokens })) {
1938
+ lastProgressAt = Date.now();
1939
+ // T1 review fix: progress AFTER a delivered nudge means the worker answered it —
1940
+ // disarm the grace deadline. Without this the expiry below fires at the next
1941
+ // quiet patch ≥ the grace (4m) measured from the rolling lastProgressAt, so the
1942
+ // exact population the nudge rescued (workers prone to long silences, e.g. a 6m
1943
+ // test run) was force-concluded as if it had ignored the nudge. The answered
1944
+ // worker returns to the full rolling window, and the consult sees the truth.
1945
+ if (nudged && nudgeDeadline !== undefined) {
1946
+ nudgeDeadline = undefined;
1947
+ journal.append("worker-nudge-answered", t.id, { slot: slot.name, attempt });
1948
+ }
1949
+ }
1950
+ const sliceNow = Date.now();
1951
+ // T1 (OBS-263): quota banners are classified IN-LOOP — two consecutive matching slices
1952
+ // plus >=3m tracker silence — then the post-loop quota failover runs NOW, not after the
1953
+ // window. The match reads the chrome-filtered tail: the bottom of a rendered TUI frame
1954
+ // is fixed composer/welcome chrome (codex pins a "usage limit resets available" line
1955
+ // there — it matched every frame of the wedged-MCP fixture), so the known chrome is
1956
+ // filtered by identity, never by novelty — a banner already on screen at launch
1957
+ // classifies exactly like one printed mid-attempt (T1 review: a novelty baseline
1958
+ // exculpated the launch-throttle case forever).
1959
+ if (QUOTA_RE.test(stallSnapshotBannerRows(paneText)))
1960
+ quotaStreak++;
1961
+ else
1962
+ quotaStreak = 0;
1963
+ if (quotaStreak >= 2 && sliceNow - lastProgressAt >= quotaBannerSilentMs) {
1964
+ // no `output =` here: the post-loop no-trailer tail re-reads the pane anyway, so an
1965
+ // assignment would only split the classification read from the verdict read.
1966
+ journal.append("quota-banner", t.id, { slot: slot.name, attempt, silentMs: sliceNow - lastProgressAt });
1967
+ break;
1968
+ }
1969
+ // T1 (OBS-262): the `paged` latch is deleted — status is sampled EVERY slice (and
1970
+ // journaled on change — see lastStatus above), so post-hoc analysis can see which
1971
+ // gate held. page on "idle" too: herdr's blocked-scrape is strict and proved flaky
1972
+ // for TUI dialogs (live check: cursor's trust dialog scraped as idle).
1973
+ // "unknown"/"working" never page.
1974
+ const st = await driver.status(slot);
1975
+ if (st !== lastStatus) {
1976
+ lastStatus = st;
1977
+ journal.append("worker-status", t.id, { slot: slot.name, status: st, attempt });
1978
+ }
1979
+ if (st === "blocked" || st === "idle") {
1980
+ // T5: once-per-slot auto-answer when the adapter declares a trust dialog and the pane
1981
+ // text matches. tickmarkr created the worktree from the operator's own repo — safe by construction.
1982
+ if (!trustAnswered && adapter.trustDialog && driver.sendKey) {
1983
+ try {
1984
+ const paneText = await driver.read(slot, 80);
1985
+ if (matchesTrustDialog(paneText, adapter.trustDialog)) {
1986
+ trustAnswered = true;
1987
+ // v1.25 T1: audit trail for live runs — prove the dialog appeared and was answered.
1988
+ // Latch + sendKey + no-page continue stay byte-identical; this append is additive only.
1989
+ journal.append("trust-auto-answer", t.id, { slot: slot.name, adapter: adapter.id });
1990
+ await driver.sendKey(slot, adapter.trustDialog.key);
1991
+ const spent = Date.now() - sliceStart;
1992
+ if (spent < slice)
1993
+ await new Promise((r) => setTimeout(r, Math.min(slice - spent, 1_000)));
1994
+ continue; // do not page — keep waiting for the trailer
1995
+ }
1996
+ }
1997
+ catch {
1998
+ /* read/send failed — fall through to page the operator */
1999
+ }
2000
+ }
2001
+ }
2002
+ // T1 (OBS-262): the daemon ACTS on a silent worker before paging anyone. Gate: monotonic
2003
+ // tracker silent ≥ the nudge threshold — the herdr status reading (idle/unknown/working)
2004
+ // no longer holds the gate hostage; only `blocked` stays page-only (nudging a dialog
2005
+ // prompt can't help). One nudge per attempt, then the grace timer owns the conclusion.
2006
+ const nudgeable = st !== "blocked" && !!driver.nudge && NUDGEABLE_ADAPTERS.has(adapter.id);
2007
+ // T1 review (answer-then-die): `nudgeable` is a per-slice property of the adapter and
2008
+ // status — it is NOT "a daemon action is pending". An action is pending only while the
2009
+ // nudge can still fire (un-nudged) or its grace window is armed; once the worker
2010
+ // ANSWERS, the disarm above clears nudgeDeadline while `nudged` stays latched, and the
2011
+ // daemon has nothing left to do — the pane falls back under the fast-kill and page
2012
+ // watchdogs like any other, instead of riding the whole rolling window untended.
2013
+ const nudgePending = nudgeable && (!nudged || nudgeDeadline !== undefined);
2014
+ // T2 (OBS-264): the liveness triad CONCLUDES the wait on finished work — commits ahead
2015
+ // of the task base (this attempt's own AND any carried forward — see the eligibility
2016
+ // comment at the dispatch site), a flat worker-tree CPU delta, and >= harvestSilentMs
2017
+ // of monotonic tracker silence (defined once, above, and run identically by the print loop).
2018
+ // T2 review (material): it carries the SAME nudge hold as the fast-kill below, by the
2019
+ // same clause and for the same reason. The CPU leg cannot tell "idle because finished"
2020
+ // from "idle because holding an unsubmitted turn in its input box" — both read flat CPU
2021
+ // under a silent tracker — and the nudge is the one signal that can. Under the shipped
2022
+ // constants (harvest 5m < nudge 10m) an unheld harvest concluded every committed
2023
+ // claude-code worker before the rescue could fire, leaving T1's nudge dead code for
2024
+ // exactly the committed-and-stalled population OBS-264 is about. Holding concludes at
2025
+ // ~14m (nudge + grace) rather than ~36m — nearly all of the OBS-264 win, and a worker
2026
+ // that only needed a submit answers with a full trailer instead of partial work. An
2027
+ // ANSWERED or twice-undeliverable nudge leaves nothing pending, so the triad governs
2028
+ // again; the hold is on a pending daemon ACTION, never on the adapter being nudgeable.
2029
+ if ((!nudgePending || nudgeFailed) && await harvestConcludes(sliceNow - lastProgressAt))
2030
+ break;
2031
+ // T1 (R1 dead-channel fast-kill): no trailer, no worktree delta, and no output growth
2032
+ // for the fast-kill window — the channel is dead, so conclude NOW
2033
+ // (journaled) and let the existing no-trailer tail classify and route the attempt.
2034
+ // The tracker is the growth signal on purpose: raw pane bytes grow on cosmetic repaint
2035
+ // (an elapsed "9s"→"10s" lengthens the read and would hide a frozen pane). But the
2036
+ // tracker SHARES the read window's ceiling: its row signal saturates once a sample
2037
+ // FILLS a PANE_READ_ROWS read on raw lines (blanks/chrome included — see stall.ts),
2038
+ // and past that point a flat tracker means "unmeasurable", not "dead". For unmetered
2039
+ // adapters (codex, cursor-agent, grok, opencode — no contextUsage, so the token signal
2040
+ // never fires) rows are the ONLY
2041
+ // signal, and those are exactly the non-nudgeable adapters this kill governs — a live
2042
+ // worker past the ceiling would be concluded dead mid-work. So the kill STANDS DOWN on
2043
+ // a saturated row signal (journaled once per attempt); the rolling window still owns
2044
+ // that pane, exactly as pre-T1. Token growth counts as life either way, so a metered
2045
+ // worker thinking through a long tool run survives.
2046
+ // The triad has NO status exemption: a pane that herdr reports as blocked, idle, working
2047
+ // or unknown dies alike once it holds no trailer, no delta and no growth — waiting the
2048
+ // rolling window out on a status reading is exactly the blindness T1 removes. A matched
2049
+ // trust dialog is auto-answered above and `continue`s before ever reaching here.
2050
+ // The NUDGE gets first crack at a nudgeable pane: the fast-kill holds while the daemon
2051
+ // still has an action of its own pending (un-nudged, or inside the grace window) —
2052
+ // under the shipped constants (kill 5m < nudge 10m) a delta-less pane would otherwise
2053
+ // die before the rescue could ever fire. An ANSWERED nudge leaves nothing pending, so
2054
+ // the hold lifts and the triad governs again. Once a nudge has failed to deliver TWICE
2055
+ // (one in-slice retry filters a driver flake — see below), the delta clause drops too:
2056
+ // a delta is past work, and a pane no signal can reach must not buy the rest of the
2057
+ // window with it (a `working` pane can't be paged either, so this is the only bound
2058
+ // that path has).
2059
+ if (stallProgress.rowSignalSaturated && !rowSaturationHeld) {
2060
+ rowSaturationHeld = true;
2061
+ journal.append("worker-dead-held", t.id, { slot: slot.name, attempt, reason: "row-signal-saturated" });
2062
+ }
2063
+ // T1 review fix: the kill's "no output growth" leg clocks off the RAW growth signals,
2064
+ // never lastProgressAt alone — the flat-token rule (stall.ts) deliberately suppresses
2065
+ // the re-arm report on row growth once tokens stick, and contextTokens is sticky across
2066
+ // read misses, so a metered non-nudgeable adapter (pi) streaming rows under a stale
2067
+ // counter presented a frozen lastProgressAt and was killed mid-work. lastRowGrowthAt is
2068
+ // recorded on every high-water advance, suppressed or not; token growth already rides
2069
+ // lastProgressAt. Either one advancing is output growth.
2070
+ const lastOutputGrowthAt = Math.max(stallProgress.lastRowGrowthAt ?? 0, lastProgressAt);
2071
+ if (!stallProgress.rowSignalSaturated
2072
+ && (!nudgePending || nudgeFailed)
2073
+ && sliceNow - lastOutputGrowthAt >= deadChannelFastKillMs
2074
+ && (nudgeFailed || !(await worktreeHasDelta(wt, taskBase)))) {
2075
+ journal.append("worker-dead", t.id, { slot: slot.name, attempt, silentMs: sliceNow - lastOutputGrowthAt });
2076
+ break;
2077
+ }
2078
+ if (nudgeable && !nudged && sliceNow - lastProgressAt >= nudgeAfterSilentMs) {
2079
+ nudged = true;
2080
+ // T1 review: a false return is a driver-delivery outcome (missing pin, readiness
2081
+ // stable-frame timeout, read-back hiccup), not proof of an unreachable channel — so
2082
+ // one failure is a flake class, retried once in-slice after a short settle. Only a
2083
+ // failed RETRY condemns the channel. Both failures happen inside this slice, so the
2084
+ // latch stays immediate and exactly one failure is journaled per attempt.
2085
+ let delivered = await driver.nudge(slot, WORKER_NUDGE_MESSAGE);
2086
+ if (!delivered) {
2087
+ await new Promise((r) => setTimeout(r, NUDGE_REDELIVER_MS));
2088
+ delivered = await driver.nudge(slot, WORKER_NUDGE_MESSAGE);
2089
+ }
2090
+ if (delivered) {
2091
+ // absorb the nudge's own echo BEFORE arming the grace timer — post-nudge progress
2092
+ // is measured against this baseline, not against the echo.
2093
+ const echo = await driver.read(slot, PANE_READ_ROWS);
2094
+ stallProgress.observe({ paneText: echo, contextTokens });
2095
+ nudgeDeadline = Date.now() + workerNudgeGraceMs;
2096
+ journal.append("worker-nudge", t.id, { slot: slot.name, attempt });
2097
+ }
2098
+ else {
2099
+ // T1: an undeliverable nudge is itself a condemnation — the channel is unreachable,
2100
+ // so the fast-kill above stops requiring a clean worktree for THIS pane (see there).
2101
+ // A blocked/idle pane is still paged below; a `working`/`unknown` one cannot be, and
2102
+ // for it the conclusion's own escalation is what notifies. Nothing here waits the
2103
+ // rolling window out.
2104
+ nudgeFailed = true;
2105
+ journal.append("worker-nudge-failed", t.id, { slot: slot.name, attempt });
2106
+ }
2107
+ }
2108
+ if (nudged && nudgeDeadline !== undefined && sliceNow >= nudgeDeadline && sliceNow - lastProgressAt >= workerNudgeGraceMs) {
2109
+ // grace spent, still no post-nudge progress: re-harvest once (the trailer may have
2110
+ // landed between polls), then conclude the wait as a stall NOW — the consult sees the
2111
+ // un-answered nudge instead of the remainder of the window.
2112
+ nudgeDeadline = undefined;
2113
+ output = await driver.read(slot, PANE_READ_ROWS);
2114
+ finished = new RegExp(trailerPattern(nonce)).test(output);
2115
+ const exit = exitRe.exec(output);
2116
+ if (finished || exit) {
2117
+ exitCode = exit ? Number(exit[1]) : null;
2118
+ await sampleContext();
2119
+ break;
2120
+ }
2121
+ journal.append("worker-nudge-expired", t.id, { slot: slot.name, attempt, graceMs: workerNudgeGraceMs });
2122
+ lastProgressAt = Date.now() - stallWindowMs; // the existing harvest/classify tail runs unmodified
2123
+ continue; // conclude via the loop condition — never page over an acted-on nudge
2124
+ }
2125
+ // Unlatched page (T1): the page DECISION fires and is journaled every slice the
2126
+ // operator is the right actor — i.e. the nudge path doesn't own it (non-allowlisted
2127
+ // adapter, no nudge surface, or a blocked dialog) or the nudge was attempted and
2128
+ // FAILED. A nudgeable worker below the silence threshold waits for its nudge; a
2129
+ // DELIVERED nudge's pending grace suppresses the page — the daemon already acted. An
2130
+ // ANSWERED nudge has no action pending (the disarm cleared the deadline), so a pane
2131
+ // that then reads blocked/idle is the operator's again.
2132
+ // Delivery is unlatched too: the operator is notified again on a status change or
2133
+ // once pageRepeatMs elapses, so a missed first page is not the last one.
2134
+ const pageable = (st === "blocked" || st === "idle")
2135
+ && (!nudgePending || nudgeFailed);
2136
+ if (pageable) {
2137
+ journal.append("operator-page", t.id, { slot: slot.name, attempt, status: st });
2138
+ if (st !== lastPagedStatus || sliceNow - lastPagedAt >= pageRepeatMs) {
2139
+ lastPagedStatus = st;
2140
+ lastPagedAt = sliceNow;
2141
+ const why = st === "blocked" ? "is blocked on a prompt — approve in its pane" : "looks idle without finishing — check its pane";
2142
+ await driver.notify(`tickmarkr ${runId}: ${slot.name} ${why}`, { tier: "attention" });
2143
+ }
2144
+ }
2145
+ // a dead pane or a false-positive marker display returns fast — sleep the unspent slice, never hot-spin
2146
+ const spent = Date.now() - sliceStart;
2147
+ if (spent < slice)
2148
+ await new Promise((r) => setTimeout(r, Math.min(slice - spent, 1_000)));
2149
+ }
2150
+ if (!finished && exitCode === null) {
2151
+ // timed out (or only ever saw false positives): harvest whatever the pane holds now
2152
+ timedOut = Date.now() - lastProgressAt >= stallWindowMs;
2153
+ output = await driver.read(slot, PANE_READ_ROWS);
2154
+ finished = new RegExp(trailerPattern(nonce)).test(output);
2155
+ const exit = exitRe.exec(output);
2156
+ exitCode = exit ? Number(exit[1]) : null;
2157
+ }
2158
+ if (finished) {
2159
+ await driver.waitAgentStatus(slot, "idle", 5_000); // settle, then re-harvest the final render
2160
+ output = await driver.read(slot, PANE_READ_ROWS);
2161
+ }
2162
+ // T5 / OBS-111: an interactive harvest can race the TUI's final paint. When the pane
2163
+ // contains the nonce token but the JSON hasn't balanced yet, settle and re-read through
2164
+ // the existing pane-read seam once or twice before recording a malformed-trailer cause.
2165
+ if (interactive) {
2166
+ const stallWindowMs = taskTimeoutMinutes * 60_000;
2167
+ const settleDeadline = attemptStart + stallWindowMs;
2168
+ const settleDelayMs = 1_000;
2169
+ const maxSettleRetries = 2;
2170
+ let settleTries = 0;
2171
+ settleParsed = adapter.parse(output, nonce);
2172
+ while (settleParsed.summary === UNPARSEABLE_TRAILER_SUMMARY && settleTries < maxSettleRetries) {
2173
+ const remaining = settleDeadline - Date.now();
2174
+ if (remaining <= 0)
2175
+ break;
2176
+ await new Promise((r) => setTimeout(r, Math.min(settleDelayMs, remaining)));
2177
+ output = await driver.read(slot, PANE_READ_ROWS);
2178
+ settleParsed = adapter.parse(output, nonce);
2179
+ settleTries++;
2180
+ }
2181
+ if (settleParsed.summary !== UNPARSEABLE_TRAILER_SUMMARY) {
2182
+ finished = settleParsed.summary !== NO_TRAILER_SUMMARY;
2183
+ }
2184
+ }
987
2185
  }
988
- output = seedResult.output;
989
2186
  }
990
2187
  else {
991
2188
  try {
@@ -998,229 +2195,68 @@ export async function runDaemon(repoRoot, opts = {}) {
998
2195
  continue attempts;
999
2196
  return;
1000
2197
  }
1001
- output = await driver.read(slot, 1000);
1002
- }
1003
- if (seedResult?.seedFailed) {
1004
- finished = false;
1005
- }
1006
- else {
1007
- let paged = false;
1008
- // v1.22 T5 / OBS-19: auto-answer a fingerprint-matched trust dialog exactly once per slot.
1009
- // Any other blocked/idle dialog still pages the operator (paged latch below).
1010
- let trustAnswered = false;
1011
- // OBS-201: one liveness nudge per attempt; the grace deadline is its OWN timer, never the
1012
- // stall window (the nudge's pane echo is absorbed before it starts, or the echo itself
1013
- // would reset the window and make the early conclusion unreachable).
1014
- let nudged = false;
1015
- let nudgeDeadline;
1016
- finished = false;
1017
- exitCode = null;
1018
- // OBS-54: reaping keys on new pane output, not dispatch wall clock. Poll at least twice per
1019
- // stall window (and at the existing 30s cadence for normal windows) so an active worker resets it.
2198
+ noteLaunched();
2199
+ // OBS-54: headless workers have the same output-inactivity budget as visible panes.
2200
+ // v1.76: same monotonic-progress measure as the interactive site; harvest stays raw.
1020
2201
  const stallWindowMs = taskTimeoutMinutes * 60_000;
1021
- // v1.76: only monotonic work (seed submission, transcript growth, or context growth) resets
1022
- // the stall clock. Raw pane differences are terminal chrome until proven otherwise.
1023
- let everHadOutput = output.length > 0;
2202
+ const initialPane = await driver.read(slot, 500);
2203
+ let everHadOutput = initialPane.length > 0;
1024
2204
  const stallProgress = new StallProgressTracker();
1025
- stallProgress.observe({ paneText: output, seedSubmitted: true, contextTokens });
2205
+ stallProgress.observe({ paneText: initialPane, seedSubmitted: true, contextTokens });
1026
2206
  let lastProgressAt = Date.now();
2207
+ finished = false;
2208
+ // T2 review (material): the exit marker proves the PROCESS EXITED, never that the worker
2209
+ // emitted a trailer — the two were the same flag here, so a headless worker that committed
2210
+ // and exited cleanly without one entered the tail as finished:true, skipping the harvest
2211
+ // synthesis entirely and reaching gates with the worker's own ok:false and no
2212
+ // worker-result-harvested row. The interactive site has always kept them apart (`finished`
2213
+ // there is the trailer regex; the exit marker only sets exitCode), and the cause taxonomy
2214
+ // already names this shape "clean-exit-no-trailer" — unreachable in print mode until now.
1027
2215
  while (Date.now() - lastProgressAt < stallWindowMs) {
1028
- const sliceStart = Date.now();
1029
- const remaining = stallWindowMs - (sliceStart - lastProgressAt);
2216
+ const remaining = stallWindowMs - (Date.now() - lastProgressAt);
1030
2217
  let slice = Math.min(BLOCKED_POLL_MS, Math.max(100, Math.min(stallWindowMs / 2, remaining)));
1031
2218
  if (!everHadOutput) {
1032
- const earlyLeft = earlyLaunchLivenessMs - (sliceStart - attemptStart);
2219
+ const earlyLeft = earlyLaunchLivenessMs - (Date.now() - attemptStart);
1033
2220
  if (earlyLeft > 0)
1034
2221
  slice = Math.min(slice, earlyLeft);
1035
2222
  }
1036
- if (await driver.waitOutput(slot, `(${trailerPattern(nonce)})|TICKMARKR_EXIT_${nonce}:\\d`, slice, { regex: true })) {
1037
- // verify before accepting: a worker that merely DISPLAYS a marker (e.g. editing tickmarkr's
1038
- // own source, where "TICKMARKR_EXIT:" is a string literal) must not end the wait. Only a
1039
- // parseable trailer or a digit-suffixed exit marker in the harvest is completion.
1040
- output = await driver.read(slot, 1000); // TUI transcripts carry chrome — read deeper than print's 500
1041
- finished = new RegExp(trailerPattern(nonce)).test(output);
1042
- const exit = exitRe.exec(output);
1043
- if (finished || exit) {
1044
- exitCode = exit ? Number(exit[1]) : null; // null ⇔ the TUI is still alive
1045
- await sampleContext(); // final poll-seam sample before leaving the wait
1046
- break;
1047
- }
2223
+ // T2 (OBS-264): same probe cadence the interactive loop uses — see harvestSliceMs.
2224
+ slice = Math.min(slice, harvestSliceMs(Date.now() - lastProgressAt));
2225
+ if (await driver.waitOutput(slot, `TICKMARKR_EXIT_${nonce}:\\d`, slice, { regex: true })) {
2226
+ processExited = true;
2227
+ break;
1048
2228
  }
1049
- const paneText = await driver.read(slot, 1000);
2229
+ const paneText = await driver.read(slot, 500);
1050
2230
  if (paneText.length > 0)
1051
2231
  everHadOutput = true;
1052
- // OBS-117 (v1.71 T6): zero raw output by the early-launch deadline is a dead channel now.
1053
2232
  if (!everHadOutput && Date.now() - attemptStart >= earlyLaunchLivenessMs) {
1054
2233
  earlyLaunchDead = true;
1055
- output = paneText;
1056
2234
  break;
1057
2235
  }
1058
- // v1.23 T2: piggyback on this poll slice — same cadence as blocked/idle checks, no new timer.
1059
2236
  await sampleContext();
1060
2237
  if (stallProgress.observe({ paneText, contextTokens }))
1061
2238
  lastProgressAt = Date.now();
1062
- // page on "idle" too: herdr's blocked-scrape is strict and proved flaky for TUI dialogs
1063
- // (live check: cursor's trust dialog scraped as idle). "unknown" never pages — that's just
1064
- // a pane the scraper can't read (subprocess, dead pane); the task timeout covers those.
1065
- const st = paged ? "" : await driver.status(slot);
1066
- if (!paged && (st === "blocked" || st === "idle")) {
1067
- // T5: once-per-slot auto-answer when the adapter declares a trust dialog and the pane
1068
- // text matches. tickmarkr created the worktree from the operator's own repo — safe by construction.
1069
- if (!trustAnswered && adapter.trustDialog && driver.sendKey) {
1070
- try {
1071
- const paneText = await driver.read(slot, 80);
1072
- if (matchesTrustDialog(paneText, adapter.trustDialog)) {
1073
- trustAnswered = true;
1074
- // v1.25 T1: audit trail for live runs — prove the dialog appeared and was answered.
1075
- // Latch + sendKey + no-page continue stay byte-identical; this append is additive only.
1076
- journal.append("trust-auto-answer", t.id, { slot: slot.name, adapter: adapter.id });
1077
- await driver.sendKey(slot, adapter.trustDialog.key);
1078
- const spent = Date.now() - sliceStart;
1079
- if (spent < slice)
1080
- await new Promise((r) => setTimeout(r, Math.min(slice - spent, 1_000)));
1081
- continue; // do not page — keep waiting for the trailer
1082
- }
1083
- }
1084
- catch {
1085
- /* read/send failed — fall through to page the operator */
1086
- }
1087
- }
1088
- // OBS-201: the daemon ACTS on an idle pane before paging anyone. Gate: idle AND the
1089
- // monotonic tracker silent ≥ the nudge threshold (a worker inside a tool run reads as
1090
- // `working` and never gets here; a briefly-settling TUI hasn't been silent long enough).
1091
- const nudgeable = st === "idle" && !!driver.nudge && NUDGEABLE_ADAPTERS.has(adapter.id);
1092
- if (nudgeable && !nudged) {
1093
- if (Date.now() - lastProgressAt >= nudgeAfterSilentMs) {
1094
- nudged = true;
1095
- if (await driver.nudge(slot, WORKER_NUDGE_MESSAGE)) {
1096
- // absorb the nudge's own echo BEFORE arming the grace timer — post-nudge progress
1097
- // is measured against this baseline, not against the echo.
1098
- const echo = await driver.read(slot, 1000);
1099
- stallProgress.observe({ paneText: echo, contextTokens });
1100
- nudgeDeadline = Date.now() + workerNudgeGraceMs;
1101
- journal.append("worker-nudge", t.id, { slot: slot.name, attempt });
1102
- }
1103
- else {
1104
- journal.append("worker-nudge-failed", t.id, { slot: slot.name, attempt });
1105
- // nudge undeliverable — fall back to today's behavior: page once, window backstop
1106
- paged = true;
1107
- await driver.notify(`tickmarkr ${runId}: ${slot.name} looks idle without finishing — check its pane`, { tier: "attention" });
1108
- }
1109
- }
1110
- // idle but not yet silent past the threshold: neither nudge nor page this slice
1111
- }
1112
- else if (nudgeable && nudged && nudgeDeadline !== undefined) {
1113
- if (Date.now() >= nudgeDeadline && Date.now() - lastProgressAt >= workerNudgeGraceMs) {
1114
- // grace spent, still idle, no post-nudge progress: re-harvest once (the trailer may
1115
- // have landed between polls), then conclude the wait as a stall NOW — the consult
1116
- // sees the un-answered nudge instead of the remainder of the window.
1117
- nudgeDeadline = undefined;
1118
- output = await driver.read(slot, 1000);
1119
- finished = new RegExp(trailerPattern(nonce)).test(output);
1120
- const exit = exitRe.exec(output);
1121
- if (finished || exit) {
1122
- exitCode = exit ? Number(exit[1]) : null;
1123
- await sampleContext();
1124
- break;
1125
- }
1126
- journal.append("worker-nudge-expired", t.id, { slot: slot.name, attempt, graceMs: workerNudgeGraceMs });
1127
- lastProgressAt = Date.now() - stallWindowMs; // the existing harvest/classify tail runs unmodified
1128
- }
1129
- }
1130
- else {
1131
- paged = true; // page once — the visible pane is the operator's to unblock; task timeout is the backstop
1132
- const why = st === "blocked" ? "is blocked on a prompt — approve in its pane" : "looks idle without finishing — check its pane";
1133
- await driver.notify(`tickmarkr ${runId}: ${slot.name} ${why}`, { tier: "attention" });
1134
- }
1135
- }
1136
- // a dead pane or a false-positive marker display returns fast — sleep the unspent slice, never hot-spin
1137
- const spent = Date.now() - sliceStart;
1138
- if (spent < slice)
1139
- await new Promise((r) => setTimeout(r, Math.min(slice - spent, 1_000)));
1140
- }
1141
- if (!finished && exitCode === null) {
1142
- // timed out (or only ever saw false positives): harvest whatever the pane holds now
1143
- timedOut = Date.now() - lastProgressAt >= stallWindowMs;
1144
- output = await driver.read(slot, 1000);
1145
- finished = new RegExp(trailerPattern(nonce)).test(output);
1146
- const exit = exitRe.exec(output);
1147
- exitCode = exit ? Number(exit[1]) : null;
1148
- }
1149
- if (finished) {
1150
- await driver.waitAgentStatus(slot, "idle", 5_000); // settle, then re-harvest the final render
1151
- output = await driver.read(slot, 1000);
1152
- }
1153
- // T5 / OBS-111: an interactive harvest can race the TUI's final paint. When the pane
1154
- // contains the nonce token but the JSON hasn't balanced yet, settle and re-read through
1155
- // the existing pane-read seam once or twice before recording a malformed-trailer cause.
1156
- if (interactive) {
1157
- const stallWindowMs = taskTimeoutMinutes * 60_000;
1158
- const settleDeadline = attemptStart + stallWindowMs;
1159
- const settleDelayMs = 1_000;
1160
- const maxSettleRetries = 2;
1161
- let settleTries = 0;
1162
- settleParsed = adapter.parse(output, nonce);
1163
- while (settleParsed.summary === UNPARSEABLE_TRAILER_SUMMARY && settleTries < maxSettleRetries) {
1164
- const remaining = settleDeadline - Date.now();
1165
- if (remaining <= 0)
1166
- break;
1167
- await new Promise((r) => setTimeout(r, Math.min(settleDelayMs, remaining)));
1168
- output = await driver.read(slot, 1000);
1169
- settleParsed = adapter.parse(output, nonce);
1170
- settleTries++;
1171
- }
1172
- if (settleParsed.summary !== UNPARSEABLE_TRAILER_SUMMARY) {
1173
- finished = settleParsed.summary !== NO_TRAILER_SUMMARY;
1174
- }
2239
+ // T2 (OBS-264): the same liveness triad the interactive loop runs. A headless worker that
2240
+ // committed and went quiet is finished work too, and before this it rode the entire
2241
+ // window out before anything looked at its commits.
2242
+ // No nudge hold here, deliberately: print mode has no nudge surface at all (no pane to
2243
+ // steer, driver.nudge is never consulted on this path), so there is no pending daemon
2244
+ // action for the triad to preempt — the asymmetry with the interactive call site above is
2245
+ // the absence of the thing being held for, not an oversight.
2246
+ if (await harvestConcludes(Date.now() - lastProgressAt))
2247
+ break;
1175
2248
  }
2249
+ output = await driver.read(slot, 500);
2250
+ exitCode = Number(exitRe.exec(output)?.[1] ?? 1);
2251
+ // Completion is the trailer, exactly as in the interactive loop. A process that exited
2252
+ // without one is finished:false with a non-null exitCode — the harvest synthesis then owns
2253
+ // it when the worktree carries work, and classifyWorkerResultCause names it otherwise.
2254
+ finished = new RegExp(trailerPattern(nonce)).test(output);
2255
+ timedOut = !processExited && !finished && Date.now() - lastProgressAt >= stallWindowMs;
1176
2256
  }
1177
2257
  }
1178
- else {
1179
- try {
1180
- await driver.run(slot, paneDispatchCommand(dispatchScript));
1181
- }
1182
- catch (error) {
1183
- if (!(error instanceof DeliveryReadinessError))
1184
- throw error;
1185
- if (await handleDeliveryReadiness(error))
1186
- continue attempts;
1187
- return;
1188
- }
1189
- // OBS-54: headless workers have the same output-inactivity budget as visible panes.
1190
- // v1.76: same monotonic-progress measure as the interactive site; harvest stays raw.
1191
- const stallWindowMs = taskTimeoutMinutes * 60_000;
1192
- const initialPane = await driver.read(slot, 500);
1193
- let everHadOutput = initialPane.length > 0;
1194
- const stallProgress = new StallProgressTracker();
1195
- stallProgress.observe({ paneText: initialPane, seedSubmitted: true, contextTokens });
1196
- let lastProgressAt = Date.now();
1197
- finished = false;
1198
- while (Date.now() - lastProgressAt < stallWindowMs) {
1199
- const remaining = stallWindowMs - (Date.now() - lastProgressAt);
1200
- let slice = Math.min(BLOCKED_POLL_MS, Math.max(100, Math.min(stallWindowMs / 2, remaining)));
1201
- if (!everHadOutput) {
1202
- const earlyLeft = earlyLaunchLivenessMs - (Date.now() - attemptStart);
1203
- if (earlyLeft > 0)
1204
- slice = Math.min(slice, earlyLeft);
1205
- }
1206
- if (await driver.waitOutput(slot, `TICKMARKR_EXIT_${nonce}:\\d`, slice, { regex: true })) {
1207
- finished = true;
1208
- break;
1209
- }
1210
- const paneText = await driver.read(slot, 500);
1211
- if (paneText.length > 0)
1212
- everHadOutput = true;
1213
- if (!everHadOutput && Date.now() - attemptStart >= earlyLaunchLivenessMs) {
1214
- earlyLaunchDead = true;
1215
- break;
1216
- }
1217
- await sampleContext();
1218
- if (stallProgress.observe({ paneText, contextTokens }))
1219
- lastProgressAt = Date.now();
1220
- }
1221
- output = await driver.read(slot, 500);
1222
- exitCode = Number(exitRe.exec(output)?.[1] ?? 1);
1223
- timedOut = !finished && Date.now() - lastProgressAt >= stallWindowMs;
2258
+ finally {
2259
+ await cpuAccountant?.stop();
1224
2260
  }
1225
2261
  // SPEND-01 interactive metering race: the harvest loop breaks on the trailer, but the worker
1226
2262
  // shell may still be running post-trailer bookkeeping (session-store flush, fake usage stamp,
@@ -1232,7 +2268,7 @@ export async function runDaemon(repoRoot, opts = {}) {
1232
2268
  }
1233
2269
  // keepPanes retains visible context, not a timed-out subprocess tree. Close before consult/retry
1234
2270
  // can recreate the worktree; Herdr and subprocesses that reached their exit marker stay unchanged.
1235
- if (keepOpen && (finished || driver.id !== "subprocess"))
2271
+ if (keepOpen && (finished || processExited || driver.id !== "subprocess"))
1236
2272
  keptSlots.push(slot);
1237
2273
  else
1238
2274
  await closeSlot(slot);
@@ -1247,15 +2283,51 @@ export async function runDaemon(repoRoot, opts = {}) {
1247
2283
  tokens = addUsage(tokens, attemptUsage);
1248
2284
  metered++;
1249
2285
  }
1250
- const result = settleParsed ?? adapter.parse(output, nonce);
1251
- const cause = classifyWorkerResultCause({ output, ok: result.ok, finished, exitCode, summary: result.summary, timedOut });
2286
+ let result = settleParsed ?? adapter.parse(output, nonce);
2287
+ const workerFinished = finished;
2288
+ const workerCause = classifyWorkerResultCause({ output, ok: result.ok, finished, exitCode, summary: result.summary, timedOut });
1252
2289
  journal.append("worker-result", t.id, {
1253
- ok: result.ok, summary: result.summary, deviations: result.deviations, finished, exitCode,
1254
- mode: interactive ? "interactive" : "print", ...(cause ? { cause } : {}),
2290
+ ok: result.ok, summary: result.summary, deviations: result.deviations, finished: workerFinished, exitCode,
2291
+ mode: interactive ? "interactive" : "print", ...(workerCause ? { cause: workerCause } : {}),
1255
2292
  });
1256
- if (result.ok && finished)
2293
+ // T2 review (routing precedence): provider-death, quota and dead-channel classification
2294
+ // derive from ONE rule — the worker's OWN outcome, workerFinished and this PRE-HARVEST
2295
+ // parse — never from the synthesized result below. A worker that committed and then walled
2296
+ // (provider outage, quota banner, dead CLI) must still route; the harvest synthesis only
2297
+ // decides whether THIS attempt's worktree goes to gates, and when routing wins instead, the
2298
+ // commits survive via the existing commitsToCarry/cherryPickCommits carry-forward.
2299
+ const preHarvestResult = result;
2300
+ // T2 (OBS-264): recognize committed no-trailer work BEFORE any no-trailer streak, provider,
2301
+ // quota or dead-channel routing. Gates never trusted the trailer, so this successful synthesis
2302
+ // must enter exactly where a worker-claimed ok enters. Preserve the parsed worker truth in the
2303
+ // worker-result row above and name the synthesized gate input in its own harvested event.
2304
+ let harvestedCommits = [];
2305
+ if (!workerFinished) {
2306
+ harvestedCommits = await commitsAheadOf(taskBase, wt);
2307
+ if (harvestedCommits.length > 0) {
2308
+ result = { ok: true, summary: HARVESTED_RESULT_SUMMARY, deviations: [], raw: output };
2309
+ finished = true;
2310
+ journal.append("worker-result-harvested", t.id, {
2311
+ attempt, commits: harvestedCommits, summary: HARVESTED_RESULT_SUMMARY, source: "harvest",
2312
+ });
2313
+ }
2314
+ }
2315
+ // T2 review: provider-death is the THIRD routing branch that must read the pre-harvest
2316
+ // outcome — the synthesis must not null it. A worker that committed and then printed the
2317
+ // outage banner without exiting (workerFinished false, so the harvest fires) still takes
2318
+ // the capped same-channel requeue below; nulling the cause here would skip that branch and
2319
+ // let classifyDeadChannel(preHarvestResult) demote the channel run-wide on a transient blip.
2320
+ const cause = harvestedCommits.length > 0 && workerCause !== "provider-death" ? undefined : workerCause;
2321
+ // T2 review (family): the no-trailer streak is accounted on the SAME ONE rule the routing
2322
+ // branches above use — workerFinished and the PRE-HARVEST parse — never the synthesized
2323
+ // result. A channel that commits but never emits a parseable trailer still burned a
2324
+ // no-trailer window (OBS-57): the synthesis decides whether THIS worktree goes to gates, it
2325
+ // never certifies the channel. Reading the synthesized `finished`/`ok` here reset the streak
2326
+ // on every harvest, so a CLI that produces commits and swallows every trailer was immune to
2327
+ // the two-window demotion and stayed first pick for the rest of the run.
2328
+ if (preHarvestResult.ok && workerFinished)
1257
2329
  noTrailerStreak.set(channelKey(assignment), 0);
1258
- else if (!finished && cause !== "provider-death") {
2330
+ else if (!workerFinished && cause !== "provider-death") {
1259
2331
  const ck = channelKey(assignment);
1260
2332
  const streak = (noTrailerStreak.get(ck) ?? 0) + 1;
1261
2333
  noTrailerStreak.set(ck, streak);
@@ -1275,8 +2347,19 @@ export async function runDaemon(repoRoot, opts = {}) {
1275
2347
  }
1276
2348
  // quota exhaustion → failover within floor; does NOT consume the ladder (spec §4)
1277
2349
  // print: guarded on exit code — exit-0 output that merely MENTIONS "rate limit" must not failover
1278
- // interactive: a harvested trailer beats quota mentions; without one, quota text fails over (spec v1.2 §2)
1279
- const quotaHit = (interactive ? !finished : exitCode !== 0) && QUOTA_RE.test(output);
2350
+ // interactive: a worker-CLAIMED trailer beats quota mentions; without one, quota text fails over
2351
+ // (spec v1.2 §2) — matched on the chrome-filtered tail, the exact discrimination the in-loop
2352
+ // classifier makes, so the two can never disagree. A TUI harvest is the whole retained pane:
2353
+ // an unscoped match failed a worker over for quoting "rate limit" in its own diff (tail
2354
+ // scoping kills that — the mention sits ABOVE the tail), and a raw-tail match fires on fixed
2355
+ // chrome (codex's welcome line — filtered by identity, so a launch-time banner this backstop
2356
+ // exists to catch is never exculpated). Print output keeps the exit-code guard.
2357
+ // T2 review: the gate is `workerFinished`, not the harvest-synthesized `finished` — a
2358
+ // committed-but-quota-walled attempt routes here FIRST (its commits ride the carry-forward
2359
+ // into the next attempt's recreated worktree), it never buys a gate run on throttled work.
2360
+ const quotaHit = interactive
2361
+ ? !workerFinished && QUOTA_RE.test(stallSnapshotBannerRows(output))
2362
+ : exitCode !== 0 && QUOTA_RE.test(output);
1280
2363
  if (quotaHit) {
1281
2364
  const next = failover("quota-failover");
1282
2365
  journal.append("quota-failover", t.id, { from: channelKey(assignment), to: next ? channelKey(next) : null });
@@ -1316,7 +2399,10 @@ export async function runDaemon(repoRoot, opts = {}) {
1316
2399
  // cap above is spent, so a transient blip still recovers in place.
1317
2400
  // OBS-117 (v1.71 T6): a silent launch failure has no CLI signature to parse — the same
1318
2401
  // setup-required typed dead-channel path a late-harvest "command not found" would take.
1319
- const dead = classifyDeadChannel(result) ?? (earlyLaunchDead ? "setup-required" : undefined);
2402
+ // T2 review: classify the PRE-HARVEST parse — classifyDeadChannel bails on any ok:true
2403
+ // result, so reading the synthesized harvest result would swallow auth-required /
2404
+ // setup-required / provider-outage for every committed-but-walled attempt (in both modes).
2405
+ const dead = classifyDeadChannel(preHarvestResult) ?? (earlyLaunchDead ? "setup-required" : undefined);
1320
2406
  if (dead) {
1321
2407
  const from = channelKey(assignment);
1322
2408
  demotedChannels.add(from); // excluded for later attempts AND later tasks in this run
@@ -1382,33 +2468,36 @@ export async function runDaemon(repoRoot, opts = {}) {
1382
2468
  }
1383
2469
  const onGate = async (e) => {
1384
2470
  if (e.phase === "start") {
1385
- journal.phaseStart(t.id, phaseForGate(e.gate), { gate: e.gate, index: e.index, total: e.total });
2471
+ notePhaseStart(e);
2472
+ journal.phaseStart(t.id, phaseForGate(e.gate), { gate: e.gate, index: e.index, total: e.total, ...(e.parentAt === undefined ? {} : { parallel: true }) });
1386
2473
  return;
1387
2474
  }
1388
2475
  const g = e.result;
1389
- // GATE-09 (ROADMAP SC-4): journal every judge retry as an attributable event — which gate flaked,
1390
- // which channel flaked, which channel retried — so `tickmarkr journal`/report can distinguish "judge
1391
- // flaked, retried" from "worker failed" (run-20260711-185020 P43-03 L70-72 billed a judge flake as
1392
- // a worker attempt; 47-01 fixed WHO retries, this closes the audit-trail half). The condition is
1393
- // META-ONLY (D-03): gate === "acceptance" + typeof-shape guards on meta.judgeRetry — never a
1394
- // details-regex. The v1.1 review regex below is grandfathered, not precedent. Appended BEFORE the
1395
- // gate-result so attribution precedes the verdict in the stream. secondUnparseable is derived from
1396
- // the final result's meta.unparseable (set by run-gates when the retry ALSO flaked — double-garbage).
1397
- if (g.gate === "acceptance" && typeof g.meta?.judgeRetry === "object" && g.meta.judgeRetry !== null) {
1398
- const jr = g.meta.judgeRetry;
1399
- if (typeof jr.flaked === "string" && typeof jr.retried === "string") {
1400
- journal.append("judge-retry", t.id, {
1401
- gate: "acceptance", flaked: jr.flaked, retried: jr.retried,
1402
- ...(g.meta.unparseable === true ? { secondUnparseable: true } : {}),
1403
- });
2476
+ inParallelOrder(g.gate, () => {
2477
+ // GATE-09 (ROADMAP SC-4): journal every judge retry as an attributable event — which gate flaked,
2478
+ // which channel flaked, which channel retried — so `tickmarkr journal`/report can distinguish "judge
2479
+ // flaked, retried" from "worker failed" (run-20260711-185020 P43-03 L70-72 billed a judge flake as
2480
+ // a worker attempt; 47-01 fixed WHO retries, this closes the audit-trail half). The condition is
2481
+ // META-ONLY (D-03): gate === "acceptance" + typeof-shape guards on meta.judgeRetry — never a
2482
+ // details-regex. The v1.1 review regex below is grandfathered, not precedent. Appended BEFORE the
2483
+ // gate-result so attribution precedes the verdict in the stream. secondUnparseable is derived from
2484
+ // the final result's meta.unparseable (set by run-gates when the retry ALSO flaked — double-garbage).
2485
+ if (g.gate === "acceptance" && typeof g.meta?.judgeRetry === "object" && g.meta.judgeRetry !== null) {
2486
+ const jr = g.meta.judgeRetry;
2487
+ if (typeof jr.flaked === "string" && typeof jr.retried === "string") {
2488
+ journal.append("judge-retry", t.id, {
2489
+ gate: "acceptance", flaked: jr.flaked, retried: jr.retried,
2490
+ ...(g.meta.unparseable === true ? { secondUnparseable: true } : {}),
2491
+ });
2492
+ }
1404
2493
  }
1405
- }
1406
- journal.append("gate-result", t.id, { gate: g.gate, pass: g.pass, details: g.details, ...(g.meta?.skipped === true ? { skipped: true } : {}) });
1407
- noteReviewRetry(g);
1408
- // v1.1 failover: never re-ask a reviewer channel that produced garbage for this task
1409
- if (g.gate === "review" && !g.pass && /unparseable/.test(g.details) && typeof g.meta?.reviewer === "string") {
1410
- badReviewers.push(g.meta.reviewer);
1411
- }
2494
+ journalGateResult(g);
2495
+ noteReviewRetry(g);
2496
+ // v1.1 failover: never re-ask a reviewer channel that produced garbage for this task
2497
+ if (g.gate === "review" && !g.pass && /unparseable/.test(g.details) && typeof g.meta?.reviewer === "string") {
2498
+ badReviewers.push(g.meta.reviewer);
2499
+ }
2500
+ });
1412
2501
  };
1413
2502
  let results = [];
1414
2503
  let commits = [];
@@ -1418,6 +2507,7 @@ export async function runDaemon(repoRoot, opts = {}) {
1418
2507
  ({ results, commits } = await runGates(t, {
1419
2508
  worktree: wt, baseRef: taskBase, result, author: assignment,
1420
2509
  commands, baseline, channels, adapters, cfg, artifactDir: journal.dir,
2510
+ pipeline: "v185", selectTests: !testGateFailed,
1421
2511
  via: cfg.visibility.llm === "pane"
1422
2512
  ? {
1423
2513
  driver: trackedDriver,
@@ -1439,7 +2529,9 @@ export async function runDaemon(repoRoot, opts = {}) {
1439
2529
  }));
1440
2530
  graph = addEvidence(graph, t.id, { commits, gateResults: results, artifacts: [promptFile] });
1441
2531
  saveGraph(repoRoot, graph);
1442
- if (results.every((g) => g.pass)) {
2532
+ if (results.some((g) => g.gate === "test" && !g.pass))
2533
+ testGateFailed = true;
2534
+ if (results.every(gateSatisfied)) {
1443
2535
  const m = await mergeSerial(taskBranch, t, gated);
1444
2536
  if (m.tipMoved) {
1445
2537
  journal.append("tip-moved", t.id, m.tipMoved);
@@ -1484,21 +2576,113 @@ export async function runDaemon(repoRoot, opts = {}) {
1484
2576
  // v1.53 T3: prefer the CLI's own session id captured from this attempt's output (kimi's resume
1485
2577
  // trailer) over the harness slot name; absent hook or no capture keeps today's slot-name id.
1486
2578
  retrySession = { channel: channelKey(assignment), id: adapter.sessionIdFrom?.(output) ?? sessionId, contextTokens };
1487
- feedback = results.filter((g) => !g.pass).map((g) => `${g.gate}: ${g.details}`).join("\n\n");
2579
+ feedback = results.filter(gateFailed).map((g) => `${g.gate}: ${g.details}`).join("\n\n");
1488
2580
  // OBS-189/G3 (park-economics patch): a request-changes review is a findings brief, not a worker
1489
2581
  // defect — the fix attempt stays on the same channel with the findings as feedback and consumes
1490
2582
  // no escalation-ladder rung. Bounded by the engagement round cap at the top of this loop.
1491
2583
  // Unparseable verdicts (already retried in-gate, OBS-193) and diff-cap trips (the diff cannot
1492
2584
  // shrink by retrying, OBS-48) fall through to the ladder unchanged. Review runs last, so a
1493
2585
  // failed review with every other gate green is exactly "the work landed, the reviewer objects".
1494
- const reviewFail = results.find((g) => g.gate === "review" && !g.pass);
2586
+ const reviewFail = results.find((g) => g.gate === "review" && gateFailed(g));
1495
2587
  const reviewFixRetry = reviewFail !== undefined
1496
2588
  && reviewFail.meta?.unparseable !== true
1497
2589
  && !isDiffCapPark(reviewFail)
1498
- && results.every((g) => g.pass || g.gate === "review");
1499
- const step = reviewFixRetry ? "retry" : r.ladder[Math.min(ladderIdx++, r.ladder.length - 1)];
1500
- journal.append("escalation", t.id, { step, attempt: attempt + 1, ...(reviewFixRetry ? { reviewFix: true } : {}) });
1501
- await driver.notify(`tickmarkr ${runId}: ${t.id} escalation: ${step}`, { tier: "attention" });
2590
+ && results.every((g) => gateSatisfied(g) || g.gate === "review");
2591
+ const failing = results.filter(gateFailed);
2592
+ // Decided before the cap, because the cap's question is whether the NEXT move would re-buy a
2593
+ // measurement already made — and a funded repair is one of the moves that would.
2594
+ const landed = await commitsAheadOf(taskBase, wt);
2595
+ // T4 (OBS-265): judge and review are now ONE round, so a round can report both failing where the
2596
+ // serial walk returned at the judge and review never ran. Eligibility is scored on the battery
2597
+ // the serial contract would have surfaced — review only speaks for a round nothing else failed —
2598
+ // so removing the waiting does not re-price the ladder. The journal below still names every
2599
+ // failing gate, and `feedback` still carries every one of them to the next attempt.
2600
+ const repairBattery = failing.some((g) => g.gate !== "review") ? failing.filter((g) => g.gate !== "review") : failing;
2601
+ const repairable = narrowRepairBattery(repairBattery) && lostCommits.length === 0 && landed.length > 0;
2602
+ const repairsDrawn = repairsSinceApproval(journal.read(), t.id);
2603
+ const repair = repairable && repairsDrawn < MAX_REPAIRS;
2604
+ // v1.85 T3: the fingerprint cap. Two normalized-identical failures of one DETERMINISTIC gate on
2605
+ // one task (volatile tokens — worktree prefixes, line refs, durations, run ids — are not
2606
+ // information) mean the round about to be bought is a re-measurement: ~663m across 5 runs went to
2607
+ // exactly this loop. The threshold is a property of the FAILURE, never of the move that would
2608
+ // follow it, so it is evaluated on EVERY such gate at EVERY ladder position — including a rung
2609
+ // that would change channel, and including a review-fix round. Conditioning it on the next rung
2610
+ // was the first shape of this and let a second identical failure buy an escalate/consult round
2611
+ // the criterion says it may not buy. What the cap does NOT reach is an LLM verdict, which is a
2612
+ // different object with its own tighter bound (see isDeterministicFailure).
2613
+ //
2614
+ // It fires on the CROSSING, not as a latch: the consult it forces and the ban it sets govern the
2615
+ // next move, so re-firing on the third identical failure would only re-buy the round it just paid
2616
+ // for — and the ladder, whose rung this failure still spends, bounds the rest.
2617
+ const repeated = failing.find((g) => isDeterministicFailure(g)
2618
+ && identicalGateFailures(journal.read(), t.id, g.gate, normalizeGateFailure(g.details)) === GATE_FINGERPRINT_CAP);
2619
+ // The rung the cap spent, when it fired — the move below executes THIS instead of drawing a
2620
+ // second one, so a cap costs exactly the rung the failure would have cost anyway.
2621
+ let capStep;
2622
+ if (repeated) {
2623
+ const normalized = normalizeGateFailure(repeated.details);
2624
+ journal.append("gate-fingerprint-cap", t.id, {
2625
+ gate: repeated.gate,
2626
+ occurrences: GATE_FINGERPRINT_CAP,
2627
+ fingerprint: normalized.slice(0, 500),
2628
+ retrySameBanned: true,
2629
+ channel: channelKey(assignment), // the ban is bound to the channel that produced the repeat
2630
+ attempt: attempt + 1,
2631
+ });
2632
+ await driver.notify(`tickmarkr ${runId}: ${t.id} ${repeated.gate} failed identically twice — consulting, identical retry banned`, { tier: "attention" });
2633
+ // The cap takes the ladder's MOVE, never its accounting: this failure still spends the rung it
2634
+ // would have spent, so a task that cannot converge still reaches ladder exhaustion on exactly
2635
+ // the budget it always had and the cap can never hand a stuck task extra rounds.
2636
+ capStep = r.ladder[Math.min(ladderIdx++, r.ladder.length - 1)];
2637
+ journal.append("escalation", t.id, { step: capStep, attempt: attempt + 1, fingerprintCap: true });
2638
+ await driver.notify(`tickmarkr ${runId}: ${t.id} escalation: ${capStep}`, { tier: "attention" });
2639
+ const v = await runConsult("gate-fail-repeat", output, feedback, results);
2640
+ // Rule: a terminal cap consult vetoes a same-channel retry, but cannot veto an `escalate`
2641
+ // rung that already satisfies retry-same-banned by changing channel. Thus terminal+retry
2642
+ // parks through the shared verdict boundary, while terminal+escalate records the consult as
2643
+ // advisory and executes the already-spent rung below. Recoverable verdicts still control the
2644
+ // move directly. The paired fixture asserts both directions of this boundary.
2645
+ const recoverable = v.action === "retry" || v.action === "reroute";
2646
+ if (recoverable || capStep !== "escalate") {
2647
+ if (await applyVerdict(v, attempt + 1, "gate-fail"))
2648
+ continue;
2649
+ return;
2650
+ }
2651
+ journal.append("consult-verdict", t.id, { action: v.action, notes: v.notes, capAdvisory: true });
2652
+ }
2653
+ // v1.85 T3: a narrow battery over fully carried commits earns a REPAIR (decided above) — the
2654
+ // next dispatch carries the findings verbatim and the diff content instead of re-onboarding a
2655
+ // fresh worker. Budget is engagement-scoped and journal-derived, so a resume inherits it rather
2656
+ // than refunding it; the third repair-eligible failure falls back to the fresh ladder.
2657
+ //
2658
+ // `landed` is measured from the worktree rather than read from runGates: runGates returns at
2659
+ // the first failure, so a red test or lint gate never reaches the evidence stage and its
2660
+ // `commits` come back empty — reading them would make the ruling's test/lint case unreachable.
2661
+ //
2662
+ // Budget spent means the FRESH LADDER owns this failure — including a review-only one, whose
2663
+ // same-channel fix retry is exactly the round the budget just declared too expensive to repeat.
2664
+ const repairExhausted = repairable && !repair;
2665
+ if (repair && !capStep) {
2666
+ journal.append("repair-attempt", t.id, {
2667
+ repair: repairsDrawn + 1, of: MAX_REPAIRS, gates: failing.map((g) => g.gate),
2668
+ commits: landed.length,
2669
+ findings: feedback, // the failure bytes this repair must carry, replayable across a resume
2670
+ });
2671
+ }
2672
+ else if (repairExhausted && !capStep) {
2673
+ journal.append("repair-exhausted", t.id, { repairs: repairsDrawn, of: MAX_REPAIRS, gates: failing.map((g) => g.gate) });
2674
+ }
2675
+ const step = capStep ?? (repair || (reviewFixRetry && !repairExhausted)
2676
+ ? "retry"
2677
+ : r.ladder[Math.min(ladderIdx++, r.ladder.length - 1)]);
2678
+ if (!capStep) { // a capped failure already journaled and announced the rung it spent
2679
+ journal.append("escalation", t.id, {
2680
+ step, attempt: attempt + 1,
2681
+ ...(reviewFixRetry && !repairExhausted ? { reviewFix: true } : {}),
2682
+ ...(repair ? { repair: repairsDrawn + 1 } : {}),
2683
+ });
2684
+ await driver.notify(`tickmarkr ${runId}: ${t.id} escalation: ${step}`, { tier: "attention" });
2685
+ }
1502
2686
  if (step === "retry")
1503
2687
  continue;
1504
2688
  if (step === "escalate") {
@@ -1574,24 +2758,7 @@ export async function runDaemon(repoRoot, opts = {}) {
1574
2758
  // OBS-34: post-merge integration-tip verify — strict exit codes, no baseline forgiveness.
1575
2759
  const lastMergedTask = [...journal.read()].reverse().find((e) => e.event === "merge" && e.taskId)?.taskId;
1576
2760
  if (summary.done.length > 0 && Object.keys(commands).length > 0) {
1577
- const tipResults = await verifyIntegrationTip(intWt, commands, journal.dir);
1578
- let tipFailed = false;
1579
- for (const r of tipResults) {
1580
- if (r.pass) {
1581
- journal.append("tip-verify", undefined, { gate: r.gate, cmd: r.cmd, pass: true, exitCode: r.exitCode, details: r.details });
1582
- }
1583
- else {
1584
- journal.append("tip-verify-failed", undefined, {
1585
- gate: r.gate,
1586
- cmd: r.cmd,
1587
- exitCode: r.exitCode,
1588
- fingerprints: r.fingerprints,
1589
- artifact: r.artifact,
1590
- lastMergedTask,
1591
- });
1592
- tipFailed = true;
1593
- }
1594
- }
2761
+ const tipFailed = await verifyIntegrationTipCached(intWt, commands, journal, { lastMergedTask });
1595
2762
  summary.tipVerify = tipFailed ? "failed" : "passed";
1596
2763
  if (tipFailed && lastMergedTask)
1597
2764
  summary.lastMergedTask = lastMergedTask;
@@ -1617,20 +2784,11 @@ export async function runDaemon(repoRoot, opts = {}) {
1617
2784
  return summary;
1618
2785
  }
1619
2786
  catch (err) {
1620
- if (runStarted && !taskLoopStarted && !journal.read().some((e) => e.event === "run-end")) {
1621
- journal.append("run-end", undefined, {
1622
- runId,
1623
- branch,
1624
- done: [],
1625
- failed: [],
1626
- human: [],
1627
- blocked: [],
1628
- pending: [],
1629
- phase: "setup",
1630
- fatal: true,
1631
- error: err instanceof Error ? err.message : String(err),
1632
- });
1633
- }
2787
+ // T7 (v1.86): guarded — a journal read/append failure while recording the fatal run-end is
2788
+ // reported alongside err, never instead of it; recordFatalRunEnd never throws, so the original
2789
+ // error (message, stack, cause) always reaches the caller verbatim.
2790
+ if (runStarted && !taskLoopStarted)
2791
+ recordFatalRunEnd(journal, runId, branch, err);
1634
2792
  throw err;
1635
2793
  }
1636
2794
  finally {