tickmarkr 1.80.0 → 1.83.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (50) hide show
  1. package/README.md +4 -4
  2. package/dist/brand.d.ts +4 -1
  3. package/dist/brand.js +46 -7
  4. package/dist/cli/commands/approve.js +53 -8
  5. package/dist/cli/commands/plan.js +17 -2
  6. package/dist/cli/commands/report.js +15 -6
  7. package/dist/cli/commands/ui.d.ts +1 -1
  8. package/dist/cli/commands/ui.js +47 -4
  9. package/dist/cli/index.d.ts +1 -1
  10. package/dist/cli/index.js +1 -1
  11. package/dist/compile/collateral.d.ts +19 -0
  12. package/dist/compile/collateral.js +90 -0
  13. package/dist/compile/index.js +18 -4
  14. package/dist/drivers/herdr.d.ts +1 -0
  15. package/dist/drivers/herdr.js +19 -0
  16. package/dist/drivers/types.d.ts +1 -0
  17. package/dist/gates/acceptance.js +50 -8
  18. package/dist/gates/llm.js +34 -19
  19. package/dist/gates/review.d.ts +9 -1
  20. package/dist/gates/review.js +173 -7
  21. package/dist/gates/run-gates.d.ts +1 -0
  22. package/dist/gates/run-gates.js +18 -1
  23. package/dist/report/bundle.d.ts +6 -1
  24. package/dist/report/bundle.js +13 -3
  25. package/dist/run/daemon.d.ts +5 -0
  26. package/dist/run/daemon.js +188 -27
  27. package/dist/run/journal.d.ts +5 -0
  28. package/dist/run/journal.js +71 -1
  29. package/dist/tui/cockpit/capture.d.ts +107 -1
  30. package/dist/tui/cockpit/capture.js +287 -10
  31. package/dist/tui/cockpit/components.d.ts +15 -86
  32. package/dist/tui/cockpit/components.js +49 -55
  33. package/dist/tui/cockpit/derive.d.ts +73 -1
  34. package/dist/tui/cockpit/derive.js +211 -15
  35. package/dist/tui/cockpit/keys.d.ts +244 -0
  36. package/dist/tui/cockpit/keys.js +634 -0
  37. package/dist/tui/cockpit/layout.d.ts +162 -0
  38. package/dist/tui/cockpit/layout.js +229 -0
  39. package/dist/tui/cockpit/live.d.ts +80 -0
  40. package/dist/tui/cockpit/live.js +311 -0
  41. package/dist/tui/cockpit/run-cockpit.d.ts +65 -2
  42. package/dist/tui/cockpit/run-cockpit.js +705 -128
  43. package/dist/tui/cockpit/setup-cockpit.d.ts +149 -1
  44. package/dist/tui/cockpit/setup-cockpit.js +432 -56
  45. package/dist/tui/cockpit/views.d.ts +72 -0
  46. package/dist/tui/cockpit/views.js +56 -0
  47. package/dist/tui/cockpit/width.d.ts +105 -0
  48. package/dist/tui/cockpit/width.js +276 -0
  49. package/fixtures/speckit-sample/tasks.md +2 -2
  50. package/package.json +3 -1
@@ -20,7 +20,8 @@ import { augmentRetryBrief, consult, renderRetryGuidance } from "./consult.js";
20
20
  import { runEnvironment } from "./environment.js";
21
21
  import { cleanupRunWorktrees, gitHead, linkNodeModules, npmDependencyInstallCommand, npmDependencyManifestChanged, sh, shGit, WORKTREE_LAYOUT_CONTRACT, worktreePath } from "./git.js";
22
22
  import { runInteractiveSeed } from "./interactive-seed.js";
23
- import { classifyWorkerResultCause, engagementComparable, Journal, loadRoutingProfile, newRunId, phaseForGate, recordedTaskFailureKind } from "./journal.js";
23
+ import { classifyTaskFailure, classifyWorkerResultCause, engagementComparable, Journal, loadRoutingProfile, newRunId, phaseForGate, recordedTaskFailureKind, reviewRoundsSinceApproval } from "./journal.js";
24
+ import { isDiffCapPark } from "../gates/review.js";
24
25
  import { acquireRunLock, releaseRunLock } from "./lock.js";
25
26
  import { ensureIntegration, integrationBranch, integrationHead, mergeTask, verifyIntegrationTip } from "./merge.js";
26
27
  import { nextChannel, route } from "../route/router.js";
@@ -71,7 +72,10 @@ const MAX_ATTEMPTS = 10; // ponytail: hard cap so a pathological ladder can neve
71
72
  // v1.70 T5: request-changes review rounds a single task may draw before it parks for a human decision
72
73
  // instead of cycling. Well below MAX_ATTEMPTS so review non-convergence is caught long before the
73
74
  // global cap. ponytail: literal constant; lift to cfg.review.roundCap only if a second knob-turner appears.
74
- const REVIEW_ROUND_CAP = 3;
75
+ // OBS-189 (park-economics patch): 3 → 2, counted per ENGAGEMENT (since the newest operator approval,
76
+ // reviewRoundsSinceApproval) — a park is now cheap (approve/--uphold both cost one decision, never a
77
+ // run), so non-convergence parks earlier and an upheld task re-enters with a fresh round budget.
78
+ const REVIEW_ROUND_CAP = 2;
75
79
  const BLOCKED_POLL_MS = 30_000; // between trailer-wait slices, check whether the pane is blocked on a prompt
76
80
  const PROVIDER_DEATH_REQUEUE_CAP = 2; // v1.46 T1: requeue same assignment twice, then fall through to the normal ladder
77
81
  const PROVIDER_DEATH_BACKOFF_MS = 500; // short backoff before provider-death requeue
@@ -88,6 +92,30 @@ export function setEarlyLaunchLivenessMsForTests(ms) {
88
92
  export function resetEarlyLaunchLivenessMsForTests() {
89
93
  earlyLaunchLivenessMs = EARLY_LAUNCH_LIVENESS_MS;
90
94
  }
95
+ // OBS-201: the liveness nudge — the daemon's ACTIVE response to an idle worker holding no trailer,
96
+ // replacing page-a-human-then-burn-the-window (289 of 692 worker-minutes in one measured day).
97
+ // Gate: herdr classifies the pane idle AND the monotonic tracker has seen nothing for
98
+ // NUDGE_AFTER_SILENT_MS (a worker grinding inside a tool run is `working` and never reaches the
99
+ // gate). One nudge per attempt; if the grace passes still idle with no progress, the wait concludes
100
+ // as a stall NOW and the consult sees the un-answered nudge instead of an hour of silence.
101
+ // Allowlist: claude-code first (steering path proven, OBS-122); widen per adapter only with a
102
+ // captured occupied-frame fixture (OBS-181 scar). The message builder takes no nonce — the
103
+ // self-reference guard holds by construction, an echoed bare token can never match the wait regex.
104
+ export const NUDGEABLE_ADAPTERS = new Set(["claude-code"]);
105
+ export const WORKER_NUDGE_MESSAGE = "tickmarkr liveness check: if the task is complete, print your TICKMARKR_RESULT completion trailer exactly as specified in your prompt now. If not, state your next concrete action and continue working.";
106
+ const NUDGE_AFTER_SILENT_MS = 3 * 60_000;
107
+ const WORKER_NUDGE_GRACE_MS = 4 * 60_000;
108
+ let nudgeAfterSilentMs = NUDGE_AFTER_SILENT_MS;
109
+ let workerNudgeGraceMs = WORKER_NUDGE_GRACE_MS;
110
+ /** Test seam — shrink the nudge gate and grace without minute-long sleeps. */
111
+ export function setNudgeTimingForTests(silentMs, graceMs) {
112
+ nudgeAfterSilentMs = silentMs;
113
+ workerNudgeGraceMs = graceMs;
114
+ }
115
+ export function resetNudgeTimingForTests() {
116
+ nudgeAfterSilentMs = NUDGE_AFTER_SILENT_MS;
117
+ workerNudgeGraceMs = WORKER_NUDGE_GRACE_MS;
118
+ }
91
119
  async function commitsAheadOf(base, wt) {
92
120
  const head = await gitHead(wt);
93
121
  if (head === base)
@@ -168,6 +196,7 @@ export async function runDaemon(repoRoot, opts = {}) {
168
196
  status: (s) => driver.status(s),
169
197
  read: (s, n) => driver.read(s, n),
170
198
  ...(driver.sendKey ? { sendKey: driver.sendKey.bind(driver) } : {}),
199
+ ...(driver.nudge ? { nudge: driver.nudge.bind(driver) } : {}),
171
200
  notify: (m, o) => driver.notify(m, o),
172
201
  close: closeSlot,
173
202
  worktree: (r, b, base) => driver.worktree(r, b, base),
@@ -453,13 +482,26 @@ export async function runDaemon(repoRoot, opts = {}) {
453
482
  const badReviewers = []; // v1.1: reviewer channels that produced unparseable output for this task
454
483
  // v1.70 T5 (review-convergence): failed review rounds this task has drawn, counted from the per-task
455
484
  // review history already in the journal — the SAME review gate-result stream onGate reads to grow the
456
- // reviewer-exclusion list (badReviewers), never a second parallel counter. request-changes rounds do
457
- // not exclude their reviewer (a fix is re-checked by the same seat), so their count lives here in the
458
- // shared history rather than in badReviewers' garbage-only exclusion subset.
459
- const reviewRoundsDrawn = () => journal.read().filter((e) => e.taskId === t.id && e.event === "gate-result" &&
460
- e.data.gate === "review" &&
461
- e.data.pass === false).length;
462
- let feedback = "";
485
+ // reviewer-exclusion list (badReviewers), never a second parallel counter. OBS-189: scoped to the
486
+ // current engagement — an operator approval (uphold or accept) resets the round budget, so an upheld
487
+ // task can dispatch its funded attempt instead of re-parking against the whole journal's history.
488
+ const reviewRoundsDrawn = () => reviewRoundsSinceApproval(journal.read(), t.id);
489
+ // OBS-193: journal the in-gate review retry (mirrors judge-retry) and exclude the flaked seat from
490
+ // later attempts' reviewer picks. One helper, called from both onGate sites (satisfied-gate + main).
491
+ const noteReviewRetry = (g) => {
492
+ const rr = g.meta?.reviewRetry;
493
+ if (g.gate === "review" && rr && typeof rr.flaked === "string" && typeof rr.retried === "string") {
494
+ journal.append("review-retry", t.id, {
495
+ gate: "review", flaked: rr.flaked, retried: rr.retried,
496
+ ...(g.meta?.unparseable === true ? { secondUnparseable: true } : {}),
497
+ });
498
+ badReviewers.push(rr.flaked);
499
+ }
500
+ };
501
+ // OBS-189: the operator upheld the reviewer — the findings ARE the brief for this funded attempt.
502
+ let feedback = rs?.upheldFeedback
503
+ ? `The operator UPHELD the reviewer's findings — address them without discarding landed work.\nreview: ${rs.upheldFeedback}`
504
+ : "";
463
505
  let ladderIdx = 0;
464
506
  let modeFallbackNoted = false; // v1.2: journal the interactive→print fallback once per task, not per attempt
465
507
  let gateFails = 0; // TEL-02: incremented ONLY where feedback is built from failing gates — never derived from attempts (quota failovers bump attempts too, Pitfall 6)
@@ -484,6 +526,23 @@ export async function runDaemon(repoRoot, opts = {}) {
484
526
  }
485
527
  return next;
486
528
  };
529
+ // OBS-202 (operator law: "you can spawn as many as you want"): channels are session FACTORIES,
530
+ // not consumed seats — a tried channel can always host a fresh worker session, and a fresh
531
+ // session carries none of the failed attempt's baggage. When the untried pool is empty, recycle
532
+ // the best LIVE channel (preferring a different seat than the current one) instead of parking
533
+ // on artificial scarcity; MAX_ATTEMPTS and the review round cap are the real bounds. Only a
534
+ // fleet whose every channel is DEMOTED (verified dead: auth/setup/provider outage) has nothing
535
+ // left to spawn — that case alone still parks.
536
+ const failoverOrRecycle = (site) => {
537
+ const next = failover(site);
538
+ if (next)
539
+ return next;
540
+ const recycled = nextChannel(assignment, t, cfg, channels, [channelKey(assignment)], profile, demotedChannels)
541
+ ?? (demotedChannels.has(channelKey(assignment)) ? null : assignment);
542
+ if (recycled)
543
+ journal.append("channel-recycle", t.id, { site, channel: channelKey(recycled) });
544
+ return recycled;
545
+ };
487
546
  const runConsult = (trigger, transcript, diffOrFeedback, gates) => {
488
547
  consults++;
489
548
  return consult({
@@ -527,13 +586,15 @@ export async function runDaemon(repoRoot, opts = {}) {
527
586
  }
528
587
  }
529
588
  }
530
- const next = failover("consult-reroute");
589
+ const next = failoverOrRecycle("consult-reroute");
531
590
  if (next) {
532
591
  assignment = next;
533
- tried.push(channelKey(next));
592
+ const k = channelKey(next);
593
+ if (!tried.includes(k))
594
+ tried.push(k);
534
595
  return true;
535
596
  }
536
- await park(t, "consult said reroute but every channel is exhausted", "reroute-exhausted", assignment, attempts, startMs, gateFails, consults, tokens, metered, retryMode);
597
+ await park(t, "consult said reroute but every channel is demoted (verified dead) — nothing left to spawn", "reroute-exhausted", assignment, attempts, startMs, gateFails, consults, tokens, metered, retryMode);
537
598
  return false;
538
599
  }
539
600
  await park(t, `consult verdict: ${v.action} — ${v.notes}`, trigger, assignment, attempts, startMs, gateFails, consults, tokens, metered, retryMode); // decompose|human
@@ -554,6 +615,23 @@ export async function runDaemon(repoRoot, opts = {}) {
554
615
  const wt = await driver.worktree(repoRoot, taskBranch, taskBase);
555
616
  const carriedCommits = await cherryPickCommits(wt, commitsToCarry);
556
617
  journal.append("worktree-recreation", t.id, { attempted: commitsToCarry, carried: carriedCommits });
618
+ // OBS-212: same fail-closed rule as the dispatch path — but this path is worse, because it runs
619
+ // ONLY the gates after the approved one and then MERGES. T3 took it on run-20260728-110135:
620
+ // approved past review at 11:22, recreated at 12:50, and phase-start{gates} / phase-start{merge}
621
+ // landed in the same second with zero gate-result events. Work missing here is merged unverified.
622
+ {
623
+ const present = new Set(carriedCommits);
624
+ for (const h of commitsToCarry) {
625
+ if (!present.has(h) && (await shGit(`git merge-base --is-ancestor ${shq(h)} HEAD`, wt)).code === 0) {
626
+ present.add(h);
627
+ }
628
+ }
629
+ const lost = commitsToCarry.filter((h) => !present.has(h));
630
+ if (lost.length > 0) {
631
+ await park(t, `carry lost ${lost.length} of ${commitsToCarry.length} verified commit(s) recreating the worktree for approved gate ${satisfiedGate} (first missing: ${lost[0].slice(0, 10)}) — refusing to merge a tree that is missing landed work`, "infra", assignment, rs?.attempts ?? 0, startMs, gateFails, consults, tokens, metered, retryMode);
632
+ return;
633
+ }
634
+ }
557
635
  if (!linkNodeModules(repoRoot, wt, { force: true })) {
558
636
  await park(t, "environmental: node_modules link could not be re-asserted before gates (OBS-47)", "setup", assignment, rs?.attempts ?? 0, startMs, gateFails, consults, tokens, metered, retryMode);
559
637
  return;
@@ -586,7 +664,7 @@ export async function runDaemon(repoRoot, opts = {}) {
586
664
  journal.phaseStart(t.id, "gates");
587
665
  const { results } = await runGates(resumedTask, {
588
666
  worktree: wt, baseRef: taskBase, result: priorResult, author: gateAuthor,
589
- commands, baseline, channels, adapters, cfg,
667
+ commands, baseline, channels, adapters, cfg, artifactDir: journal.dir,
590
668
  via: cfg.visibility.llm === "pane"
591
669
  ? {
592
670
  driver: trackedDriver,
@@ -607,6 +685,7 @@ export async function runDaemon(repoRoot, opts = {}) {
607
685
  gate: g.gate, pass: g.pass, details: g.details,
608
686
  ...(g.meta?.skipped === true ? { skipped: true } : {}),
609
687
  });
688
+ noteReviewRetry(g);
610
689
  if (g.gate === "review" && !g.pass && /unparseable/.test(g.details)
611
690
  && typeof g.meta?.reviewer === "string") {
612
691
  badReviewers.push(g.meta.reviewer);
@@ -669,11 +748,11 @@ export async function runDaemon(repoRoot, opts = {}) {
669
748
  }
670
749
  // v1.70 T5: a task that has already drawn REVIEW_ROUND_CAP request-changes review rounds parks for
671
750
  // a human decision instead of dispatching another round. Condition on the review history (data),
672
- // never the code path — and let an operator's approval (the human decision the cap asked for)
673
- // release it, mirroring the humanGate guard's condition-on-approval precedent. Rounds only accrue
674
- // after attempt 0, so the guard skips the journal read on the happy path.
675
- if (attempt > 0 && !approved.has(t.id) && reviewRoundsDrawn() >= REVIEW_ROUND_CAP) {
676
- await park(t, `review round cap (${REVIEW_ROUND_CAP}) reached — request-changes reviews not converging; a human should decide`, "gate-fail", assignment, attempt, startMs, gateFails, consults, tokens, metered, retryMode);
751
+ // never the code path. OBS-189: the count is engagement-scoped — any operator approval resets it
752
+ // by construction (reviewRoundsSinceApproval), so the old blanket approved-task exemption is gone
753
+ // and an upheld task that STILL cannot converge re-parks after another full round budget.
754
+ if (attempt > 0 && reviewRoundsDrawn() >= REVIEW_ROUND_CAP) {
755
+ await park(t, `review round cap (${REVIEW_ROUND_CAP}) reached this engagement — \`tickmarkr approve\` accepts the diff past review; \`tickmarkr approve --uphold\` funds one fixed attempt carrying the findings`, "gate-fail", assignment, attempt, startMs, gateFails, consults, tokens, metered, retryMode);
677
756
  return;
678
757
  }
679
758
  // OBS-57: a demoted channel must not be re-dispatched on consult retry or provider requeue.
@@ -736,6 +815,27 @@ export async function runDaemon(repoRoot, opts = {}) {
736
815
  presentCommits.add(h);
737
816
  }
738
817
  }
818
+ // OBS-212: losing carried work is WORK DESTRUCTION, never an ordinary continue. Every commit in
819
+ // commitsToCarry is one a prior attempt landed and the gates verified. If a commit is neither
820
+ // cherry-picked nor already an ancestor of the new base, dispatching a worker onto this tree
821
+ // silently makes it re-buy verified work — the failure this cherry-pick exists to prevent
822
+ // (OBS-58, comment above) happening silently inside the mechanism itself. Measured on
823
+ // run-20260728-110135: T2 carried 0 of 17 at 17:07 and T1 carried 0 of 15 at 21:37, twenty
824
+ // minutes after T2's merge advanced the integration tip. Nothing failed and nothing parked;
825
+ // the run simply re-paid for the work. Fail closed instead — a visible park beats a silent loss.
826
+ // On the DISPATCH path a drop is not always a defect: a human can re-pend a task with new
827
+ // intent, and the superseded commit then SHOULD be left behind (pinned by the worktree-cleanup
828
+ // resume test, where T1 and T2 both write shared.txt and the loser is re-scripted). We cannot
829
+ // tell supersession from destruction here, so this path stays loud rather than fail-closed —
830
+ // the harm that cost this run ~9 hours was the SILENCE, not the drop. The merge path, where a
831
+ // drop is never legitimate, does fail closed (see the satisfied-gate site above).
832
+ const lostCommits = commitsToCarry.filter((h) => !presentCommits.has(h));
833
+ if (lostCommits.length > 0) {
834
+ journal.append("work-loss", t.id, {
835
+ site: "dispatch", base: taskBase, attempted: commitsToCarry, carried: carriedCommits, lost: lostCommits,
836
+ });
837
+ await driver.notify(`tickmarkr ${runId}: ${t.id} lost ${lostCommits.length} of ${commitsToCarry.length} landed commit(s) recreating its worktree — it will re-do that work`, { tier: "attention" });
838
+ }
739
839
  if (feedback || priorNamed.length > 0) {
740
840
  feedback = augmentRetryBrief(feedback, { attempted: commitsToCarry, carried: carriedCommits, present: presentCommits });
741
841
  }
@@ -908,6 +1008,11 @@ export async function runDaemon(repoRoot, opts = {}) {
908
1008
  // v1.22 T5 / OBS-19: auto-answer a fingerprint-matched trust dialog exactly once per slot.
909
1009
  // Any other blocked/idle dialog still pages the operator (paged latch below).
910
1010
  let trustAnswered = false;
1011
+ // OBS-201: one liveness nudge per attempt; the grace deadline is its OWN timer, never the
1012
+ // stall window (the nudge's pane echo is absorbed before it starts, or the echo itself
1013
+ // would reset the window and make the early conclusion unreachable).
1014
+ let nudged = false;
1015
+ let nudgeDeadline;
911
1016
  finished = false;
912
1017
  exitCode = null;
913
1018
  // OBS-54: reaping keys on new pane output, not dispatch wall clock. Poll at least twice per
@@ -980,9 +1085,53 @@ export async function runDaemon(repoRoot, opts = {}) {
980
1085
  /* read/send failed — fall through to page the operator */
981
1086
  }
982
1087
  }
983
- paged = true; // page once — the visible pane is the operator's to unblock; task timeout is the backstop
984
- const why = st === "blocked" ? "is blocked on a prompt — approve in its pane" : "looks idle without finishing — check its pane";
985
- await driver.notify(`tickmarkr ${runId}: ${slot.name} ${why}`, { tier: "attention" });
1088
+ // OBS-201: the daemon ACTS on an idle pane before paging anyone. Gate: idle AND the
1089
+ // monotonic tracker silent ≥ the nudge threshold (a worker inside a tool run reads as
1090
+ // `working` and never gets here; a briefly-settling TUI hasn't been silent long enough).
1091
+ const nudgeable = st === "idle" && !!driver.nudge && NUDGEABLE_ADAPTERS.has(adapter.id);
1092
+ if (nudgeable && !nudged) {
1093
+ if (Date.now() - lastProgressAt >= nudgeAfterSilentMs) {
1094
+ nudged = true;
1095
+ if (await driver.nudge(slot, WORKER_NUDGE_MESSAGE)) {
1096
+ // absorb the nudge's own echo BEFORE arming the grace timer — post-nudge progress
1097
+ // is measured against this baseline, not against the echo.
1098
+ const echo = await driver.read(slot, 1000);
1099
+ stallProgress.observe({ paneText: echo, contextTokens });
1100
+ nudgeDeadline = Date.now() + workerNudgeGraceMs;
1101
+ journal.append("worker-nudge", t.id, { slot: slot.name, attempt });
1102
+ }
1103
+ else {
1104
+ journal.append("worker-nudge-failed", t.id, { slot: slot.name, attempt });
1105
+ // nudge undeliverable — fall back to today's behavior: page once, window backstop
1106
+ paged = true;
1107
+ await driver.notify(`tickmarkr ${runId}: ${slot.name} looks idle without finishing — check its pane`, { tier: "attention" });
1108
+ }
1109
+ }
1110
+ // idle but not yet silent past the threshold: neither nudge nor page this slice
1111
+ }
1112
+ else if (nudgeable && nudged && nudgeDeadline !== undefined) {
1113
+ if (Date.now() >= nudgeDeadline && Date.now() - lastProgressAt >= workerNudgeGraceMs) {
1114
+ // grace spent, still idle, no post-nudge progress: re-harvest once (the trailer may
1115
+ // have landed between polls), then conclude the wait as a stall NOW — the consult
1116
+ // sees the un-answered nudge instead of the remainder of the window.
1117
+ nudgeDeadline = undefined;
1118
+ output = await driver.read(slot, 1000);
1119
+ finished = new RegExp(trailerPattern(nonce)).test(output);
1120
+ const exit = exitRe.exec(output);
1121
+ if (finished || exit) {
1122
+ exitCode = exit ? Number(exit[1]) : null;
1123
+ await sampleContext();
1124
+ break;
1125
+ }
1126
+ journal.append("worker-nudge-expired", t.id, { slot: slot.name, attempt, graceMs: workerNudgeGraceMs });
1127
+ lastProgressAt = Date.now() - stallWindowMs; // the existing harvest/classify tail runs unmodified
1128
+ }
1129
+ }
1130
+ else {
1131
+ paged = true; // page once — the visible pane is the operator's to unblock; task timeout is the backstop
1132
+ const why = st === "blocked" ? "is blocked on a prompt — approve in its pane" : "looks idle without finishing — check its pane";
1133
+ await driver.notify(`tickmarkr ${runId}: ${slot.name} ${why}`, { tier: "attention" });
1134
+ }
986
1135
  }
987
1136
  // a dead pane or a false-positive marker display returns fast — sleep the unspent slice, never hot-spin
988
1137
  const spent = Date.now() - sliceStart;
@@ -1172,7 +1321,7 @@ export async function runDaemon(repoRoot, opts = {}) {
1172
1321
  const from = channelKey(assignment);
1173
1322
  demotedChannels.add(from); // excluded for later attempts AND later tasks in this run
1174
1323
  journal.append("channel-exclusion", t.id, { channel: from, reason: dead, kind: "dead-channel" });
1175
- const next = failover("dead-channel");
1324
+ const next = failoverOrRecycle("dead-channel");
1176
1325
  journal.append("dead-channel-failover", t.id, { reason: dead, from, to: next ? channelKey(next) : null });
1177
1326
  if (next) {
1178
1327
  await driver.notify(`tickmarkr ${runId}: ${t.id} dead channel (${dead}) failover`, { tier: "attention" });
@@ -1255,6 +1404,7 @@ export async function runDaemon(repoRoot, opts = {}) {
1255
1404
  }
1256
1405
  }
1257
1406
  journal.append("gate-result", t.id, { gate: g.gate, pass: g.pass, details: g.details, ...(g.meta?.skipped === true ? { skipped: true } : {}) });
1407
+ noteReviewRetry(g);
1258
1408
  // v1.1 failover: never re-ask a reviewer channel that produced garbage for this task
1259
1409
  if (g.gate === "review" && !g.pass && /unparseable/.test(g.details) && typeof g.meta?.reviewer === "string") {
1260
1410
  badReviewers.push(g.meta.reviewer);
@@ -1267,7 +1417,7 @@ export async function runDaemon(repoRoot, opts = {}) {
1267
1417
  journal.phaseStart(t.id, "gates");
1268
1418
  ({ results, commits } = await runGates(t, {
1269
1419
  worktree: wt, baseRef: taskBase, result, author: assignment,
1270
- commands, baseline, channels, adapters, cfg,
1420
+ commands, baseline, channels, adapters, cfg, artifactDir: journal.dir,
1271
1421
  via: cfg.visibility.llm === "pane"
1272
1422
  ? {
1273
1423
  driver: trackedDriver,
@@ -1335,8 +1485,19 @@ export async function runDaemon(repoRoot, opts = {}) {
1335
1485
  // trailer) over the harness slot name; absent hook or no capture keeps today's slot-name id.
1336
1486
  retrySession = { channel: channelKey(assignment), id: adapter.sessionIdFrom?.(output) ?? sessionId, contextTokens };
1337
1487
  feedback = results.filter((g) => !g.pass).map((g) => `${g.gate}: ${g.details}`).join("\n\n");
1338
- const step = r.ladder[Math.min(ladderIdx++, r.ladder.length - 1)];
1339
- journal.append("escalation", t.id, { step, attempt: attempt + 1 });
1488
+ // OBS-189/G3 (park-economics patch): a request-changes review is a findings brief, not a worker
1489
+ // defect — the fix attempt stays on the same channel with the findings as feedback and consumes
1490
+ // no escalation-ladder rung. Bounded by the engagement round cap at the top of this loop.
1491
+ // Unparseable verdicts (already retried in-gate, OBS-193) and diff-cap trips (the diff cannot
1492
+ // shrink by retrying, OBS-48) fall through to the ladder unchanged. Review runs last, so a
1493
+ // failed review with every other gate green is exactly "the work landed, the reviewer objects".
1494
+ const reviewFail = results.find((g) => g.gate === "review" && !g.pass);
1495
+ const reviewFixRetry = reviewFail !== undefined
1496
+ && reviewFail.meta?.unparseable !== true
1497
+ && !isDiffCapPark(reviewFail)
1498
+ && results.every((g) => g.pass || g.gate === "review");
1499
+ const step = reviewFixRetry ? "retry" : r.ladder[Math.min(ladderIdx++, r.ladder.length - 1)];
1500
+ journal.append("escalation", t.id, { step, attempt: attempt + 1, ...(reviewFixRetry ? { reviewFix: true } : {}) });
1340
1501
  await driver.notify(`tickmarkr ${runId}: ${t.id} escalation: ${step}`, { tier: "attention" });
1341
1502
  if (step === "retry")
1342
1503
  continue;
@@ -1374,8 +1535,8 @@ export async function runDaemon(repoRoot, opts = {}) {
1374
1535
  .catch(async (err) => {
1375
1536
  const taskEvents = journal.read().filter((e) => e.taskId === t.id);
1376
1537
  const dispatch = [...taskEvents].reverse().find((e) => e.event === "task-dispatch");
1377
- const verifiedFailure = taskEvents.some((e) => e.event === "gate-result" && e.data.pass === false);
1378
- const kind = verifiedFailure ? "gate-fail" : dispatch ? "dispatch" : "infra";
1538
+ // OBS-206: shared rule with `resume --retry-failed` — see classifyTaskFailure.
1539
+ const kind = classifyTaskFailure(taskEvents);
1379
1540
  const attempts = dispatch && Number.isInteger(dispatch.data.attempt) ? dispatch.data.attempt : 0;
1380
1541
  graph = setStatus(graph, t.id, "failed");
1381
1542
  saveGraph(repoRoot, graph);
@@ -16,11 +16,15 @@ export interface ResumeState {
16
16
  attempts: number;
17
17
  tried: string[];
18
18
  lastAssignment?: Assignment;
19
+ upheldFeedback?: string;
19
20
  }
20
21
  export declare const CHANNEL_EXCLUSION_KINDS: readonly ["dead-channel"];
21
22
  export type ChannelExclusionKind = (typeof CHANNEL_EXCLUSION_KINDS)[number];
22
23
  export declare const ATTEMPT_CAP_RELEASE: "attempt-cap";
23
24
  export declare const GATE_SATISFIED_RELEASE: "gate-satisfied";
25
+ export declare const REVIEW_UPHELD_RELEASE: "review-upheld";
26
+ export declare const RECHECK_RELEASE: "recheck";
27
+ export declare function reviewRoundsSinceApproval(events: JournalEvent[], taskId: string): number;
24
28
  export declare const PARK_KINDS: readonly ["human-gate", "ladder-exhausted", "attempt-cap", "gate-fail", "quota", "reroute-exhausted", "setup", "stall", "merge-conflict", "tip-moved", "infra", "dispatch"];
25
29
  export type ParkKind = (typeof PARK_KINDS)[number];
26
30
  export declare const RETRY_MODES: readonly ["resume", "fresh"];
@@ -28,6 +32,7 @@ export type RetryMode = (typeof RETRY_MODES)[number];
28
32
  export declare const WORKER_RESULT_CAUSES: readonly ["provider-death", "stall-timeout", "malformed-trailer", "clean-exit-no-trailer"];
29
33
  export type WorkerResultCause = (typeof WORKER_RESULT_CAUSES)[number];
30
34
  export declare function isQualityFailureParkKind(kind: ParkKind): boolean;
35
+ export declare function classifyTaskFailure(taskEvents: JournalEvent[]): ParkKind;
31
36
  export declare function recordedTaskFailureKind(events: JournalEvent[], taskId: string): ParkKind | undefined;
32
37
  export declare function runHasEnded(events: JournalEvent[]): boolean;
33
38
  /** OBS-53: classify worker-result failures so retries and routing see the true signal, not one lumped bucket. */
@@ -44,6 +44,32 @@ export const ATTEMPT_CAP_RELEASE = "attempt-cap";
44
44
  // OBS-130: task-approved carries the exact failed gate the operator satisfied. The release tag keeps
45
45
  // ordinary humanGate and attempt-cap approvals byte-compatible while making this authority explicit.
46
46
  export const GATE_SATISFIED_RELEASE = "gate-satisfied";
47
+ // OBS-189: the second human decision a review park needs. `approve` (gate-satisfied) accepts the diff
48
+ // the reviewer rejected; `approve --uphold` sides WITH the reviewer and funds one fixed worker attempt
49
+ // carrying the findings — a park costs an attempt, never a fresh run that re-executes green tasks.
50
+ export const REVIEW_UPHELD_RELEASE = "review-upheld";
51
+ // OBS-203: the third decision a gate-fail park needs. Plain approve WAIVES the failed gate (and every
52
+ // gate before it — daemon.ts remainingGates), which is wrong when the gate failed against a stale task
53
+ // DECLARATION rather than a bad diff: amend the spec's files[], recompile, and the gate now passes
54
+ // honestly. This release re-dispatches with the whole gate suite intact and no gate marked satisfied,
55
+ // so the corrected declaration is the thing that earns the green. Budget semantics match attempt-cap
56
+ // (fresh attempts, tried survives) because the park cost the task its remaining budget.
57
+ export const RECHECK_RELEASE = "recheck";
58
+ // OBS-189: review rounds are scoped to the current ENGAGEMENT — the stretch since the newest operator
59
+ // approval for the task. A whole-journal count re-parks an upheld task before its funded attempt can
60
+ // dispatch (measured live on run-20260726-213539), making a fresh journal the only escape.
61
+ export function reviewRoundsSinceApproval(events, taskId) {
62
+ let rounds = 0;
63
+ for (const e of events) {
64
+ if (e.taskId !== taskId)
65
+ continue;
66
+ if (e.event === "task-approved")
67
+ rounds = 0;
68
+ else if (e.event === "gate-result" && e.data.gate === "review" && e.data.pass === false)
69
+ rounds++;
70
+ }
71
+ return rounds;
72
+ }
47
73
  // Fail-closed shape for a dispatched assignment (journal.ts:75-90 posture): a malformed assignment in
48
74
  // one dispatch degrades that single task toward today's behavior — counts toward attempts, contributes
49
75
  // nothing to tried, poisons only lastAssignment — never crashes resume, never poisons other tasks.
@@ -66,6 +92,25 @@ export function isQualityFailureParkKind(kind) {
66
92
  outcome: "human", durationMs: 0, parkKind: kind,
67
93
  }) === 0;
68
94
  }
95
+ // OBS-206: ONE classification rule, called at write time by the daemon's failure handler and at read
96
+ // time by `resume --retry-failed`, so the two can never drift and a label written by an older binary
97
+ // can never outvote the evidence sitting in the same journal. `taskEvents` is this task's events up to
98
+ // (not including) the failure. The gate evidence is scoped to the CURRENT attempt: a whole-history
99
+ // scan asks "did this task EVER fail a gate", which mislabels an infra death as a verified failure for
100
+ // any task that ever failed one gate — and, since only "dispatch" is retryable, locks it out forever.
101
+ export function classifyTaskFailure(taskEvents) {
102
+ let dispatchIdx = -1;
103
+ for (let i = taskEvents.length - 1; i >= 0; i--) {
104
+ if (taskEvents[i].event === "task-dispatch") {
105
+ dispatchIdx = i;
106
+ break;
107
+ }
108
+ }
109
+ if (dispatchIdx < 0)
110
+ return "infra"; // never dispatched — nothing to retry
111
+ return taskEvents.slice(dispatchIdx + 1)
112
+ .some((e) => e.event === "gate-result" && e.data.pass === false) ? "gate-fail" : "dispatch";
113
+ }
69
114
  // The newest terminal event owns the cause. Unknown/malformed kinds fail toward undefined so a
70
115
  // task-failed row keeps the legacy red treatment instead of being mistaken for recoverable noise.
71
116
  export function recordedTaskFailureKind(events, taskId) {
@@ -73,6 +118,12 @@ export function recordedTaskFailureKind(events, taskId) {
73
118
  const e = events[i];
74
119
  if (e.taskId !== taskId || (e.event !== "task-human" && e.event !== "task-failed"))
75
120
  continue;
121
+ // A park kind is a daemon DECISION and is read exactly as recorded. A failure kind is a
122
+ // CLASSIFICATION over evidence that is still in this journal, so it is re-derived (OBS-206) —
123
+ // recomputation agrees with every correctly-written label and repairs the wrong ones in place.
124
+ if (e.event === "task-failed") {
125
+ return classifyTaskFailure(events.slice(0, i).filter((x) => x.taskId === taskId));
126
+ }
76
127
  return typeof e.data.kind === "string" && PARK_KINDS.includes(e.data.kind)
77
128
  ? e.data.kind
78
129
  : undefined;
@@ -461,9 +512,14 @@ export class Journal {
461
512
  replayResumeState() {
462
513
  const m = new Map();
463
514
  const pendingReroute = new Set(); // reroute verdicts not yet cleared by a later dispatch
515
+ const lastReviewFail = new Map(); // OBS-189: newest failed review details per task
464
516
  for (const e of this.read()) {
465
517
  if (!e.taskId)
466
518
  continue;
519
+ if (e.event === "gate-result" && e.data.gate === "review" && e.data.pass === false
520
+ && typeof e.data.details === "string") {
521
+ lastReviewFail.set(e.taskId, e.data.details);
522
+ }
467
523
  if (e.event === "task-dispatch") {
468
524
  // A subsequent dispatch clears the pending reroute — the reroute was acted on pre-kill.
469
525
  pendingReroute.delete(e.taskId);
@@ -490,7 +546,8 @@ export class Journal {
490
546
  // A reroute bans the in-force channel; retry/decompose/human verdicts ban nothing (D-03).
491
547
  pendingReroute.add(e.taskId);
492
548
  }
493
- else if (e.event === "task-approved" && e.data.release === ATTEMPT_CAP_RELEASE) {
549
+ else if (e.event === "task-approved"
550
+ && (e.data.release === ATTEMPT_CAP_RELEASE || e.data.release === RECHECK_RELEASE)) {
494
551
  // v1.24 OBS-18: operator released an attempt-cap park. Pre-v1.24 task-approved events have no
495
552
  // `release` key ⇒ this branch never fires (corpus criterion: identical statuses + resume state).
496
553
  // attempts reset to 0 so the daemon's attempt-cap check does not re-park in the same tick;
@@ -503,6 +560,19 @@ export class Journal {
503
560
  st.lastAssignment = undefined;
504
561
  }
505
562
  }
563
+ else if (e.event === "task-approved" && e.data.release === REVIEW_UPHELD_RELEASE) {
564
+ // OBS-189: the operator upheld the reviewer. Same budget semantics as the attempt-cap release
565
+ // (fresh attempts, tried survives, no burned-channel restore) PLUS the findings carry: the
566
+ // newest failed review's details ride into the funded attempt as its retry brief.
567
+ const st = m.get(e.taskId);
568
+ if (st) {
569
+ st.attempts = 0;
570
+ st.lastAssignment = undefined;
571
+ const details = lastReviewFail.get(e.taskId);
572
+ if (details)
573
+ st.upheldFeedback = details;
574
+ }
575
+ }
506
576
  }
507
577
  // Trailing-reroute edge (D-01 kill between verdict and dispatch): a reroute verdict with NO
508
578
  // subsequent dispatch means the last-dispatched channel is itself banned — add it to tried (if