tickmarkr 2.5.2 → 2.5.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (41) hide show
  1. package/dist/adapters/qwen.d.ts +1 -0
  2. package/dist/adapters/qwen.js +8 -0
  3. package/dist/adapters/types.d.ts +1 -0
  4. package/dist/cli/commands/doctor.d.ts +10 -0
  5. package/dist/cli/commands/doctor.js +50 -1
  6. package/dist/cli/commands/plan.js +144 -3
  7. package/dist/cli/commands/resume.js +2 -0
  8. package/dist/cli/commands/run.js +12 -0
  9. package/dist/cli/commands/status.js +8 -14
  10. package/dist/compile/collateral.d.ts +2 -0
  11. package/dist/compile/collateral.js +50 -9
  12. package/dist/compile/ownership.d.ts +16 -0
  13. package/dist/compile/ownership.js +113 -13
  14. package/dist/config/config.d.ts +9 -0
  15. package/dist/config/config.js +13 -2
  16. package/dist/drivers/herdr.d.ts +5 -0
  17. package/dist/drivers/herdr.js +14 -3
  18. package/dist/drivers/types.d.ts +1 -0
  19. package/dist/gates/baseline.d.ts +5 -1
  20. package/dist/gates/baseline.js +18 -5
  21. package/dist/gates/llm.d.ts +1 -1
  22. package/dist/gates/llm.js +34 -12
  23. package/dist/gates/review.d.ts +46 -2
  24. package/dist/gates/review.js +111 -21
  25. package/dist/gates/run-gates.d.ts +2 -0
  26. package/dist/gates/run-gates.js +19 -10
  27. package/dist/route/router.d.ts +13 -0
  28. package/dist/route/router.js +64 -11
  29. package/dist/run/daemon.d.ts +3 -1
  30. package/dist/run/daemon.js +332 -53
  31. package/dist/run/git.d.ts +25 -0
  32. package/dist/run/git.js +68 -1
  33. package/dist/run/journal.js +16 -7
  34. package/dist/run/operator-state.d.ts +1 -1
  35. package/dist/run/operator-state.js +5 -9
  36. package/dist/tui/cockpit/live-runtime.js +12 -10
  37. package/package.json +1 -1
  38. package/skills/tickmarkr-auto/SKILL.md +1 -1
  39. package/skills/tickmarkr-loop/SKILL.md +1 -1
  40. package/skills/tickmarkr-overseer/SKILL.md +82 -4
  41. package/skills/tickmarkr-overseer/scripts/watch-launch.sh +38 -0
@@ -31,8 +31,9 @@ import { activeRetryBan, classifyTaskFailure, classifyWorkerResultCause, deferre
31
31
  import { isDiffCapPark, pickReviewer } from "../gates/review.js";
32
32
  import { acquireApprovalSerialization, acquireRunLock, isPidLive, releaseRunLock } from "./lock.js";
33
33
  import { ensureIntegration, integrationBranch, integrationHead, mergeTask, verifyIntegrationTip } from "./merge.js";
34
- import { nextChannel, route } from "../route/router.js";
34
+ import { climbChannel, marginalCostRank, nextChannel, route } from "../route/router.js";
35
35
  import { desiredPanes } from "./reconcile.js";
36
+ import { readTierLiveness, readWatchBoard, supervisionBeatPath } from "./supervision.js";
36
37
  import { harvestCpuFlatWindowMs, NUDGEABLE_ADAPTERS, PANE_READ_ROWS, StallProgressTracker, stallSnapshotBannerRows, WorkerTreeCpuAccountant, } from "./stall.js";
37
38
  // The live set is also the ownership claim shared by task cleanup and termination.
38
39
  export async function closeLiveSlot(liveSlots, driver, slot) {
@@ -307,7 +308,13 @@ let suiteWaitCeilingMs = SUITE_WAIT_CEILING_MS;
307
308
  export const setSuiteWaitCeilingForTests = (ms) => { suiteWaitCeilingMs = ms; };
308
309
  export const resetSuiteWaitCeilingForTests = () => { suiteWaitCeilingMs = SUITE_WAIT_CEILING_MS; };
309
310
  export const APPROVAL_POLL_MS = 250;
310
- export const APPROVAL_WINDOW_MS = 1_000;
311
+ export const APPROVAL_WINDOW_MS = 120_000;
312
+ // Keep ordinary park tests off the operator's production wait. Explicit timing tests
313
+ // use the same setter/reset pattern as the suite wait ceiling above.
314
+ const DEFAULT_APPROVAL_WINDOW_MS = process.env.VITEST ? 1 : APPROVAL_WINDOW_MS;
315
+ let approvalWindowMs = DEFAULT_APPROVAL_WINDOW_MS;
316
+ export const setApprovalWindowForTests = (ms) => { approvalWindowMs = ms; };
317
+ export const resetApprovalWindowForTests = () => { approvalWindowMs = DEFAULT_APPROVAL_WINDOW_MS; };
311
318
  const PROVIDER_DEATH_REQUEUE_CAP = 2; // v1.46 T1: requeue same assignment twice, then fall through to the normal ladder
312
319
  const PROVIDER_DEATH_BACKOFF_MS = 500; // short backoff before provider-death requeue
313
320
  const NO_TRAILER_DEMOTION_STREAK = 2; // OBS-57: consecutive no-trailer windows demote a channel for the rest of the run
@@ -1352,7 +1359,7 @@ export async function runDaemon(repoRoot, opts = {}) {
1352
1359
  return await runWithVerificationBudget(occupancyCapacity, async () => {
1353
1360
  // v1.51 T4: every dispatch provenance line begins with the mode and its source; when a pin won
1354
1361
  // the route (the final "→ " segment is a pin, not a degraded-to-auto tail) it names the mode it bypassed.
1355
- const dispatchProvenance = (p) => `mode ${rm.mode.mode} (${rm.source})${p.split("→ ").pop().startsWith("pin ") ? ` — pin bypasses mode ${rm.mode.mode}` : ""} · ${p}`;
1362
+ const dispatchProvenance = (p) => `mode ${rm.mode.mode} (${rm.source})${p.split("→ ").pop().startsWith("pin ") && !p.includes("not re-tried") ? ` — pin bypasses mode ${rm.mode.mode}` : ""} · ${p}`;
1356
1363
  // v1.87 T2: one pool per seat role, built once. `channels` stays the WORKER pool — every routing
1357
1364
  // call below reads it exactly as before — while the judge, review and consult seats each receive
1358
1365
  // the pool their own deny scope allows, so routing.deny.workers benches a channel for dispatch
@@ -1389,6 +1396,7 @@ export async function runDaemon(repoRoot, opts = {}) {
1389
1396
  liveSlots.add(s);
1390
1397
  return s;
1391
1398
  } } : {}),
1399
+ ...(driver.retireLostWatch ? { retireLostWatch: driver.retireLostWatch.bind(driver) } : {}),
1392
1400
  ...(driver.project ? { project: driver.project.bind(driver) } : {}),
1393
1401
  ...(driver.reconcile ? { reconcile: driver.reconcile.bind(driver) } : {}),
1394
1402
  notify: (m, o) => driver.notify(m, o),
@@ -1462,6 +1470,15 @@ export async function runDaemon(repoRoot, opts = {}) {
1462
1470
  const reviewHistory = journal.read()
1463
1471
  .filter((event) => event.event === "gate-result" && event.data.gate === "review" && typeof event.data.reviewer === "string")
1464
1472
  .map((event) => event.data.reviewer);
1473
+ // RF-1 (OBS-922 add.3): the seats THIS task's review rows name — the prior reviewers a later round's
1474
+ // floor holds. Task-scoped on purpose: reviewHistory rotates seats run-wide and is never floor evidence.
1475
+ // Each row's journaled dispatch tier travels with it, so a seat that left the pool on resume still counts.
1476
+ // Leg-2 T4 M2: a `review-no-verdict` note is dispatch evidence too — it lands BEFORE the in-gate retry's
1477
+ // gate-result, so a daemon killed between the two leaves it as the task's only record of that seat.
1478
+ const taskReviewers = (taskId) => journal.read()
1479
+ .filter((event) => event.taskId === taskId && typeof event.data.reviewer === "string"
1480
+ && ((event.event === "gate-result" && event.data.gate === "review") || event.event === "review-no-verdict"))
1481
+ .map((event) => ({ reviewer: event.data.reviewer, tier: event.data.reviewerTier }));
1465
1482
  // Capture this before this observer appends run-resume. A prior run-end belongs to a completed
1466
1483
  // lifecycle and must use the ordinary resume/redispatch rules; only an interrupted live attempt
1467
1484
  // owns reusable measurements or unclean-death residue.
@@ -1493,19 +1510,99 @@ export async function runDaemon(repoRoot, opts = {}) {
1493
1510
  const commands = detectGateCommands(repoRoot, cfg);
1494
1511
  const watchName = formatOwnedName({ role: "watch", taskId: "run", attempt: 0, runId });
1495
1512
  let watchSlot;
1496
- const openBoard = async () => {
1513
+ let boardOpened = false; // WB-1: only a board this daemon placed is watched below
1514
+ // Leg-2 T7 M1 (OBS-988): a reopen first retires the ghost through the driver's seam so the narrator
1515
+ // launches a NEW pane; a driver that answers with the lost pane anyway (no seam, or a cache it
1516
+ // would not drop) has not reopened anything, and that answer is a FAILED reopen, never a success.
1517
+ const openBoard = async (ghost) => {
1518
+ try {
1519
+ if (ghost?.slot)
1520
+ await trackedDriver.retireLostWatch?.(ghost.slot);
1521
+ const slot = await trackedDriver.narrator(repoRoot, watchCommand(runId), runId);
1522
+ if (ghost && slot.id === ghost.pane) {
1523
+ liveSlots.delete(slot); // the ghost again: no close is ever aimed at it
1524
+ return { ok: false, error: `narrator answered with the lost pane ${ghost.pane}` };
1525
+ }
1526
+ watchSlot = slot;
1527
+ boardOpened = true;
1528
+ return { ok: true, slot };
1529
+ }
1530
+ catch (error) {
1531
+ return { ok: false, error: error instanceof Error ? error.message : String(error) };
1532
+ }
1533
+ };
1534
+ const placeBoard = async () => {
1497
1535
  if (!trackedDriver.narrator) {
1498
1536
  noteCapabilityAbsent("narrator");
1537
+ return undefined;
1538
+ }
1539
+ const placed = await openBoard();
1540
+ if (placed.ok)
1541
+ return placed.slot;
1542
+ journal.append("watch-placement-failed", undefined, { error: placed.error });
1543
+ console.error(`tickmarkr: narrator not opened: ${placed.error}`);
1544
+ return undefined;
1545
+ };
1546
+ // WB-1 (OBS-988): the daemon watches its own cockpit. A board is lost when the watch tier's beat
1547
+ // aged past the supervision stale bound, the recorded arm has no presence, or the owner pid is dead.
1548
+ // The beat and presence checks wait for the UI to have armed (armId recorded) — a freshly reopened
1549
+ // board that has not armed yet is not a second loss. Reopens are bounded per run; past the bound
1550
+ // the loss is journaled `boardless` and the narrator is never called again.
1551
+ // Leg-2 T7 M2: one `watch-board-lost` row per LOSS. A loss is identified by the pane and pid the
1552
+ // owner record names; a reopen that fails leaves that record in place, so the next poll sees the
1553
+ // same loss, journals only its own reopen attempt, and the bound retires the run boardless from the
1554
+ // last failed attempt rather than from a repeated lost row.
1555
+ const MAX_BOARD_REOPENS = 3;
1556
+ let boardReopens = journal.read().filter((e) => e.event === "watch-board-reopened" || e.event === "watch-board-reopen-failed").length;
1557
+ let boardless = false;
1558
+ let journaledLoss;
1559
+ const boardLoss = () => {
1560
+ const owner = readWatchBoard(repoRoot, runId);
1561
+ if (!owner)
1562
+ return undefined;
1563
+ const lost = { pane: owner.pane, ...(owner.pid !== undefined ? { pid: owner.pid } : {}) };
1564
+ if (owner.armId !== undefined) {
1565
+ const beat = readTierLiveness(repoRoot, "watch");
1566
+ if (beat.state === "STALE")
1567
+ return { ...lost, beatAgeMs: beat.beatAgeMs };
1568
+ // ponytail: presence path math mirrors supervision.ts's private helper (`<tier>.live.<armId>`)
1569
+ if (!existsSync(join(dirname(supervisionBeatPath(repoRoot, "watch")), `watch.live.${owner.armId}`)))
1570
+ return lost;
1571
+ }
1572
+ if (owner.pid !== undefined && !isPidLive(owner.pid))
1573
+ return lost;
1574
+ return undefined;
1575
+ };
1576
+ const watchBoard = async () => {
1577
+ if (!boardOpened || boardless || !trackedDriver.narrator)
1578
+ return;
1579
+ const loss = boardLoss();
1580
+ if (!loss) {
1581
+ journaledLoss = undefined;
1499
1582
  return;
1500
1583
  }
1501
- try {
1502
- watchSlot = await trackedDriver.narrator(repoRoot, watchCommand(runId), runId);
1584
+ const identity = `${loss.pane}:${loss.pid ?? ""}`;
1585
+ if (identity !== journaledLoss) {
1586
+ journaledLoss = identity;
1587
+ boardless = boardReopens >= MAX_BOARD_REOPENS;
1588
+ journal.append("watch-board-lost", undefined, { ...loss, ...(boardless ? { boardless: true } : {}) });
1589
+ if (boardless)
1590
+ return;
1503
1591
  }
1504
- catch (error) {
1505
- const reason = error instanceof Error ? error.message : String(error);
1506
- journal.append("watch-placement-failed", undefined, { error: reason });
1507
- console.error(`tickmarkr: narrator not opened: ${reason}`);
1592
+ const attempt = ++boardReopens;
1593
+ // The ghost pane is gone: drop it from this run's live set so no close is ever aimed at it.
1594
+ const ghost = watchSlot;
1595
+ if (ghost)
1596
+ liveSlots.delete(ghost);
1597
+ watchSlot = undefined;
1598
+ const reopened = await openBoard({ slot: ghost, pane: loss.pane });
1599
+ if (reopened.ok) {
1600
+ journal.append("watch-board-reopened", undefined, { pane: readWatchBoard(repoRoot, runId)?.pane ?? reopened.slot.id, attempt });
1601
+ return;
1508
1602
  }
1603
+ boardless = attempt >= MAX_BOARD_REOPENS;
1604
+ journal.append("watch-board-reopen-failed", undefined, { pane: loss.pane, attempt, error: reopened.error, ...(boardless ? { boardless: true } : {}) });
1605
+ console.error(`tickmarkr: board not reopened (attempt ${attempt}): ${reopened.error}`);
1509
1606
  };
1510
1607
  // OBS-829/OBS-854: one full-suite verdict round at a time in this run, and do not begin beside an
1511
1608
  // externally live suite attributable to this repository. The process scan catches nested scratch
@@ -1626,12 +1723,12 @@ export async function runDaemon(repoRoot, opts = {}) {
1626
1723
  ...(replayedExclusions.size > 0 ? { excludedChannels: [...replayedExclusions].sort() } : {}),
1627
1724
  ...(opts.retryFailed ? { retryFailed: true } : {}),
1628
1725
  });
1629
- await openBoard();
1726
+ await placeBoard();
1630
1727
  }
1631
1728
  else {
1632
1729
  baseRef = await gitHead(repoRoot);
1633
1730
  journal.append("baseline-start", undefined, { baseRef, commands, capacity: resolvedCapacity() });
1634
- await openBoard();
1731
+ await placeBoard();
1635
1732
  baselinePending = true;
1636
1733
  // Workers can run beside capture, but no gate may observe an absent or partial baseline.
1637
1734
  // Keep publication and warnings inside the same barrier as the suite's final verdict.
@@ -1767,7 +1864,20 @@ export async function runDaemon(repoRoot, opts = {}) {
1767
1864
  ];
1768
1865
  return `${identity} — release with ${commands.map((command) => `\`${command}\``).join(" or ")}`;
1769
1866
  };
1867
+ const namedPaths = (text) => [...text.matchAll(/(?:^|[\s`'"(])((?:[A-Za-z0-9_@.()[\]-]+\/)+[A-Za-z0-9_@.[\]-]+|[A-Za-z0-9_@-]+(?:\.[A-Za-z0-9_-]+)+)(?=$|[\s`'"),:;.!?])/g)]
1868
+ .map((match) => match[1].replace(/^\.\//, "").replace(/\.$/, ""))
1869
+ .filter((path) => !path.split("/").includes("..")
1870
+ && (/\.[A-Za-z][A-Za-z0-9_-]*$/.test(basename(path)) || existsSync(join(repoRoot, path))));
1770
1871
  const park = async (t, reason, kind, assignment, attempts, startMs, gateFails = 0, consults = 0, tokens, metered = 0, retryMode = "fresh", details = {}) => {
1872
+ // OBS-979: a worker refusal can identify the missing authoring scope even when gates
1873
+ // subsequently supply the park's disposition. Keep that actionable path on the park itself.
1874
+ const worker = journal.read().reverse().find((e) => e.taskId === t.id && e.event === "worker-result");
1875
+ if (t.files.length > 0 && worker?.data.ok === false && typeof worker.data.summary === "string") {
1876
+ const allowed = filesGlob(t.files);
1877
+ const paths = namedPaths(worker.data.summary).filter((path) => !allowed(path));
1878
+ if (paths.length)
1879
+ reason += ` — files[] repair hint: ${[...new Set(paths)].join(", ")}`;
1880
+ }
1771
1881
  graph = setStatus(graph, t.id, "human");
1772
1882
  saveGraph(repoRoot, graph);
1773
1883
  journal.append("task-human", t.id, { ...details, reason, kind });
@@ -1797,6 +1907,30 @@ export async function runDaemon(repoRoot, opts = {}) {
1797
1907
  const dispositionScopeRed = async (t, results, assignment, attempts, startMs, gateFails, consults, tokens, metered, retryMode) => {
1798
1908
  const scopeRed = results.find((g) => g.gate === "scope" && gateFailed(g));
1799
1909
  const verdict = scopeRed?.meta?.collateral;
1910
+ if (!verdict && t.files.length > 0) {
1911
+ const allowed = filesGlob(t.files);
1912
+ const worker = journal.read().reverse().find((e) => e.taskId === t.id && e.event === "worker-result");
1913
+ const refusal = worker?.data.ok === false && typeof worker.data.summary === "string"
1914
+ && /outside|out.of.scope|allowlist|scope expansion|not (?:in|own)|unowned/i.test(worker.data.summary)
1915
+ ? namedPaths(worker.data.summary).filter((path) => !allowed(path)) : [];
1916
+ const reds = results.filter(gateFailed);
1917
+ // A scope verdict already attributes actual out-of-scope EDITS using the collateral map.
1918
+ // Path-only diagnostics from the other gates name code that needs fixing, not an edit the
1919
+ // worker made. Do not replace the scope gate's stronger attribution with that inference.
1920
+ const paths = reds.filter((g) => g.gate !== "scope" || !g.meta?.collateral).flatMap((g) => namedPaths(g.details));
1921
+ if (refusal.length || (paths.length > 0 && paths.every((path) => !allowed(path)))) {
1922
+ journal.append("scope-authoring", t.id, {
1923
+ gate: refusal.length ? "worker" : reds[0]?.gate,
1924
+ paths: [...new Set(refusal.length ? refusal : paths)],
1925
+ repair: `files[] repair hint: ${[...new Set(refusal.length ? refusal : paths)].join(", ")}`,
1926
+ attempt: attempts + 1, chargeable: false, source: "diagnostic",
1927
+ });
1928
+ // OBS-547 (Leg-2 T3 material): unchargeable ⇒ metered 0, exactly as the predicted path below —
1929
+ // passing the physical count would write `meteredAttempts: 1` beside `attempts: 0`.
1930
+ await park(t, `authoring defect — files[] repair hint: ${[...new Set(refusal.length ? refusal : paths)].join(", ")}; current files[]: ${t.files.join(", ")}`, "authoring", assignment, attempts, startMs, gateFails, consults, tokens, 0, retryMode);
1931
+ return true;
1932
+ }
1933
+ }
1800
1934
  if (!verdict)
1801
1935
  return false;
1802
1936
  if (verdict.authoring) {
@@ -1915,12 +2049,28 @@ export async function runDaemon(repoRoot, opts = {}) {
1915
2049
  // not re-tried first (consult bans / prior failovers survive the release).
1916
2050
  const rs = resume.get(t.id);
1917
2051
  const contentDigest = taskContentDigest(t);
1918
- if (rs?.lastAssignment && rs.attempts > 0
2052
+ const previousDispatch = journal.read().reverse().find((e) => e.taskId === t.id && e.event === "task-dispatch");
2053
+ const recordedGraphPath = join(journal.dir, "graph.json");
2054
+ const recordedTask = existsSync(recordedGraphPath)
2055
+ ? JSON.parse(readFileSync(recordedGraphPath, "utf8")).tasks.find((task) => task.id === t.id)
2056
+ : undefined;
2057
+ const previousHints = previousDispatch && "routingHints" in previousDispatch.data
2058
+ ? previousDispatch.data.routingHints : recordedTask?.routingHints;
2059
+ const hintsChanged = !!rs && (!!recordedTask || !!previousDispatch && "routingHints" in previousDispatch.data)
2060
+ && (JSON.stringify(previousHints?.pin) !== JSON.stringify(t.routingHints?.pin)
2061
+ || previousHints?.floor !== t.routingHints?.floor);
2062
+ if (hintsChanged)
2063
+ journal.append("restore-rerouted", t.id, {
2064
+ from: rs.lastAssignment ? channelKey(rs.lastAssignment) : rs.tried.at(-1) ?? null,
2065
+ to: channelKey(assignment),
2066
+ reason: JSON.stringify(previousHints?.pin) !== JSON.stringify(t.routingHints?.pin) ? "pin changed" : "floor changed",
2067
+ });
2068
+ if (!hintsChanged && rs?.lastAssignment && rs.attempts > 0
1919
2069
  && channels.some((c) => channelKey(c) === channelKey(rs.lastAssignment))
1920
2070
  && !demotedChannels.has(channelKey(rs.lastAssignment))) {
1921
2071
  assignment = rs.lastAssignment; // restore the consult-chosen assignment (bypasses route()'s static re-pick)
1922
2072
  }
1923
- else if (rs && rs.tried.length) {
2073
+ else if (!hintsChanged && rs && rs.tried.length) {
1924
2074
  // trailing-reroute edge (kill between verdict and dispatch), a stale fleet, OR a fresh-budget
1925
2075
  // release (attempts 0 + non-empty tried): pick a failover over the replayed exclusions via the
1926
2076
  // EXISTING nextChannel `tried` parameter — zero router changes (D-03).
@@ -1931,6 +2081,14 @@ export async function runDaemon(repoRoot, opts = {}) {
1931
2081
  // assignment and proceed. Dispatching on a previously-tried channel beats deadlocking a resumed
1932
2082
  // run; a park-instead policy can come later if it ever bites.
1933
2083
  }
2084
+ const taskHistory = journal.read().filter((e) => e.taskId === t.id);
2085
+ const lastApproval = taskHistory.map((e) => e.event).lastIndexOf("task-approved");
2086
+ const priorClimb = taskHistory.slice(lastApproval + 1).reverse().find((e) => e.event === "tier-escalated");
2087
+ if (!hintsChanged && priorClimb && taskHistory.indexOf(priorClimb) > taskHistory.map((e) => e.event).lastIndexOf("task-dispatch")) {
2088
+ const target = channels.find((c) => channelKey(c) === priorClimb.data.to && !demotedChannels.has(channelKey(c)));
2089
+ if (target)
2090
+ assignment = { adapter: target.adapter, model: target.model, channel: target.channel, tier: target.tier };
2091
+ }
1934
2092
  // pre-kill invariant: tried always contains the current assignment. Spread, never alias the
1935
2093
  // journal-derived array (no hidden mutation of replayed state).
1936
2094
  const tried = rs?.tried.length ? [...rs.tried] : [channelKey(assignment)];
@@ -2026,6 +2184,7 @@ export async function runDaemon(repoRoot, opts = {}) {
2026
2184
  gate: g.gate, ...(unverdicted ? {} : { pass: g.pass }), details: g.details,
2027
2185
  ...(gateSubject ? { commit: gateSubject.commit, attempt: gateSubject.attempt } : {}),
2028
2186
  ...(gateSubject?.replayMeasurement ? { replayMeasurement: true } : {}),
2187
+ ...(gateSubject?.replayedFromAttempt !== undefined ? { replayedFromAttempt: gateSubject.replayedFromAttempt } : {}),
2029
2188
  ...(g.meta?.skipped === true || noVerdictReview ? { skipped: true } : {}),
2030
2189
  // T9: an infra-only exit is journaled AS one. The operator reading a red `test` row has to
2031
2190
  // be able to tell "the suite found a defect" from "the runner never ran", and the merge
@@ -2052,7 +2211,7 @@ export async function runDaemon(repoRoot, opts = {}) {
2052
2211
  ...(g.meta?.reapedGroup === true ? { reapedGroup: true } : {}),
2053
2212
  ...(g.gate === "review" && typeof g.meta?.reviewer === "string" ? {
2054
2213
  reviewer: g.meta.reviewer,
2055
- ...Object.fromEntries(["cause", "seatAuthoredBytes", "bytes", "rawPath", "briefPath", "timeoutMs", "unparseable", "noVerdict", "resolved", "reraised"]
2214
+ ...Object.fromEntries(["cause", "seatAuthoredBytes", "bytes", "rawPath", "briefPath", "timeoutMs", "unparseable", "noVerdict", "resolved", "reraised", "reviewerFloor", "reviewerFloorCause", "reviewerTier"]
2056
2215
  .filter((key) => g.meta?.[key] !== undefined).map((key) => [key, g.meta[key]])),
2057
2216
  ...(typeof g.meta.vendor === "string" ? { vendor: g.meta.vendor } : {}),
2058
2217
  ...(typeof g.meta.provider === "string" ? { provider: g.meta.provider } : {}),
@@ -2152,6 +2311,43 @@ export async function runDaemon(repoRoot, opts = {}) {
2152
2311
  ? `The operator UPHELD the reviewer's findings — address them without discarding landed work.\nreview: ${upheldFeedback}`
2153
2312
  : "");
2154
2313
  let ladderIdx = 0;
2314
+ const engagementRows = () => {
2315
+ const rows = journal.read().filter((e) => e.taskId === t.id);
2316
+ const approval = rows.map((e) => e.event).lastIndexOf("task-approved");
2317
+ return rows.slice(approval + 1);
2318
+ };
2319
+ let ownedGateFailures = engagementRows().filter((e) => e.event === "escalation" && e.data.workerAttributed === true).length;
2320
+ let climbProvenance = !hintsChanged && priorClimb
2321
+ ? `tier-escalated ${priorClimb.data.from} → ${priorClimb.data.to} (${priorClimb.data.cause})` : "";
2322
+ const climbSkips = new Set(!hintsChanged && priorClimb
2323
+ ? channels.filter((c) => c.tier === priorClimb.data.fromTier).map(channelKey) : []);
2324
+ const climb = async (cause, gate, attempt) => {
2325
+ if (cfg.routing.escalateTier === "off" || engagementRows().some((e) => e.event === "tier-escalated"))
2326
+ return;
2327
+ const pin = t.routingHints?.pin;
2328
+ if (pin) {
2329
+ await park(t, `pin ${pin.via}:${pin.model} held: ${cause}`, "gate-fail", assignment, attempt, startMs, gateFails, consults, tokens, metered, retryMode);
2330
+ return "parked";
2331
+ }
2332
+ const next = climbChannel(assignment, t, cfg, channels, tried, profile, demotedChannels);
2333
+ if (!next?.climbed)
2334
+ return;
2335
+ const from = channelKey(assignment);
2336
+ const to = channelKey(next);
2337
+ const poolBefore = channels.filter((c) => !tried.includes(channelKey(c)) && !demotedChannels.has(channelKey(c))).map(channelKey);
2338
+ journal.append("tier-escalated", t.id, {
2339
+ attempt, cause, gate: gate.gate, fingerprint: normalizeGateFailure(gate.details).slice(0, 500),
2340
+ from, to, fromTier: assignment.tier, toTier: next.tier, poolBefore,
2341
+ costDelta: marginalCostRank(channels.find((c) => channelKey(c) === to)) - marginalCostRank(channels.find((c) => channelKey(c) === from)),
2342
+ });
2343
+ for (const c of channels)
2344
+ if (c.tier === assignment.tier && channelKey(c) !== to)
2345
+ climbSkips.add(channelKey(c));
2346
+ climbProvenance = `tier-escalated ${from} → ${to} (${cause})`;
2347
+ assignment = { adapter: next.adapter, model: next.model, channel: next.channel, tier: next.tier };
2348
+ tried.push(to);
2349
+ return "climbed";
2350
+ };
2155
2351
  let modeFallbackNoted = false; // v1.2: journal the interactive→print fallback once per task, not per attempt
2156
2352
  let gateFails = 0; // TEL-02: incremented ONLY where feedback is built from failing gates — never derived from attempts (quota failovers bump attempts too, Pitfall 6)
2157
2353
  let consults = 0; // TEL-02: bumped in the runConsult wrapper so one counter covers all three trigger sites, across the attempt loop
@@ -2447,7 +2643,7 @@ export async function runDaemon(repoRoot, opts = {}) {
2447
2643
  }
2448
2644
  : undefined,
2449
2645
  excludeReviewers: badReviewers,
2450
- reviewHistory, demotedReviewers,
2646
+ reviewHistory, demotedReviewers, priorReviewers: taskReviewers(t.id),
2451
2647
  onGate: async (e) => {
2452
2648
  if (e.phase === "start") {
2453
2649
  notePhaseStart(e);
@@ -2723,8 +2919,17 @@ export async function runDaemon(repoRoot, opts = {}) {
2723
2919
  // ledger cannot tell a carried dispatch from an amnesiac one — the exact question a run that
2724
2920
  // spends two frontier attempts re-deriving a known defect has to be able to answer afterwards.
2725
2921
  journal.append("task-dispatch", t.id, {
2726
- assignment, attempt, provenance: dispatchProvenance(r.provenance), retryMode,
2727
- excludedChannels: [...demotedChannels].sort(),
2922
+ assignment, attempt, provenance: dispatchProvenance([
2923
+ channelKey(assignment) === channelKey(r.assignment) ? r.provenance
2924
+ : t.routingHints?.pin ? `pin ${t.routingHints.pin.via}:${t.routingHints.pin.model} not re-tried` : "ladder assignment",
2925
+ `dispatch ${channelKey(assignment)}`, climbProvenance,
2926
+ ].filter(Boolean).join(" · ")), retryMode,
2927
+ routingHints: t.routingHints ?? {},
2928
+ excludedChannels: [...new Set([...demotedChannels, ...tried, ...climbSkips])]
2929
+ .filter((key) => key !== channelKey(assignment)).sort(),
2930
+ exclusionReasons: Object.fromEntries([...new Set([...demotedChannels, ...tried, ...climbSkips])]
2931
+ .filter((key) => key !== channelKey(assignment))
2932
+ .map((key) => [key, demotedChannels.has(key) ? "demoted" : tried.includes(key) ? "already tried" : "tier climb skipped"])),
2728
2933
  ...(outstandingFindings.length > 0 ? { carriedFindings: outstandingFindings } : {}),
2729
2934
  ...(carriedConsultGuidance ? { carriedConsultGuidance } : {}),
2730
2935
  });
@@ -3878,6 +4083,8 @@ export async function runDaemon(repoRoot, opts = {}) {
3878
4083
  // (provider outage, quota banner, dead CLI) must still route; the harvest synthesis only
3879
4084
  // decides whether THIS attempt's worktree goes to gates, and when routing wins instead, the
3880
4085
  // commits survive via the existing commitsToCarry/cherryPickCommits carry-forward.
4086
+ if (workerFinished && !result.ok && await dispositionScopeRed(t, [], assignment, attempt, startMs, gateFails, consults, tokens, metered, retryMode))
4087
+ return;
3881
4088
  const preHarvestResult = result;
3882
4089
  // T2 (OBS-264): recognize committed no-trailer work BEFORE any no-trailer streak, provider,
3883
4090
  // quota or dead-channel routing. Gates never trusted the trailer, so this successful synthesis
@@ -4136,33 +4343,64 @@ export async function runDaemon(repoRoot, opts = {}) {
4136
4343
  const gated = await gitHead(wt);
4137
4344
  gateSubject = { commit: await gateCommitSubject(taskBase, gated, wt), attempt };
4138
4345
  await trackedDriver.project?.(t.id, "in-review");
4346
+ // Only the immediately preceding attempt can lend a red. Re-journal its results
4347
+ // as this attempt's verdicts so all existing disposition and fingerprint accounting
4348
+ // sees the replay, without buying another command or reviewer invocation.
4349
+ const taskEvents = journal.read().filter((e) => e.taskId === t.id);
4350
+ const previousRound = taskEvents.map((e) => e.event === "phase-start" && e.data.phase === "gates").lastIndexOf(true);
4351
+ const previousRows = retryMode === "repair"
4352
+ ? taskEvents.slice(previousRound + 1).filter((e) => e.event === "gate-result"
4353
+ && e.data.attempt === attempt - 1)
4354
+ : [];
4139
4355
  journal.phaseStart(t.id, "gates");
4140
- ({ results, commits } = await withSuiteWindow(t.id, t.gates.includes("test") && commands.test !== undefined, () => runReviewRecovery(t, {
4141
- carriedFindings: outstandingFindings,
4142
- worktree: wt, baseRef: taskBase, result, author: assignment,
4143
- commands, baseline, channels: pools.review, judgeChannels: pools.judge, adapters, cfg, artifactDir: journal.dir,
4144
- collateral: collateral.get(t.id) ?? [],
4145
- pipeline: "v185", selectTests: !testGateFailed,
4146
- via: cfg.visibility.llm === "pane"
4147
- ? {
4148
- driver: trackedDriver,
4149
- // D-07: judge/review panes self-clean when their verdict is read (keepLlm) — only "forever" keeps them.
4150
- keep: keepLlm,
4151
- onSlot: keepLlm ? (s) => keptSlots.push(s) : undefined,
4152
- // T2 ownership contract: canonical names (tickmarkr:<role>:<task>:0:<runId>) so reconcile
4153
- // owns judge/review panes; run-gates' -r1 retry suffix becomes attempt 1 in llm.ts.
4154
- // Same-name reuse across worker attempts is safe: panes self-clean when read (keepLlm),
4155
- // and herdr's DEFECT-01 reclaim covers a kept holdover under keepPanes:forever.
4156
- nameFor: (role) => formatOwnedName({ role, taskId: t.id, attempt: 0, runId }),
4157
- // role-tab label (SUP-01): role-first + task id, unique per concurrent instance within a run.
4158
- // Duplicate labels from a resumed run or operator-made tabs are accepted (per-process state).
4159
- labelFor: (role) => `${role.toUpperCase()} ${t.id}`,
4160
- }
4161
- : undefined,
4162
- excludeReviewers: badReviewers,
4163
- reviewHistory, demotedReviewers,
4164
- onGate,
4165
- })));
4356
+ const replay = previousRows.length > 0
4357
+ && previousRows.every((e) => e.data.commit === gateSubject.commit)
4358
+ && previousRows.some((e) => e.data.pass === false && e.data.skipped !== true);
4359
+ if (replay) {
4360
+ gateSubject.replayedFromAttempt = attempt - 1;
4361
+ results = previousRows.map(({ data }) => ({
4362
+ gate: String(data.gate), pass: data.pass === true, details: String(data.details),
4363
+ // The verdict is reused; its old timing is not a measurement of this attempt.
4364
+ meta: Object.fromEntries(Object.entries(data).filter(([key]) => !GATE_TELEMETRY_KEYS.includes(key) && key !== "capacity")),
4365
+ }));
4366
+ commits = await commitsAheadOf(taskBase, wt);
4367
+ for (const g of results) {
4368
+ journal.append("gate-replayed", t.id, {
4369
+ attempt, priorAttempt: attempt - 1, gate: g.gate, commit: gateSubject.commit,
4370
+ ...(g.meta?.skipped === true ? { skipped: true } : { pass: g.pass }),
4371
+ details: g.details,
4372
+ });
4373
+ journalGateResult(g);
4374
+ }
4375
+ }
4376
+ else {
4377
+ ({ results, commits } = await withSuiteWindow(t.id, t.gates.includes("test") && commands.test !== undefined, () => runReviewRecovery(t, {
4378
+ carriedFindings: outstandingFindings,
4379
+ worktree: wt, baseRef: taskBase, result, author: assignment,
4380
+ commands, baseline, channels: pools.review, judgeChannels: pools.judge, adapters, cfg, artifactDir: journal.dir,
4381
+ collateral: collateral.get(t.id) ?? [],
4382
+ pipeline: "v185", selectTests: !testGateFailed,
4383
+ via: cfg.visibility.llm === "pane"
4384
+ ? {
4385
+ driver: trackedDriver,
4386
+ // D-07: judge/review panes self-clean when their verdict is read (keepLlm) — only "forever" keeps them.
4387
+ keep: keepLlm,
4388
+ onSlot: keepLlm ? (s) => keptSlots.push(s) : undefined,
4389
+ // T2 ownership contract: canonical names (tickmarkr:<role>:<task>:0:<runId>) so reconcile
4390
+ // owns judge/review panes; run-gates' -r1 retry suffix becomes attempt 1 in llm.ts.
4391
+ // Same-name reuse across worker attempts is safe: panes self-clean when read (keepLlm),
4392
+ // and herdr's DEFECT-01 reclaim covers a kept holdover under keepPanes:forever.
4393
+ nameFor: (role) => formatOwnedName({ role, taskId: t.id, attempt: 0, runId }),
4394
+ // role-tab label (SUP-01): role-first + task id, unique per concurrent instance within a run.
4395
+ // Duplicate labels from a resumed run or operator-made tabs are accepted (per-process state).
4396
+ labelFor: (role) => `${role.toUpperCase()} ${t.id}`,
4397
+ }
4398
+ : undefined,
4399
+ excludeReviewers: badReviewers,
4400
+ reviewHistory, demotedReviewers, priorReviewers: taskReviewers(t.id),
4401
+ onGate,
4402
+ })));
4403
+ }
4166
4404
  results.forEach(classifySignalOnlyTest);
4167
4405
  graph = addEvidence(graph, t.id, { commits, gateResults: results, artifacts: [promptFile] });
4168
4406
  saveGraph(repoRoot, graph);
@@ -4251,6 +4489,9 @@ export async function runDaemon(repoRoot, opts = {}) {
4251
4489
  && !isDiffCapPark(reviewFail)
4252
4490
  && results.every((g) => gateSatisfied(g) || g.gate === "review");
4253
4491
  const failing = results.filter(gateFailed);
4492
+ const deterministic = failing.find(isDeterministicFailure);
4493
+ if (deterministic)
4494
+ ownedGateFailures++;
4254
4495
  // Decided before the cap, because the cap's question is whether the NEXT move would re-buy a
4255
4496
  // measurement already made — and a funded repair is one of the moves that would.
4256
4497
  const landed = await commitsAheadOf(taskBase, wt);
@@ -4292,13 +4533,18 @@ export async function runDaemon(repoRoot, opts = {}) {
4292
4533
  channel: channelKey(assignment), // the ban is bound to the channel that produced the repeat
4293
4534
  attempt: attempt + 1,
4294
4535
  });
4295
- await driver.notify(`tickmarkr ${runId}: ${t.id} ${repeated.gate} failed identically twice — consulting, identical retry banned`, { tier: "attention" });
4536
+ await driver.notify(`tickmarkr ${runId}: ${t.id} ${repeated.gate} failed identically twice — identical retry banned`, { tier: "attention" });
4296
4537
  // The cap takes the ladder's MOVE, never its accounting: this failure still spends the rung it
4297
4538
  // would have spent, so a task that cannot converge still reaches ladder exhaustion on exactly
4298
4539
  // the budget it always had and the cap can never hand a stuck task extra rounds.
4299
4540
  capStep = r.ladder[Math.min(ladderIdx++, r.ladder.length - 1)];
4300
- journal.append("escalation", t.id, { step: capStep, attempt: attempt + 1, fingerprintCap: true });
4541
+ journal.append("escalation", t.id, { step: capStep, attempt: attempt + 1, fingerprintCap: true, workerAttributed: !!deterministic });
4301
4542
  await driver.notify(`tickmarkr ${runId}: ${t.id} escalation: ${capStep}`, { tier: "attention" });
4543
+ const climbed = await climb("gate-fingerprint-cap", repeated, attempt + 1);
4544
+ if (climbed === "parked")
4545
+ return;
4546
+ if (climbed === "climbed")
4547
+ continue;
4302
4548
  const v = await runConsult("gate-fail-repeat", output, feedback, results);
4303
4549
  // Rule: a terminal cap consult vetoes a same-channel retry, but cannot veto an `escalate`
4304
4550
  // rung that already satisfies retry-same-banned by changing channel. Thus terminal+retry
@@ -4348,12 +4594,21 @@ export async function runDaemon(repoRoot, opts = {}) {
4348
4594
  : r.ladder[Math.min(ladderIdx++, r.ladder.length - 1)]);
4349
4595
  if (!capStep) { // a capped failure already journaled and announced the rung it spent
4350
4596
  journal.append("escalation", t.id, {
4351
- step, attempt: attempt + 1,
4597
+ step, attempt: attempt + 1, workerAttributed: !!deterministic,
4352
4598
  ...(reviewFixRetry && !repairExhausted ? { reviewFix: true } : {}),
4353
4599
  ...(repair ? { repair: repairHistory.length + 1, repairCharge: repairsDrawn + 1 } : {}),
4354
4600
  });
4355
4601
  await driver.notify(`tickmarkr ${runId}: ${t.id} escalation: ${step}`, { tier: "attention" });
4356
4602
  }
4603
+ const climbCause = repairExhausted ? "repair-exhausted"
4604
+ : deterministic && ownedGateFailures === 2 ? "gate-fail" : undefined;
4605
+ if (climbCause) {
4606
+ const climbed = await climb(climbCause, deterministic ?? failing[0], attempt + 1);
4607
+ if (climbed === "parked")
4608
+ return;
4609
+ if (climbed === "climbed")
4610
+ continue;
4611
+ }
4357
4612
  if (step === "retry")
4358
4613
  continue;
4359
4614
  if (step === "escalate") {
@@ -4377,6 +4632,23 @@ export async function runDaemon(repoRoot, opts = {}) {
4377
4632
  };
4378
4633
  taskLoopStarted = true;
4379
4634
  const inflight = new Map();
4635
+ // A settled worker can release a dependency or an approval in the same poll tick.
4636
+ // Audit the current graph at every empty-flight boundary, never a prior ready snapshot.
4637
+ // WB-1 rider: `end-condition-held` names a CLOSE the daemon refused, so only a close attempt journals
4638
+ // it — the post-race sweep still dispatches what it finds but says nothing.
4639
+ const holdEndCondition = (closing = true) => {
4640
+ sweepLiveApprovals();
4641
+ const freeSlots = Math.max(0, concurrency - inflight.size);
4642
+ const dispatchable = readyTasks(graph).filter((t) => !inflight.has(t.id));
4643
+ if (freeSlots === 0 || dispatchable.length === 0)
4644
+ return false;
4645
+ if (closing) {
4646
+ for (const task of dispatchable) {
4647
+ journal.append("end-condition-held", task.id, { deps: task.deps, freeSlots });
4648
+ }
4649
+ }
4650
+ return true;
4651
+ };
4380
4652
  closeLoop: while (true) {
4381
4653
  let approvalDeadline;
4382
4654
  while (true) {
@@ -4384,6 +4656,7 @@ export async function runDaemon(repoRoot, opts = {}) {
4384
4656
  // must still stop the run before it can dispatch more work or write run-end.
4385
4657
  if (termSignal)
4386
4658
  throw new Error(`terminated by ${termSignal}`);
4659
+ await watchBoard();
4387
4660
  sweepLiveApprovals();
4388
4661
  const ready = readyTasks(graph)
4389
4662
  .filter((t) => !inflight.has(t.id))
@@ -4412,7 +4685,7 @@ export async function runDaemon(repoRoot, opts = {}) {
4412
4685
  .filter((t) => t.status === "pending").every(behindPark);
4413
4686
  if (onlyParks) {
4414
4687
  if (approvalDeadline === undefined) {
4415
- const windowMs = opts.approvalWindowMs ?? APPROVAL_WINDOW_MS;
4688
+ const windowMs = opts.approvalWindowMs ?? approvalWindowMs;
4416
4689
  approvalDeadline = Date.now() + windowMs;
4417
4690
  journal.append("approval-window-start", undefined, { windowMs, parked: [...parked] });
4418
4691
  // The narrator may itself append a decision at this boundary.
@@ -4428,16 +4701,23 @@ export async function runDaemon(repoRoot, opts = {}) {
4428
4701
  }
4429
4702
  journal.append("approval-window-expired", undefined, { parked: [...parked] });
4430
4703
  }
4704
+ if (holdEndCondition()) {
4705
+ approvalDeadline = undefined;
4706
+ continue;
4707
+ }
4431
4708
  break;
4432
4709
  }
4433
4710
  approvalDeadline = undefined;
4434
4711
  const waiters = [...inflight.values(), aborted];
4435
4712
  // A free slot is itself a scheduling boundary: poll the append-only approval stream instead of
4436
4713
  // sleeping until an unrelated long-running task settles.
4437
- if (inflight.size < concurrency) {
4714
+ // An open board is polled on the same cadence, so a dead cockpit is noticed mid-task.
4715
+ if (inflight.size < concurrency || boardOpened) {
4438
4716
  waiters.push(new Promise((wake) => setTimeout(wake, APPROVAL_POLL_MS)));
4439
4717
  }
4440
4718
  await Promise.race(waiters); // aborted rejects on termination — unwinds the run
4719
+ if (inflight.size === 0)
4720
+ holdEndCondition(false);
4441
4721
  }
4442
4722
  // D-07: the sweep now closes only what's LEFT in keptSlots — done-closed worker slots were removed
4443
4723
  // (no double-close) and self-cleaned LLM/consult panes were never added under keepLlm:false. This
@@ -4515,12 +4795,11 @@ export async function runDaemon(repoRoot, opts = {}) {
4515
4795
  const approvalSerialization = await acquireApprovalSerialization(repoRoot, runId);
4516
4796
  releaseApprovalSerialization = approvalSerialization.release;
4517
4797
  // Close the last poll-to-run-end race while holding the same serializer as approve.
4518
- sweepLiveApprovals();
4519
- if (readyTasks(graph).length) {
4798
+ if (holdEndCondition()) {
4520
4799
  releaseApprovalSerialization();
4521
4800
  releaseApprovalSerialization = undefined;
4522
- if (summary.tipVerify)
4523
- journal.append("tip-verify-cancelled", undefined, { reason: "approval", lastMergedTask });
4801
+ // Verification has settled: retain its verdict. The cache key will require a
4802
+ // new battery if dispatch actually moves the tip or changes its commands.
4524
4803
  continue closeLoop;
4525
4804
  }
4526
4805
  const outstanding = outstandingApprovals(journal.read());