tickmarkr 2.6.0 → 2.6.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (47) hide show
  1. package/dist/adapters/prompt.d.ts +2 -1
  2. package/dist/adapters/prompt.js +10 -0
  3. package/dist/adapters/types.d.ts +1 -0
  4. package/dist/adapters/types.js +10 -0
  5. package/dist/cli/commands/fleet.js +26 -1
  6. package/dist/cli/commands/plan.js +7 -3
  7. package/dist/cli/commands/status.js +13 -3
  8. package/dist/cli/help.d.ts +2 -0
  9. package/dist/cli/help.js +2 -0
  10. package/dist/drivers/orca.js +6 -1
  11. package/dist/gates/cache.d.ts +3 -1
  12. package/dist/gates/cache.js +10 -3
  13. package/dist/gates/llm.js +4 -1
  14. package/dist/gates/review.js +7 -11
  15. package/dist/gates/run-gates.d.ts +4 -0
  16. package/dist/gates/run-gates.js +7 -4
  17. package/dist/gates/test-manifest.d.ts +12 -0
  18. package/dist/gates/test-manifest.js +26 -6
  19. package/dist/graph/graph.d.ts +6 -2
  20. package/dist/graph/graph.js +15 -4
  21. package/dist/route/role-pick.d.ts +16 -0
  22. package/dist/route/role-pick.js +15 -0
  23. package/dist/route/router.d.ts +14 -0
  24. package/dist/route/router.js +9 -1
  25. package/dist/run/consult.js +5 -9
  26. package/dist/run/daemon.d.ts +9 -0
  27. package/dist/run/daemon.js +395 -59
  28. package/dist/run/git.d.ts +36 -1
  29. package/dist/run/git.js +89 -5
  30. package/dist/run/host-health.d.ts +20 -0
  31. package/dist/run/host-health.js +64 -0
  32. package/dist/run/journal.js +9 -0
  33. package/dist/run/operator-state.d.ts +24 -2
  34. package/dist/run/operator-state.js +41 -5
  35. package/dist/run/stall.d.ts +38 -2
  36. package/dist/run/stall.js +276 -6
  37. package/dist/tui/cockpit/board.d.ts +1 -1
  38. package/dist/tui/cockpit/board.js +27 -19
  39. package/dist/tui/cockpit/derive.d.ts +2 -0
  40. package/dist/tui/cockpit/derive.js +4 -0
  41. package/dist/tui/cockpit/live-store.d.ts +1 -0
  42. package/dist/tui/cockpit/live-store.js +31 -8
  43. package/dist/tui/cockpit/run-cockpit.js +2 -1
  44. package/dist/tui/cockpit/run-view.d.ts +2 -4
  45. package/dist/tui/cockpit/run-view.js +9 -8
  46. package/package.json +1 -1
  47. package/skills/tickmarkr-overseer/SKILL.md +43 -6
@@ -1,3 +1,5 @@
1
+ import { HOST_PROBE_SAMPLE_MS, HOST_PROBE_SAMPLES, HostDegradedError, hostDegraded, observeHost } from "./host-health.js";
2
+ import { VITEST_CACHE_ENV, worktreeVitestCache } from "../gates/test-manifest.js";
1
3
  import { COMMAND_LEASE_TOKEN_ENV, commandLeaseEnvironment, CommandLeases, currentCommandLeaseToken, isRunnerCommand, runWithCommandLease, withCommandLease } from "./lease.js";
2
4
  import { execFileSync, spawn } from "node:child_process";
3
5
  import { createHash, randomBytes } from "node:crypto";
@@ -8,9 +10,9 @@ import { tmpdir } from "node:os";
8
10
  import { basename, dirname, isAbsolute, join, posix, relative, resolve, sep } from "node:path";
9
11
  import { fileURLToPath } from "node:url";
10
12
  import { stringify } from "yaml";
11
- import { classifyDeadChannel, NO_TRAILER_SUMMARY, trailerPattern, UNPARSEABLE_TRAILER_SUMMARY, writePrompt } from "../adapters/prompt.js";
13
+ import { classifyDeadChannel, classifyTransientCapacity, NO_TRAILER_SUMMARY, trailerPattern, UNPARSEABLE_TRAILER_SUMMARY, writePrompt } from "../adapters/prompt.js";
12
14
  import { allAdapters, getAdapter, probeAll, readDoctor, rolePools } from "../adapters/registry.js";
13
- import { SettledTrailerTracker, addUsage, channelKey, matchesInputBox, matchesTrustDialog, QUOTA_RE } from "../adapters/types.js";
15
+ import { SettledTrailerTracker, addUsage, CAPACITY_RE, channelKey, matchesInputBox, matchesTrustDialog, QUOTA_RE } from "../adapters/types.js";
14
16
  import { bannerShell, paneDispatchCommand } from "../brand.js";
15
17
  import { collateralHits } from "../compile/collateral.js";
16
18
  import { ExecutionPolicySchema, DEFAULT_DIFF_CAP, globalConfigDir, loadConfigWithMode, readOverlayFile, repoOverlayPath, } from "../config/config.js";
@@ -20,8 +22,9 @@ import { herdrSealShellPrefix, MAX_BUF, SubprocessDriver } from "../drivers/subp
20
22
  import { formatOwnedName } from "../drivers/types.js";
21
23
  import { captureBaseline, detectGateCommands, detectVacuousOracles } from "../gates/baseline.js";
22
24
  import { runGates } from "../gates/run-gates.js";
25
+ import { isInfraResult } from "../gates/cache.js";
23
26
  import { filesGlob } from "../graph/files-glob.js";
24
- import { addEvidence, attributeBlocked, blockedTasks, getTask, graphDefinitionHash, loadGraph, pendingTasks, readyTasks, saveGraph, setStatus, taskContentDigest, tickmarkrDir } from "../graph/graph.js";
27
+ import { addEvidence, attributeBlocked, batteryPriority, blockedTasks, getTask, graphDefinitionHash, loadGraph, pendingTasks, readyTasks, saveGraph, setStatus, taskContentDigest, tickmarkrDir } from "../graph/graph.js";
25
28
  import { GATE_NAMES } from "../graph/schema.js";
26
29
  import { distFingerprint } from "../cli/commands/version.js";
27
30
  import { augmentRetryBrief, consult, renderRetryGuidance } from "./consult.js";
@@ -29,7 +32,7 @@ import { executionSignal, remainingExecutionMs, withExecutionBudget, withoutExec
29
32
  import { repairSelectionDecision } from "./repair-selection.js";
30
33
  import { failureDisposition, reserveInfrastructureRetry } from "./recovery.js";
31
34
  import { runEnvironment } from "./environment.js";
32
- import { cleanupRunWorktrees, deriveForkCap, FORK_CAP_ENV, gitHead, linkNodeModules, npmDependencyInstallCommand, npmDependencyManifestChanged, preserveWorktree, resolvedCapacity, runWithForkBudget, runWithVerificationBudget, sameCapacity, sameVerification, sh, shGit, SUITE_PARENT_ENV, verificationProtocol, WORKTREE_LAYOUT_CONTRACT, worktreePath } from "./git.js";
35
+ import { cleanupRunWorktrees, deriveForkCap, FORK_CAP_ENV, gitHead, linkNodeModules, npmDependencyInstallCommand, npmDependencyManifestChanged, PRESERVE_COMMIT_SUBJECT, PRESERVE_PRODUCER_TRAILER, preserveWorktree, producerFields, resolvedCapacity, runWithForkBudget, runWithVerificationBudget, sameCapacity, sameVerification, sh, shGit, SUITE_PARENT_ENV, verificationProtocol, WORKTREE_LAYOUT_CONTRACT, worktreePath } from "./git.js";
33
36
  import { runInteractiveSeed } from "./interactive-seed.js";
34
37
  import { classifyRepairDisposition, resolveScopeHints } from "./repair-disposition.js";
35
38
  import { applyScopeAmendments, activeRetryBan, classifyTaskFailure, classifyWorkerResultCause, deferredReviewFindings, engagementComparable, formatPriorFindingEvidence, GATE_FINGERPRINT_CAP, GATE_SATISFIED_RELEASE, identicalGateFailures, isDeferredFinding, journaledFailureBrief, Journal, loadRoutingProfile, newRunId, normalizeGateFailure, outstandingConsultGuidance, outstandingReviewFindings, pendingApprovalActions, pendingRechecks, pendingRepairFindings, phaseForGate, readPriorRunEvidence, recordedTaskFailureKind, RECHECK_RELEASE, renderStructuredReviewFinding, repairReachSinceApproval, repairsSinceApproval, reviewRoundsSinceApproval, runHasEnded, structuredFindings, upheldFeedbackByTask } from "./journal.js";
@@ -199,7 +202,8 @@ const RECHECK_ENACTMENT = "recheck-battery";
199
202
  const GATE_SATISFIED_ENACTMENT = "worktree-recreation";
200
203
  // Older daemons enacted rechecks through worker-launch, before recheck-battery existed.
201
204
  // Preserve that consumption at every scheduling read without changing continuing permission.
202
- function pendingDaemonApprovalActions(events) {
205
+ // OBS-1158: exported so plan folds battery priority through the SAME consumption-aware seam.
206
+ export function pendingDaemonApprovalActions(events) {
203
207
  const actions = pendingApprovalActions(events);
204
208
  const rechecks = pendingRechecks(events);
205
209
  for (const [id, action] of actions) {
@@ -310,6 +314,17 @@ function classifySignalOnlyTest(g) {
310
314
  return;
311
315
  g.meta = { ...g.meta, classification: "infra", infra: true, retryable: false, kind: "signal-exit" };
312
316
  }
317
+ /** OBS-1106: ONE infrastructure predicate for classification, persistence and repair admission. A red
318
+ * that carries an infra fingerprint or classification without `meta.infra` (a legacy journal row, an
319
+ * oracle that named only its classification) is still a non-verdict about the work: it is normalized
320
+ * here BEFORE its journal row and before any park/repair seam reads it, so those seams can key on the
321
+ * same `isInfraResult` the gate cache keys on and never on one metadata field alone. */
322
+ function classifyInfraResult(g) {
323
+ classifySignalOnlyTest(g);
324
+ if (g.pass || g.meta?.infra === true || !isInfraResult(g))
325
+ return;
326
+ g.meta = { ...g.meta, classification: "infra", infra: true };
327
+ }
313
328
  // v1.85 T3: the gates whose failure IS a deterministic measurement — a machine re-ran a command over a
314
329
  // tree and printed the same bytes. Those are the failures the fingerprint cap governs (the ruling names
315
330
  // it a "deterministic-gate" cap): a third identical answer to a question already answered twice is the
@@ -455,6 +470,15 @@ export const setApprovalWindowForTests = (ms) => { approvalWindowMs = ms; };
455
470
  export const resetApprovalWindowForTests = () => { approvalWindowMs = DEFAULT_APPROVAL_WINDOW_MS; };
456
471
  const PROVIDER_DEATH_REQUEUE_CAP = 2; // v1.46 T1: requeue same assignment twice, then fall through to the normal ladder
457
472
  const PROVIDER_DEATH_BACKOFF_MS = 500; // short backoff before provider-death requeue
473
+ // OBS-1161: transient capacity ("Selected model is at capacity") — bounded same-seat requeues with a
474
+ // real wait between them, THEN a same-floor failover; never a demotion. The budget is per task and
475
+ // seat, counted from the journal's own capacity-requeue rows so a resume continues it, never restarts it.
476
+ const CAPACITY_REQUEUE_CAP = 2;
477
+ const CAPACITY_BACKOFF_MS = 60_000;
478
+ let capacityBackoffMs = CAPACITY_BACKOFF_MS;
479
+ /** Test seam — shrink the capacity backoff without minute-long sleeps. */
480
+ export function setCapacityBackoffMsForTests(ms) { capacityBackoffMs = ms; }
481
+ export function resetCapacityBackoffMsForTests() { capacityBackoffMs = CAPACITY_BACKOFF_MS; }
458
482
  const NO_TRAILER_DEMOTION_STREAK = 2; // OBS-57: consecutive no-trailer windows demote a channel for the rest of the run
459
483
  // OBS-117 (v1.71 T6): a worker pane that never prints a byte by T+60s after dispatch is a dead
460
484
  // channel — don't burn the full stall window waiting for a silent launch failure. Checked on the
@@ -1488,6 +1512,32 @@ async function cherryPickCommits(wt, commits) {
1488
1512
  }
1489
1513
  return carried;
1490
1514
  }
1515
+ const PRESERVED_REF_PREFIX = "refs/tickmarkr/preserved/";
1516
+ const preservedRefOf = (data) => [data.ref, data.preservedRef].find((v) => typeof v === "string" && v.startsWith(PRESERVED_REF_PREFIX));
1517
+ /** The attempt whose worker last launched into the task's current checkout; "unknown" once a
1518
+ * recreation replaced that tree without a new launch (recheck, gate-only restore) or when no
1519
+ * dispatch assignment is on record. Never the newest dispatch by itself. */
1520
+ export function knownProducer(events, taskId) {
1521
+ let dispatched = "unknown";
1522
+ let producer = "unknown";
1523
+ for (const row of events) {
1524
+ if (row.taskId !== taskId)
1525
+ continue;
1526
+ if (row.event === "task-dispatch") {
1527
+ const a = row.data.assignment;
1528
+ dispatched = typeof a?.adapter === "string" && typeof a?.model === "string"
1529
+ ? { channel: `${a.adapter}:${a.model}`, attempt: typeof row.data.attempt === "number" ? row.data.attempt : 0 } : "unknown";
1530
+ }
1531
+ else if (row.event === "worker-launch")
1532
+ producer = dispatched;
1533
+ else if (row.event === "worktree-recreation")
1534
+ producer = "unknown";
1535
+ }
1536
+ return producer;
1537
+ }
1538
+ /** Distinct author channels of the subject for task-done/status/board rows: "unknown" replaces every
1539
+ * unresolvable owner (legacy unattributed preservation, dispatch without assignment). */
1540
+ const mergedAuthors = (authors) => [...new Set(authors.map((a) => a.startsWith("unknown author") ? "unknown" : a))].sort();
1491
1541
  /** Fold lifetime dispatch/carry evidence, independent of attempt budgets and routing exclusions.
1492
1542
  * Recreation rows name SOURCE hashes, so ownership is joined by stable patch identity. Each
1493
1543
  * attempt owns only what the next carry (or current subject) adds beyond its own incoming set.
@@ -1517,12 +1567,15 @@ async function subjectAuthors(events, taskId, wt, base) {
1517
1567
  let previous;
1518
1568
  let awaitingCarry = false;
1519
1569
  const owners = new Map();
1570
+ // Preserved engine commits carry their producing attempt on the row; a row without one is legacy
1571
+ // and stays explicitly unknown rather than inheriting the seat that later carried the patch.
1572
+ const preservedOwner = new Map();
1520
1573
  const attribute = (ids, attempt) => {
1521
1574
  for (const id of ids) {
1522
1575
  if (attempt?.incoming.has(id))
1523
1576
  continue;
1524
1577
  const authors = owners.get(id) ?? new Set();
1525
- authors.add(attempt?.author ?? "unknown author (missing task-dispatch assignment)");
1578
+ authors.add(preservedOwner.get(id) ?? attempt?.author ?? "unknown author (missing task-dispatch assignment)");
1526
1579
  owners.set(id, authors);
1527
1580
  }
1528
1581
  };
@@ -1530,6 +1583,24 @@ async function subjectAuthors(events, taskId, wt, base) {
1530
1583
  for (const row of events) {
1531
1584
  if (row.taskId !== taskId)
1532
1585
  continue;
1586
+ const preserved = preservedRefOf(row.data);
1587
+ if (preserved) {
1588
+ // Only an engine preserve commit is owned by its row; a ref naming a worker's own commit keeps
1589
+ // that commit's dispatch attribution.
1590
+ const shown = await shGit(`git show -s ${shq(`--format=%H%n%s%n%(trailers:key=${PRESERVE_PRODUCER_TRAILER},valueonly)`)} ${shq(`${preserved}^{commit}`)}`, wt);
1591
+ const [commit, subject, trailer] = shown.stdout.trim().split("\n");
1592
+ if (shown.code === 0 && commit && subject === PRESERVE_COMMIT_SUBJECT) {
1593
+ // A row that merely mentions the ref (a park naming it) defers to the commit's own trailer;
1594
+ // only a commit with neither is legacy. A known owner is never downgraded by a later mention.
1595
+ const owner = typeof row.data.producer === "string" ? row.data.producer
1596
+ : trailer?.trim().replace(/ attempt \d+$/, "") || "unknown author (legacy unattributed preservation)";
1597
+ for (const id of await patches([commit])) {
1598
+ const prior = preservedOwner.get(id);
1599
+ if (prior === undefined || prior === "unknown" || prior.startsWith("unknown author"))
1600
+ preservedOwner.set(id, owner);
1601
+ }
1602
+ }
1603
+ }
1533
1604
  if (row.event === "task-dispatch") {
1534
1605
  previous = current;
1535
1606
  const a = row.data.assignment;
@@ -1706,6 +1777,8 @@ export async function runDaemon(repoRoot, opts = {}) {
1706
1777
  let fatalPhase = "setup";
1707
1778
  let deliberateTermination = false;
1708
1779
  const fatalStop = new AbortController();
1780
+ const hostStop = new AbortController();
1781
+ const hostChecks = new Set();
1709
1782
  const inflight = new Map();
1710
1783
  let retireFatalSlots;
1711
1784
  let branch = "";
@@ -1802,21 +1875,50 @@ export async function runDaemon(repoRoot, opts = {}) {
1802
1875
  if (owner) {
1803
1876
  const processGroup = readOwnedProcessGroup(owner.groupFile);
1804
1877
  let survivors;
1878
+ const strays = [];
1879
+ let session;
1805
1880
  try {
1806
1881
  const shared = processGroup !== undefined && [...workerOwners].some(([other, otherOwner]) => other !== slot && liveSlots.has(other) && readOwnedProcessGroup(otherOwner.groupFile) === processGroup);
1807
1882
  if (shared)
1808
1883
  throw new Error(`worker group ${processGroup} is shared with another live attempt`);
1809
- survivors = await reapOwnedProcessGroup(processGroup, slot.cwd);
1884
+ try {
1885
+ session = readFileSync(`${owner.groupFile}.session`, "utf8").trim() || undefined;
1886
+ }
1887
+ catch { /* pre-launch or older driver */ }
1888
+ let parent;
1889
+ try {
1890
+ const row = /^\s*(\d+)\s+(.+)$/.exec(readFileSync(`${owner.groupFile}.parent`, "utf8").trim());
1891
+ if (row)
1892
+ parent = { pid: Number(row[1]), startedAt: row[2].trim().replace(/\s+/g, " ") };
1893
+ }
1894
+ catch { /* a live dispatch root can still prove its parent */ }
1895
+ const others = [...workerOwners].filter(([other]) => other !== slot && liveSlots.has(other));
1896
+ survivors = await reapOwnedProcessGroup(processGroup, slot.cwd, {
1897
+ marker: owner.marker, session, parent, identities: owner.identities, descendants: owner.descendants, strays,
1898
+ excludedGroups: others.flatMap(([, other]) => {
1899
+ const group = readOwnedProcessGroup(other.groupFile);
1900
+ return group === undefined ? [] : [group];
1901
+ }),
1902
+ excludedWorktrees: others.map(([other]) => other.cwd).filter((cwd) => cwd !== slot.cwd),
1903
+ });
1810
1904
  }
1811
1905
  catch (error) {
1812
1906
  journal.append("worker-process-reaped", owner.taskId, { slot: slot.name, attempt: owner.attempt,
1813
- processGroup: processGroup ?? null, survivors: null, error: String(error) });
1907
+ processGroup: processGroup ?? null, strays, survivors: null, error: String(error) });
1908
+ workerOwners.delete(slot); // one reap row per attempt: a later close never re-sweeps
1814
1909
  throw error;
1815
1910
  }
1816
- reapReports.set(slot, { processGroup: processGroup ?? null, survivors });
1911
+ reapReports.set(slot, { processGroup: processGroup ?? null, strays, survivors });
1817
1912
  journal.append("worker-process-reaped", owner.taskId, {
1818
- slot: slot.name, attempt: owner.attempt, processGroup: processGroup ?? null, survivors,
1913
+ slot: slot.name, attempt: owner.attempt, processGroup: processGroup ?? null, strays, survivors,
1819
1914
  });
1915
+ // Every outcome retires the claim: one reap row per attempt, whatever the verdict.
1916
+ workerOwners.delete(slot);
1917
+ // Pre-launch and in-process drivers have no OS dispatch claim; their close owns retirement.
1918
+ // Once any dispatch ownership exists, an unreadable sweep must block the next gate.
1919
+ if (survivors === null && (processGroup !== undefined || session !== undefined || owner.descendants.size > 0)) {
1920
+ throw new Error(`worker group ${processGroup} cleanup unknown`);
1921
+ }
1820
1922
  if (survivors && survivors.length > 0)
1821
1923
  throw new Error(`worker group ${processGroup} survivors: ${survivors.join(", ")}`);
1822
1924
  }
@@ -1915,6 +2017,8 @@ export async function runDaemon(repoRoot, opts = {}) {
1915
2017
  // new attempt while this reaper is still closing the old ones.
1916
2018
  const termination = new Error(`terminated by ${sig}`);
1917
2019
  abortRun(termination);
2020
+ hostStop.abort(termination);
2021
+ await Promise.allSettled(hostChecks);
1918
2022
  if (activeTipVerify) {
1919
2023
  activeTipVerify.controller.abort(termination);
1920
2024
  await activeTipVerify.settled;
@@ -2107,32 +2211,143 @@ export async function runDaemon(repoRoot, opts = {}) {
2107
2211
  journal.append("watch-board-reopen-failed", undefined, { pane: loss.pane, attempt, error: reopened.error, ...(boardless ? { boardless: true } : {}) });
2108
2212
  console.error(`tickmarkr: board not reopened (attempt ${attempt}): ${reopened.error}`);
2109
2213
  };
2110
- // Context carries attribution through gates and remote inference; only shell commands acquire.
2111
- const commandLeases = new CommandLeases();
2112
- const withCommandContext = (taskId, run, signal = executionSignal()) => runWithCommandLease((_command, execute) => commandLeases.run(async () => {
2113
- let lastCount = -1;
2214
+ // Reference rows are the durable source of truth; old journals establish one on first resume.
2215
+ const priorReference = opts.resume ? [...journal.read()].reverse().find(e => ["host-reference", "host-reference-reset"].includes(e.event)
2216
+ && typeof e.data.medianMs === "number" && Number.isFinite(e.data.medianMs) && e.data.medianMs > 0) : undefined;
2217
+ let hostReferenceMs = priorReference?.data.medianMs;
2218
+ const hostSignal = (signal) => AbortSignal.any([hostStop.signal, fatalStop.signal, signal, executionSignal()].filter((s) => !!s));
2219
+ const trackHost = (promise) => {
2220
+ hostChecks.add(promise);
2221
+ void promise.finally(() => hostChecks.delete(promise)).catch(() => { });
2222
+ return promise;
2223
+ };
2224
+ const recordReference = (observation) => {
2225
+ if (observation.medianMs === null)
2226
+ return;
2227
+ journal.append("host-reference", undefined, { ...observation });
2228
+ hostReferenceMs = observation.medianMs;
2229
+ };
2230
+ // Health and occupancy share a deadline, but only healthy occupancy gets the bounded fallback.
2231
+ const admitHost = async (taskId, signal, initial, resuming = false, gate, onWait) => {
2114
2232
  const startedAt = Date.now();
2233
+ let lastCount = -1;
2234
+ let observation = initial;
2235
+ let first = true;
2236
+ let resetEligible = resuming && hostReferenceMs !== undefined;
2115
2237
  for (;;) {
2116
- signal?.throwIfAborted();
2117
- fatalStop.signal.throwIfAborted();
2118
- executionSignal()?.throwIfAborted();
2238
+ signal.throwIfAborted();
2119
2239
  const count = await liveSuiteCount(repoRoot);
2120
- if (count === 0)
2121
- break;
2122
- if (count !== lastCount)
2123
- journal.append("suite-wait", taskId, { count });
2240
+ signal.throwIfAborted();
2241
+ const remaining = suiteWaitCeilingMs - (Date.now() - startedAt);
2242
+ // Reserve a full bounded batch. A partial last batch would manufacture an unreadable
2243
+ // host at an otherwise healthy occupancy deadline. Retain the latest complete observation.
2244
+ const probeBudget = HOST_PROBE_SAMPLE_MS * HOST_PROBE_SAMPLES;
2245
+ if (!observation || (!first && remaining >= probeBudget)) {
2246
+ observation = await observeHost(signal, remaining > 0 ? Math.min(probeBudget, remaining) : undefined);
2247
+ journal.append("host-observation", taskId, { ...observation, referenceMs: hostReferenceMs ?? null, resuming });
2248
+ }
2249
+ first = false;
2250
+ if (hostReferenceMs === undefined && observation.medianMs !== null)
2251
+ recordReference(observation);
2252
+ const degraded = hostDegraded(observation, hostReferenceMs);
2253
+ resetEligible &&= count === 0 && observation.medianMs !== null && degraded;
2254
+ if (!degraded && (count === 0 || resuming))
2255
+ return count;
2256
+ onWait?.();
2257
+ if (degraded)
2258
+ journal.append("host-degraded", taskId, {
2259
+ ...observation, referenceMs: hostReferenceMs ?? null, ...(gate ? { gate } : {}), count, waitedMs: Date.now() - startedAt,
2260
+ });
2261
+ else if (count !== lastCount)
2262
+ journal.append("suite-wait", taskId, { count, ...(gate ? { gate } : {}) });
2124
2263
  lastCount = count;
2125
2264
  if (Date.now() - startedAt >= suiteWaitCeilingMs) {
2265
+ if (degraded) {
2266
+ if (resetEligible) {
2267
+ // Append BOTH medians before adoption. An unreadable sample or live suite vetoes reset.
2268
+ journal.append("host-reference-reset", undefined, {
2269
+ referenceMs: hostReferenceMs, medianMs: observation.medianMs, waitedMs: Date.now() - startedAt,
2270
+ });
2271
+ hostReferenceMs = observation.medianMs;
2272
+ return count;
2273
+ }
2274
+ throw new HostDegradedError("host latency remained degraded or unreadable through suite-wait deadline");
2275
+ }
2126
2276
  journal.append("suite-wait-ceiling", taskId, { count, waitedMs: Date.now() - startedAt });
2127
- journal.append("suite-budget", taskId, {
2128
- count, occupancyCap: occupancyCapacity.forkCap, conservativeCap: conservativeCapacity.forkCap,
2129
- });
2130
- return await runWithVerificationBudget(conservativeCapacity, execute);
2277
+ return count;
2131
2278
  }
2132
- await new Promise((wake) => setTimeout(wake, SUITE_POLL_MS));
2279
+ await new Promise((resolve, reject) => {
2280
+ const abort = () => { clearTimeout(timer); reject(signal.reason); };
2281
+ const timer = setTimeout(() => { signal.removeEventListener("abort", abort); resolve(); }, Math.min(SUITE_POLL_MS, Math.max(0, suiteWaitCeilingMs - (Date.now() - startedAt))));
2282
+ signal.addEventListener("abort", abort, { once: true });
2283
+ if (signal.aborted)
2284
+ abort();
2285
+ });
2133
2286
  }
2134
- return await execute();
2135
- }, (count) => journal.append("suite-wait", taskId, { count }), SUITE_POLL_MS, signal), run);
2287
+ };
2288
+ const initializeHost = () => trackHost((async () => {
2289
+ const signal = hostSignal();
2290
+ const observation = await observeHost(signal);
2291
+ journal.append("host-observation", undefined, { ...observation, referenceMs: hostReferenceMs ?? null, resuming: !!opts.resume });
2292
+ if (hostReferenceMs === undefined)
2293
+ recordReference(observation);
2294
+ if (opts.resume || observation.medianMs === null)
2295
+ await admitHost(undefined, signal, observation, !!opts.resume);
2296
+ })());
2297
+ // Context carries attribution through gates and remote inference; only shell commands acquire.
2298
+ const commandLeases = new CommandLeases();
2299
+ // Only an unambiguous, currently open phase can attribute a command wait. Parallel siblings
2300
+ // deliberately leave gate absent; neither the last phase nor the last red is a safe substitute.
2301
+ const activeGatePhases = new Map();
2302
+ const withCommandContext = (taskId, run, signal = executionSignal()) => {
2303
+ let hostFailure;
2304
+ return runWithCommandLease((_command, execute) => {
2305
+ const active = taskId ? activeGatePhases.get(taskId) : undefined;
2306
+ const gate = active?.size === 1 ? [...active][0] : undefined;
2307
+ let waited = false;
2308
+ return commandLeases.run(async () => {
2309
+ if (hostFailure)
2310
+ throw hostFailure;
2311
+ let count;
2312
+ try {
2313
+ count = await trackHost(admitHost(taskId, hostSignal(signal), undefined, false, gate, () => { waited = true; }));
2314
+ }
2315
+ catch (error) {
2316
+ if (error instanceof HostDegradedError)
2317
+ hostFailure = error;
2318
+ throw error;
2319
+ }
2320
+ if (waited && taskId) {
2321
+ if (gate)
2322
+ journal.phaseStart(taskId, phaseForGate(gate), { gate, admitted: true });
2323
+ else
2324
+ journal.append("suite-admitted", taskId, {});
2325
+ }
2326
+ if (count > 0) {
2327
+ journal.append("suite-budget", taskId, {
2328
+ count, occupancyCap: occupancyCapacity.forkCap, conservativeCap: conservativeCapacity.forkCap,
2329
+ });
2330
+ return await runWithVerificationBudget(conservativeCapacity, execute);
2331
+ }
2332
+ return await execute();
2333
+ }, (count) => {
2334
+ waited = true;
2335
+ journal.append("suite-wait", taskId, { count, ...(gate ? { gate } : {}) });
2336
+ }, SUITE_POLL_MS, signal);
2337
+ }, async () => {
2338
+ try {
2339
+ const result = await run();
2340
+ // Some command oracles turn launch errors into results. Admission failure still parks infra.
2341
+ if (hostFailure)
2342
+ throw hostFailure;
2343
+ return result;
2344
+ }
2345
+ finally {
2346
+ if (taskId)
2347
+ activeGatePhases.delete(taskId);
2348
+ }
2349
+ });
2350
+ };
2136
2351
  let baseRef;
2137
2352
  let baseline;
2138
2353
  let baselinePending = false;
@@ -2264,6 +2479,11 @@ export async function runDaemon(repoRoot, opts = {}) {
2264
2479
  });
2265
2480
  runStarted = true;
2266
2481
  await placeBoard();
2482
+ // A terminal resume with no commands has no execution to admit.
2483
+ if (Object.keys(commands).length > 0 || graph.tasks.some(t => ["pending", "running", "gated"].includes(t.status))) {
2484
+ baselineCapture = initializeHost();
2485
+ void baselineCapture.catch(() => { baselineFailed = true; });
2486
+ }
2267
2487
  }
2268
2488
  else {
2269
2489
  baseRef = await gitHead(repoRoot);
@@ -2273,7 +2493,13 @@ export async function runDaemon(repoRoot, opts = {}) {
2273
2493
  // Workers can run beside capture, but no gate may observe an absent or partial baseline.
2274
2494
  // Keep publication and warnings inside the same barrier as the suite's final verdict.
2275
2495
  baselineCapture = withCommandContext(undefined, async () => {
2276
- const captured = await captureBaseline(repoRoot, commands);
2496
+ // With no commands, capture is already complete: persist it before the probe can wait.
2497
+ // Otherwise a kill during startup can strand a resumable run without baseline.json.
2498
+ const emptyCapture = Object.keys(commands).length === 0 ? await captureBaseline(repoRoot, commands) : undefined;
2499
+ if (emptyCapture)
2500
+ writeFileSync(join(journal.dir, "baseline.json"), JSON.stringify(emptyCapture, null, 2));
2501
+ await initializeHost();
2502
+ const captured = emptyCapture ?? await captureBaseline(repoRoot, commands);
2277
2503
  writeFileSync(join(journal.dir, "baseline.json"), JSON.stringify(captured, null, 2));
2278
2504
  baseline = captured;
2279
2505
  for (const warning of captured.warnings ?? [])
@@ -2519,8 +2745,12 @@ export async function runDaemon(repoRoot, opts = {}) {
2519
2745
  }
2520
2746
  return false;
2521
2747
  };
2748
+ // OBS-1158: admission reads the same journal snapshot the sweep folded, through the shared seam.
2749
+ let admissionPriority = batteryPriority(startupActions.values());
2750
+ const admissible = () => readyTasks(graph, admissionPriority);
2522
2751
  const sweepLiveApprovals = () => {
2523
2752
  const events = journal.read();
2753
+ admissionPriority = batteryPriority(pendingDaemonApprovalActions(events).values());
2524
2754
  const approvals = events.slice(approvalSweepCursor)
2525
2755
  .filter((e) => e.event === "task-approved" && e.taskId);
2526
2756
  approvalSweepCursor = events.length;
@@ -2589,10 +2819,12 @@ export async function runDaemon(repoRoot, opts = {}) {
2589
2819
  // old path. A preservation failure throws and therefore leaves the old checkout in place. The
2590
2820
  // row is deliberately written before the later worktree-recreation row so the journal cannot
2591
2821
  // describe only the commits it carried while omitting uncommitted work the removal destroyed.
2822
+ const producerNow = () => knownProducer(journal.read(), t.id);
2592
2823
  const recreateTaskWorktree = async (taskBranch, taskBase, priorWt) => {
2593
- const ref = await preserveWorktree(priorWt);
2824
+ const producer = producerNow();
2825
+ const ref = await preserveWorktree(priorWt, producer);
2594
2826
  if (ref)
2595
- journal.append("worktree-preserved", t.id, { ref });
2827
+ journal.append("worktree-preserved", t.id, { ref, ...producerFields(producer) });
2596
2828
  return driver.worktree(repoRoot, taskBranch, taskBase);
2597
2829
  };
2598
2830
  // A dead worker with a clean checkout still needs a durable recovery handle: there may be no
@@ -2625,7 +2857,7 @@ export async function runDaemon(repoRoot, opts = {}) {
2625
2857
  // IS the fix; the tried seed and the attempt-loop start close RES-01/RES-02 alongside it.
2626
2858
  //
2627
2859
  // v1.24 OBS-18: a task-approved{release:attempt-cap} zeros rs.attempts (fresh budget) and clears
2628
- // lastAssignment while keeping tried. Only restore lastAssignment when attempts > 0 — after a
2860
+ // lastAssignment while keeping tried, so the restore below is skipped — after a
2629
2861
  // fresh-budget release, prefer nextChannel over the surviving tried-list so burned channels are
2630
2862
  // not re-tried first (consult bans / prior failovers survive the release).
2631
2863
  const rs = resume.get(t.id);
@@ -2646,7 +2878,10 @@ export async function runDaemon(repoRoot, opts = {}) {
2646
2878
  to: channelKey(assignment),
2647
2879
  reason: JSON.stringify(previousHints?.pin) !== JSON.stringify(t.routingHints?.pin) ? "pin changed" : "floor changed",
2648
2880
  });
2649
- if (!hintsChanged && rs?.lastAssignment && rs.attempts > 0
2881
+ // OBS-1161: no `attempts > 0` guard — every release already clears lastAssignment in the replay,
2882
+ // and a lastAssignment at zero attempts is a first dispatch whose capacity requeue was taken back:
2883
+ // the seat is still in force, so restore it instead of failing over its own tried[] entry early.
2884
+ if (!hintsChanged && rs?.lastAssignment
2650
2885
  && channels.some((c) => channelKey(c) === channelKey(rs.lastAssignment))
2651
2886
  && !demotedChannels.has(channelKey(rs.lastAssignment))) {
2652
2887
  assignment = rs.lastAssignment; // restore the consult-chosen assignment (bypasses route()'s static re-pick)
@@ -2807,6 +3042,8 @@ export async function runDaemon(repoRoot, opts = {}) {
2807
3042
  dirtyWorktree: true, dirtyPaths: g.meta.paths,
2808
3043
  ...(typeof g.meta.culprit === "string" ? { culprit: g.meta.culprit } : {}),
2809
3044
  ...(typeof g.meta.preservedRef === "string" ? { preservedRef: g.meta.preservedRef } : {}),
3045
+ ...(typeof g.meta.producer === "string" ? { producer: g.meta.producer } : {}),
3046
+ ...(typeof g.meta.producerAttempt === "number" ? { producerAttempt: g.meta.producerAttempt } : {}),
2810
3047
  } : {}),
2811
3048
  ...(cfg.executionPolicy && !g.pass ? { disposition: failureDisposition(g) } : {}),
2812
3049
  ...Object.fromEntries(["runnerInfraRerun", "hostStarvedRerun", "recoveryBlocked", "failingFiles", "selectionDecision", "failureEvidence"]
@@ -2910,6 +3147,9 @@ export async function runDaemon(repoRoot, opts = {}) {
2910
3147
  const parallelPending = new Set();
2911
3148
  let heldParallel;
2912
3149
  const notePhaseStart = (e) => {
3150
+ const active = activeGatePhases.get(t.id) ?? new Set();
3151
+ active.add(e.gate);
3152
+ activeGatePhases.set(t.id, active);
2913
3153
  if (e.parentAt !== undefined)
2914
3154
  parallelPending.add(e.gate);
2915
3155
  };
@@ -3010,6 +3250,13 @@ export async function runDaemon(repoRoot, opts = {}) {
3010
3250
  }
3011
3251
  return next;
3012
3252
  };
3253
+ // OBS-1161: capacity requeues spent on a seat for this task since its last operator release —
3254
+ // journal-derived so the budget survives a resume instead of restarting with the process.
3255
+ const capacityRequeuesOn = (channel) => {
3256
+ const rows = journal.read().filter((e) => e.taskId === t.id);
3257
+ const since = rows.map((e) => e.event).lastIndexOf("task-approved");
3258
+ return rows.slice(since + 1).filter((e) => e.event === "capacity-requeue" && e.data.channel === channel).length;
3259
+ };
3013
3260
  // OBS-202 (operator law: "you can spawn as many as you want"): channels are session FACTORIES,
3014
3261
  // not consumed seats — a tried channel can always host a fresh worker session, and a fresh
3015
3262
  // session carries none of the failed attempt's baggage. When the untried pool is empty, recycle
@@ -3318,6 +3565,7 @@ export async function runDaemon(repoRoot, opts = {}) {
3318
3565
  journal.phaseStart(t.id, "gates");
3319
3566
  const { results } = await withCommandContext(t.id, async () => runReviewRecovery(resumedTask, {
3320
3567
  carriedAuthors: await subjectAuthors(journal.read(), t.id, wt, taskBase),
3568
+ producer: producerNow(),
3321
3569
  carriedFindings: outstandingReviewFindings(journal.read(), t.id),
3322
3570
  operatorContext,
3323
3571
  worktree: wt, baseRef: taskBase, result: priorResult, author: gateAuthor,
@@ -3350,7 +3598,8 @@ export async function runDaemon(repoRoot, opts = {}) {
3350
3598
  return;
3351
3599
  }
3352
3600
  const g = e.result;
3353
- classifySignalOnlyTest(g);
3601
+ activeGatePhases.get(t.id)?.delete(e.gate);
3602
+ classifyInfraResult(g);
3354
3603
  inParallelOrder(g.gate, () => {
3355
3604
  journalGateResult(g);
3356
3605
  noteReviewRetry(g);
@@ -3361,7 +3610,7 @@ export async function runDaemon(repoRoot, opts = {}) {
3361
3610
  });
3362
3611
  },
3363
3612
  }, false));
3364
- results.forEach(classifySignalOnlyTest);
3613
+ results.forEach(classifyInfraResult);
3365
3614
  if (pendingDaemonApprovalActions(journal.read()).get(t.id)?.authority === "battery") {
3366
3615
  journal.append("recheck-battery", t.id, {
3367
3616
  commit: gateSubject.commit,
@@ -3378,7 +3627,8 @@ export async function runDaemon(repoRoot, opts = {}) {
3378
3627
  await park(t, gateFailApprovalReason(t.id, unavailableReview.details, true), "gate-fail", gateAuthor, rs?.attempts ?? 0, startMs, gateFails, consults, tokens, metered, retryMode);
3379
3628
  return;
3380
3629
  }
3381
- const infra = results.find((g) => g.meta?.infra === true);
3630
+ // OBS-1106: same predicate as classification — an infra replay is parked, never repaired.
3631
+ const infra = results.find((g) => gateFailed(g) && isInfraResult(g));
3382
3632
  if (infra) {
3383
3633
  await park(t, `${infra.gate}: ${infra.details}`, "infra", gateAuthor, rs?.attempts ?? 0, startMs, gateFails, consults, tokens, metered, retryMode);
3384
3634
  return;
@@ -3461,6 +3711,7 @@ export async function runDaemon(repoRoot, opts = {}) {
3461
3711
  saveGraph(repoRoot, graph);
3462
3712
  journal.append("task-done", t.id, {
3463
3713
  attempts: rs?.attempts ?? 0, assignment: gateAuthor, taskContentDigest: contentDigest,
3714
+ authors: mergedAuthors(await subjectAuthors(journal.read(), t.id, wt, taskBase)),
3464
3715
  });
3465
3716
  journal.append("merge", t.id, { branch: taskBranch, commit: await integrationHead(intWt) });
3466
3717
  await trackedDriver.project?.(t.id, "completed");
@@ -3852,13 +4103,16 @@ export async function runDaemon(repoRoot, opts = {}) {
3852
4103
  // Approval can reset the attempt counter while the old pane remains retained.
3853
4104
  // Its ownership claim must survive a new engagement reusing the script path.
3854
4105
  const groupFile = `${dispatchScript}.${nonce}.pgid`;
3855
- workerOwners.set(slot, { taskId: t.id, attempt, groupFile });
4106
+ workerOwners.set(slot, { taskId: t.id, attempt, groupFile, marker: dispatchScript, identities: new Map(), descendants: new Map() });
3856
4107
  writeFileSync(dispatchScript, [
3857
4108
  // Shell startup can swallow the driver's leading cd; the payload owns its checkout too.
3858
4109
  `cd ${shq(wt)} || exit 1`,
4110
+ `export ${VITEST_CACHE_ENV}=${shq(worktreeVitestCache(wt))}`,
3859
4111
  // A driver may launch inside the daemon's group. That group is never worker-owned.
3860
4112
  `worker_pgid=$(ps -o pgid= -p $$ 2>/dev/null); daemon_pgid=$(ps -o pgid= -p ${process.pid} 2>/dev/null)`,
3861
4113
  `if [ -n "$worker_pgid" ] && [ -n "$daemon_pgid" ] && [ "$worker_pgid" != "$daemon_pgid" ]; then printf '%s\\n' "$worker_pgid" > ${shq(groupFile)}; fi`,
4114
+ `ps -o sess= -p $$ > ${shq(`${groupFile}.session`)} 2>/dev/null`,
4115
+ `ps -o pid=,lstart= -p $PPID > ${shq(`${groupFile}.parent`)} 2>/dev/null`,
3862
4116
  "export BASH_SILENCE_DEPRECATION_WARNING=1",
3863
4117
  bannerShell(),
3864
4118
  `printf '%s\\n' 'TICKMARKR_DISPATCH_${nonce}'`,
@@ -3928,7 +4182,7 @@ export async function runDaemon(repoRoot, opts = {}) {
3928
4182
  }
3929
4183
  if (hasSeed || cpuAccountant !== undefined)
3930
4184
  return;
3931
- cpuAccountant = new WorkerTreeCpuAccountant(dispatchScript, wt, () => readOwnedProcessGroup(groupFile));
4185
+ cpuAccountant = new WorkerTreeCpuAccountant(dispatchScript, wt, () => readOwnedProcessGroup(groupFile), workerOwners.get(slot).descendants);
3932
4186
  await cpuAccountant.start();
3933
4187
  };
3934
4188
  const readCpuLeg = () => {
@@ -4024,6 +4278,7 @@ export async function runDaemon(repoRoot, opts = {}) {
4024
4278
  let deadChannelKilled = false;
4025
4279
  let hardTimedOut = false;
4026
4280
  let quotaBannerKilled = false;
4281
+ let capacityBannerKilled = false; // OBS-1161: same banner filter and gates, transient outcome
4027
4282
  let driverProbeFailed = false;
4028
4283
  let heldLegs = [];
4029
4284
  const noteDriverUnreadable = (error) => {
@@ -4352,19 +4607,32 @@ export async function runDaemon(repoRoot, opts = {}) {
4352
4607
  // filtered by identity, never by novelty — a banner already on screen at launch
4353
4608
  // classifies exactly like one printed mid-attempt (T1 review: a novelty baseline
4354
4609
  // exculpated the launch-throttle case forever).
4355
- const bannerMatch = QUOTA_RE.exec(stallSnapshotBannerRows(paneText));
4610
+ // OBS-1161: a transient-capacity banner rides the SAME filter, streak and silence gates,
4611
+ // and concludes the attempt the same way — only its post-loop outcome differs (bounded
4612
+ // requeue on this seat, then same-floor failover, never demotion). Quota wins a tie.
4613
+ const bannerRows = stallSnapshotBannerRows(paneText);
4614
+ const quotaBanner = QUOTA_RE.exec(bannerRows);
4615
+ const bannerMatch = quotaBanner ?? CAPACITY_RE.exec(bannerRows);
4356
4616
  if (bannerMatch)
4357
4617
  quotaStreak++;
4358
4618
  else
4359
4619
  quotaStreak = 0;
4360
- if (bannerMatch && !stallProgress.rowSignalSaturated && !nudgeFailed
4620
+ if (bannerMatch && !quotaBanner && !stallProgress.rowSignalSaturated && !nudgeFailed
4621
+ && !(driver.nudge && NUDGEABLE_ADAPTERS.has(adapter.id) && (!nudged || nudgeDeadline !== undefined))
4622
+ && readCpuLeg().state === "flat"
4623
+ && quotaStreak >= 2 && sliceNow - lastProgressAt >= quotaBannerSilentMs) {
4624
+ capacityBannerKilled = true;
4625
+ journal.append("capacity-banner", t.id, { slot: slot.name, attempt, silentMs: sliceNow - lastProgressAt, matched: bannerMatch[0], excerpt: bannerMatch.input, regex: CAPACITY_RE.source });
4626
+ break;
4627
+ }
4628
+ if (quotaBanner && !stallProgress.rowSignalSaturated && !nudgeFailed
4361
4629
  && !(driver.nudge && NUDGEABLE_ADAPTERS.has(adapter.id) && (!nudged || nudgeDeadline !== undefined))
4362
4630
  && readCpuLeg().state === "flat"
4363
4631
  && quotaStreak >= 2 && sliceNow - lastProgressAt >= quotaBannerSilentMs) {
4364
4632
  // no `output =` here: the post-loop no-trailer tail re-reads the pane anyway, so an
4365
4633
  // assignment would only split the classification read from the verdict read.
4366
4634
  quotaBannerKilled = true;
4367
- journal.append("quota-banner", t.id, { slot: slot.name, attempt, silentMs: sliceNow - lastProgressAt, matched: bannerMatch[0], excerpt: bannerMatch.input, regex: QUOTA_RE.source });
4635
+ journal.append("quota-banner", t.id, { slot: slot.name, attempt, silentMs: sliceNow - lastProgressAt, matched: quotaBanner[0], excerpt: quotaBanner.input, regex: QUOTA_RE.source });
4368
4636
  break;
4369
4637
  }
4370
4638
  // T1 (OBS-262): the `paged` latch is deleted — status is sampled EVERY slice (and
@@ -4520,7 +4788,7 @@ export async function runDaemon(repoRoot, opts = {}) {
4520
4788
  const ref = preservation.ref;
4521
4789
  const reason = `worker is unambiguously dead: pane absent, process tree empty, and worktree unchanged; preserved at ${ref}`;
4522
4790
  deadWorkerPark = { ref, reason };
4523
- journal.append("worktree-preserved", t.id, { ref });
4791
+ journal.append("worktree-preserved", t.id, { ref, ...producerFields(producerNow()) });
4524
4792
  noteWorkerLiveness("worker-dead-held", {
4525
4793
  slot: slot.name, attempt, reason: "unambiguous-worker-death", ref,
4526
4794
  });
@@ -4892,9 +5160,10 @@ export async function runDaemon(repoRoot, opts = {}) {
4892
5160
  try {
4893
5161
  await closeSlot(slot);
4894
5162
  if (!workerFinished) {
4895
- const ref = await preserveWorktree(wt);
5163
+ const producer = producerNow();
5164
+ const ref = await preserveWorktree(wt, producer);
4896
5165
  if (ref) {
4897
- journal.append("worktree-preserved", t.id, { ref });
5166
+ journal.append("worktree-preserved", t.id, { ref, ...producerFields(producer) });
4898
5167
  reapedWorktreeRef = ref;
4899
5168
  }
4900
5169
  journal.append("worker-reaped-before-harvest", t.id, {
@@ -4910,6 +5179,7 @@ export async function runDaemon(repoRoot, opts = {}) {
4910
5179
  }
4911
5180
  }
4912
5181
  else if (keepOpen && (workerFinished || processExited || driver.id !== "subprocess")) {
5182
+ await reapWorker(slot);
4913
5183
  keptSlots.push(slot);
4914
5184
  supersededWorkerSlot = slot;
4915
5185
  }
@@ -4964,9 +5234,10 @@ export async function runDaemon(repoRoot, opts = {}) {
4964
5234
  stallReaps = stallSeat === seat ? stallReaps + 1 : 1;
4965
5235
  stallSeat = seat;
4966
5236
  if (stallReaps >= 2) {
4967
- const ref = await preserveWorktree(wt);
5237
+ const producer = producerNow();
5238
+ const ref = await preserveWorktree(wt, producer);
4968
5239
  if (ref)
4969
- journal.append("worktree-preserved", t.id, { ref });
5240
+ journal.append("worktree-preserved", t.id, { ref, ...producerFields(producer) });
4970
5241
  await park(t, `two consecutive stall reaps without a gate on seat ${seat}`, "stall", assignment, attempt + 1, startMs, gateFails, consults, tokens, metered, retryMode, { seat, stallReaps });
4971
5242
  return;
4972
5243
  }
@@ -5011,6 +5282,12 @@ export async function runDaemon(repoRoot, opts = {}) {
5011
5282
  const quotaMatch = (interactive ? !workerFinished : exitCode !== 0)
5012
5283
  ? QUOTA_RE.exec(stallSnapshotBannerRows(output))
5013
5284
  : null;
5285
+ // OBS-1161: transient capacity reads the SAME tail under the same guards — a live idle banner
5286
+ // and a no-trailer capacity exit classify identically — through the parse-boundary rule that a
5287
+ // parsed verdict (either way) is work, so a trailer QUOTING the phrase never lands here.
5288
+ const capacityMatch = !quotaMatch && (interactive ? !workerFinished : exitCode !== 0)
5289
+ ? classifyTransientCapacity({ ...preHarvestResult, raw: stallSnapshotBannerRows(output) })
5290
+ : null;
5014
5291
  // Q-1: a graph pin is an operator instruction — a quota match ALONE never overrides it; only a
5015
5292
  // channel-attributed error (auth/setup/outage/timeout, the typed dead-channel classes) may.
5016
5293
  const pin = t.routingHints?.pin;
@@ -5021,7 +5298,7 @@ export async function runDaemon(repoRoot, opts = {}) {
5021
5298
  // demote the pin two tails in and the demotion re-dispatch would move the task off it.
5022
5299
  if (preHarvestResult.ok && workerFinished)
5023
5300
  noTrailerStreak.set(channelKey(assignment), 0);
5024
- else if (!workerFinished && cause !== "provider-death" && !pinRefusesQuota) {
5301
+ else if (!workerFinished && cause !== "provider-death" && !pinRefusesQuota && !capacityMatch) {
5025
5302
  const ck = channelKey(assignment);
5026
5303
  const streak = (noTrailerStreak.get(ck) ?? 0) + 1;
5027
5304
  noTrailerStreak.set(ck, streak);
@@ -5039,6 +5316,48 @@ export async function runDaemon(repoRoot, opts = {}) {
5039
5316
  attempt--;
5040
5317
  continue;
5041
5318
  }
5319
+ // OBS-1161: transient capacity → bounded same-seat requeue with backoff (no attempt burn, no
5320
+ // consult), then a same-floor failover exactly like quota — but the seat is NEVER demoted or
5321
+ // excluded: it is busy, not dead, and a later task may find it free. The budget is read from
5322
+ // the journal, never a loop-local counter, so a resume continues the count it left off at.
5323
+ if (capacityMatch) {
5324
+ const from = channelKey(assignment);
5325
+ const requeues = capacityRequeuesOn(from);
5326
+ const source = capacityBannerKilled ? "banner" : "exit";
5327
+ if (requeues < CAPACITY_REQUEUE_CAP) {
5328
+ journal.append("capacity-requeue", t.id, {
5329
+ attempt, requeue: requeues + 1, of: CAPACITY_REQUEUE_CAP, channel: from, assignment,
5330
+ matched: capacityMatch[0], source, backoffMs: capacityBackoffMs,
5331
+ });
5332
+ await new Promise((r) => setTimeout(r, capacityBackoffMs));
5333
+ attempt--;
5334
+ continue;
5335
+ }
5336
+ const next = failover("capacity-failover");
5337
+ journal.append("capacity-failover", t.id, {
5338
+ from, to: next ? channelKey(next) : null, matched: capacityMatch[0], source, requeues, cause: "capacity",
5339
+ });
5340
+ if (next) {
5341
+ await driver.notify(`tickmarkr ${runId}: ${t.id} capacity failover`, { tier: "attention" });
5342
+ if (!keepForever) {
5343
+ const idx = keptSlots.indexOf(slot);
5344
+ if (idx >= 0) {
5345
+ keptSlots.splice(idx, 1);
5346
+ try {
5347
+ await closeSlot(slot);
5348
+ }
5349
+ catch { /* cosmetic — reconcile is the backstop */ }
5350
+ }
5351
+ if (supersededWorkerSlot === slot)
5352
+ supersededWorkerSlot = undefined;
5353
+ }
5354
+ assignment = next;
5355
+ tried.push(channelKey(next));
5356
+ continue;
5357
+ }
5358
+ await park(t, `capacity exhausted on ${from} after ${requeues} requeues and no eligible channel at floor`, "quota", assignment, attempt + 1, startMs, gateFails, consults, tokens, metered, retryMode, { cause: "capacity", channel: from, requeues });
5359
+ return;
5360
+ }
5042
5361
  // quota exhaustion → failover within floor; does NOT consume the ladder (spec §4)
5043
5362
  // print: guarded on exit code — exit-0 output that merely MENTIONS "rate limit" must not failover
5044
5363
  // interactive: a worker-CLAIMED trailer beats quota mentions; without one, quota text fails over
@@ -5202,7 +5521,8 @@ export async function runDaemon(repoRoot, opts = {}) {
5202
5521
  return;
5203
5522
  }
5204
5523
  const g = e.result;
5205
- classifySignalOnlyTest(g);
5524
+ activeGatePhases.get(t.id)?.delete(e.gate);
5525
+ classifyInfraResult(g);
5206
5526
  inParallelOrder(g.gate, () => {
5207
5527
  // GATE-09 (ROADMAP SC-4): journal every judge retry as an attributable event — which gate flaked,
5208
5528
  // which channel flaked, which channel retried — so `tickmarkr journal`/report can distinguish "judge
@@ -5258,6 +5578,9 @@ export async function runDaemon(repoRoot, opts = {}) {
5258
5578
  meta: Object.fromEntries(Object.entries(data).filter(([key]) => !GATE_TELEMETRY_KEYS.includes(key) && key !== "capacity")),
5259
5579
  }));
5260
5580
  commits = await commitsAheadOf(taskBase, wt);
5581
+ // OBS-1106: a replayed legacy infra row lacking `infra` is re-classified before it is
5582
+ // re-journaled and before the infra park below reads it.
5583
+ results.forEach(classifyInfraResult);
5261
5584
  for (const g of results) {
5262
5585
  journal.append("gate-replayed", t.id, {
5263
5586
  attempt, priorAttempt: attempt - 1, gate: g.gate, commit: gateSubject.commit,
@@ -5270,6 +5593,7 @@ export async function runDaemon(repoRoot, opts = {}) {
5270
5593
  else {
5271
5594
  ({ results, commits } = await withCommandContext(t.id, async () => runReviewRecovery(t, {
5272
5595
  carriedAuthors: await subjectAuthors(journal.read(), t.id, wt, taskBase),
5596
+ producer: producerNow(),
5273
5597
  carriedFindings: outstandingFindings,
5274
5598
  operatorContext,
5275
5599
  worktree: wt, baseRef: taskBase, result, author: assignment,
@@ -5299,7 +5623,7 @@ export async function runDaemon(repoRoot, opts = {}) {
5299
5623
  onGate,
5300
5624
  })));
5301
5625
  }
5302
- results.forEach(classifySignalOnlyTest);
5626
+ results.forEach(classifyInfraResult);
5303
5627
  graph = addEvidence(graph, t.id, { commits, gateResults: results, artifacts: [promptFile] });
5304
5628
  saveGraph(repoRoot, graph);
5305
5629
  if (results.some((g) => g.gate === "test" && !g.pass
@@ -5336,6 +5660,7 @@ export async function runDaemon(repoRoot, opts = {}) {
5336
5660
  saveGraph(repoRoot, graph);
5337
5661
  journal.append("task-done", t.id, {
5338
5662
  attempts: attempt + 1, assignment, taskContentDigest: contentDigest,
5663
+ authors: mergedAuthors(await subjectAuthors(journal.read(), t.id, wt, taskBase)),
5339
5664
  });
5340
5665
  journal.append("merge", t.id, { branch: taskBranch, commit: await integrationHead(intWt) });
5341
5666
  await trackedDriver.project?.(t.id, "completed");
@@ -5369,7 +5694,7 @@ export async function runDaemon(repoRoot, opts = {}) {
5369
5694
  await park(t, gateFailApprovalReason(t.id, unavailableReview.details, true), "gate-fail", assignment, attempt + 1, startMs, gateFails, consults, tokens, metered, retryMode);
5370
5695
  return;
5371
5696
  }
5372
- const infraFailure = results.find((g) => gateFailed(g) && g.meta?.infra === true);
5697
+ const infraFailure = results.find((g) => gateFailed(g) && isInfraResult(g));
5373
5698
  if (infraFailure) {
5374
5699
  await park(t, `${infraFailure.gate}: ${infraFailure.details}${infraFailure.meta?.recoveryBlocked ? ` — ${infraFailure.meta.recoveryBlocked}` : ""}`, "infra", assignment, attempt + 1, startMs, gateFails, consults, tokens, metered, retryMode);
5375
5700
  return;
@@ -5586,7 +5911,7 @@ export async function runDaemon(repoRoot, opts = {}) {
5586
5911
  const holdEndCondition = (closing = true) => {
5587
5912
  sweepLiveApprovals();
5588
5913
  const freeSlots = Math.max(0, concurrency - inflight.size);
5589
- const dispatchable = readyTasks(graph).filter((t) => !inflight.has(t.id));
5914
+ const dispatchable = admissible().filter((t) => !inflight.has(t.id));
5590
5915
  if (freeSlots === 0 || dispatchable.length === 0)
5591
5916
  return false;
5592
5917
  if (closing) {
@@ -5606,7 +5931,7 @@ export async function runDaemon(repoRoot, opts = {}) {
5606
5931
  throw new Error(`terminated by ${termSignal}`);
5607
5932
  await watchBoard();
5608
5933
  sweepLiveApprovals();
5609
- const ready = readyTasks(graph)
5934
+ const ready = admissible()
5610
5935
  .filter((t) => !inflight.has(t.id))
5611
5936
  .slice(0, Math.max(0, concurrency - inflight.size));
5612
5937
  for (const t of ready) {
@@ -5627,9 +5952,14 @@ export async function runDaemon(repoRoot, opts = {}) {
5627
5952
  if (fatalStop.signal.aborted)
5628
5953
  return;
5629
5954
  const cleanupEvidence = cleanupErrors.length ? { cleanupErrors } : {};
5955
+ if (err instanceof HostDegradedError) {
5956
+ await park(t, err.message, "infra", null, 0, Date.now(), 0, 0, undefined, 0, "fresh", { disposition: "host-degraded", ...cleanupEvidence });
5957
+ return;
5958
+ }
5630
5959
  if (err instanceof HeldProbeExhausted) {
5631
5960
  const wt = worktreePath(repoRoot, `${branch}--${t.id}`);
5632
- let ref = await withoutExecutionBudget(() => preserveWorktree(wt));
5961
+ const producer = knownProducer(journal.read(), t.id);
5962
+ let ref = await withoutExecutionBudget(() => preserveWorktree(wt, producer));
5633
5963
  if (!ref) {
5634
5964
  const head = await gitHead(wt);
5635
5965
  ref = `refs/tickmarkr/preserved/${head}`;
@@ -5637,15 +5967,16 @@ export async function runDaemon(repoRoot, opts = {}) {
5637
5967
  if (saved.code !== 0)
5638
5968
  throw new Error(`could not preserve ${head}: ${saved.stderr}`);
5639
5969
  }
5640
- journal.append("worktree-preserved", t.id, { ref });
5970
+ journal.append("worktree-preserved", t.id, { ref, ...producerFields(producer) });
5641
5971
  await park(t, err.message, "infra", null, 0, Date.now(), 0, 0, undefined, 0, "fresh", { disposition: "transport-uncertain", ref, ...cleanupEvidence });
5642
5972
  return;
5643
5973
  }
5644
5974
  if (err instanceof ExecutionBudgetExceeded) {
5645
5975
  const wt = worktreePath(repoRoot, `${branch}--${t.id}`);
5646
- const ref = await preserveWorktree(wt);
5976
+ const producer = knownProducer(journal.read(), t.id);
5977
+ const ref = await preserveWorktree(wt, producer);
5647
5978
  if (ref)
5648
- journal.append("worktree-preserved", t.id, { ref });
5979
+ journal.append("worktree-preserved", t.id, { ref, ...producerFields(producer) });
5649
5980
  const dispatch = journal.read().reverse().find((e) => e.taskId === t.id && e.event === "task-dispatch");
5650
5981
  await park(t, err.message, "infra", null, 0, Date.now(), 0, 0, undefined, 0, "fresh", { disposition: "execution-budget-exhausted", limitMs: cfg.executionPolicy.taskExecutionLimitMs,
5651
5982
  ...cleanupEvidence,
@@ -5679,7 +6010,7 @@ export async function runDaemon(repoRoot, opts = {}) {
5679
6010
  journal.append("approval-window-start", undefined, { windowMs, parked: [...parked] });
5680
6011
  // The narrator may itself append a decision at this boundary.
5681
6012
  sweepLiveApprovals();
5682
- if (readyTasks(graph).length) {
6013
+ if (admissible().length) {
5683
6014
  approvalDeadline = undefined;
5684
6015
  continue;
5685
6016
  }
@@ -5729,7 +6060,12 @@ export async function runDaemon(repoRoot, opts = {}) {
5729
6060
  pending: pendingTasks(graph).map((t) => t.id),
5730
6061
  };
5731
6062
  fatalPhase = "baseline";
5732
- await baselineCapture; // also drain capture when every worker parks or fails before its gates
6063
+ await baselineCapture.catch(error => {
6064
+ // A host park is resumable only after baseline publication. Before that, retain the
6065
+ // fatal baseline failure: resume requires baseline.json and cannot recover this capture.
6066
+ if (!(error instanceof HostDegradedError) || !existsSync(join(journal.dir, "baseline.json")))
6067
+ throw error;
6068
+ });
5733
6069
  fatalPhase = "tip-verify";
5734
6070
  // OBS-34: post-merge integration-tip verify — strict exit codes, no baseline forgiveness.
5735
6071
  const lastMergedTask = [...journal.read()].reverse().find((e) => e.event === "merge" && e.taskId)?.taskId;
@@ -5739,7 +6075,7 @@ export async function runDaemon(repoRoot, opts = {}) {
5739
6075
  const checkApprovals = () => {
5740
6076
  try {
5741
6077
  sweepLiveApprovals();
5742
- if (readyTasks(graph).length)
6078
+ if (admissible().length)
5743
6079
  controller.abort(cancellation);
5744
6080
  if (termSignal)
5745
6081
  controller.abort(new Error(`terminated by ${termSignal}`));