tickmarkr 2.1.4 → 2.1.6

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -19,9 +19,9 @@ import { addEvidence, attributeBlocked, blockedTasks, getTask, graphDefinitionHa
19
19
  import { GATE_NAMES } from "../graph/schema.js";
20
20
  import { augmentRetryBrief, consult, renderRetryGuidance } from "./consult.js";
21
21
  import { runEnvironment } from "./environment.js";
22
- import { cleanupRunWorktrees, gitHead, linkNodeModules, npmDependencyInstallCommand, npmDependencyManifestChanged, preserveWorktree, runWithForkBudget, sh, shGit, WORKTREE_LAYOUT_CONTRACT, worktreePath } from "./git.js";
22
+ import { cleanupRunWorktrees, gitHead, linkNodeModules, npmDependencyInstallCommand, npmDependencyManifestChanged, preserveWorktree, resolvedCapacity, runWithForkBudget, sameCapacity, sh, shGit, WORKTREE_LAYOUT_CONTRACT, worktreePath } from "./git.js";
23
23
  import { runInteractiveSeed } from "./interactive-seed.js";
24
- import { activeRetryBan, classifyTaskFailure, classifyWorkerResultCause, engagementComparable, formatPriorFindingEvidence, GATE_FINGERPRINT_CAP, GATE_SATISFIED_RELEASE, identicalGateFailures, journaledFailureBrief, Journal, loadRoutingProfile, newRunId, normalizeGateFailure, outstandingReviewFindings, pendingRepairFindings, phaseForGate, readPriorRunEvidence, recordedTaskFailureKind, repairsSinceApproval, reviewRoundsSinceApproval, runHasEnded, structuredFindings, upheldFeedbackByTask } from "./journal.js";
24
+ import { activeRetryBan, classifyTaskFailure, classifyWorkerResultCause, deferredReviewFindings, engagementComparable, formatPriorFindingEvidence, GATE_FINGERPRINT_CAP, GATE_SATISFIED_RELEASE, identicalGateFailures, isDeferredFinding, journaledFailureBrief, Journal, loadRoutingProfile, newRunId, normalizeGateFailure, outstandingReviewFindings, pendingRepairFindings, phaseForGate, readPriorRunEvidence, recordedTaskFailureKind, renderStructuredReviewFinding, repairsSinceApproval, reviewRoundsSinceApproval, runHasEnded, structuredFindings, upheldFeedbackByTask } from "./journal.js";
25
25
  import { isDiffCapPark } from "../gates/review.js";
26
26
  import { acquireApprovalSerialization, acquireRunLock, releaseRunLock } from "./lock.js";
27
27
  import { ensureIntegration, integrationBranch, integrationHead, mergeTask, verifyIntegrationTip } from "./merge.js";
@@ -242,6 +242,10 @@ function repairBrief(findings, diff, baseRef) {
242
242
  // v1.70 T5: default request-changes rounds a task may draw before it parks. OBS-419 keeps this as the
243
243
  // no-ceiling behavior; an operator may narrow only the next engagement on the approval that releases it.
244
244
  const REVIEW_ROUND_CAP = 2;
245
+ // T2: one heading per fact. The first is owed a passing review; the second already drew one and was
246
+ // accepted with the reviewer's own rationale, which travels beside (but never changes) its identity.
247
+ const OUTSTANDING_FINDINGS_HEADING = "## Outstanding review findings — a review has NOT passed on these yet";
248
+ const DEFERRED_FINDINGS_HEADING = "## Deferred review findings — a reviewer ACCEPTED these with a rationale and did NOT block on them; do not re-litigate, fix only if your change touches them";
245
249
  // OBS-419: the newest approval starts the current engagement, so it is also the sole authority for
246
250
  // that engagement's optional ceiling. Stop at the newest approval even when the field is absent: a
247
251
  // later ordinary release restores the module default instead of inheriting an older operator limit.
@@ -392,7 +396,7 @@ function lastVerifyCycle(events) {
392
396
  if (e.event === "tip-verify-start") {
393
397
  const { tip, cmdHash } = e.data;
394
398
  cur = typeof tip === "string" && typeof cmdHash === "string"
395
- ? { tip, cmdHash, gates: new Set(), failed: false, forgiven: false }
399
+ ? { tip, cmdHash, gates: new Set(), failed: false, forgiven: false, capacities: [e.data.capacity] }
396
400
  : undefined;
397
401
  afterRunEnd = false;
398
402
  continue;
@@ -410,9 +414,14 @@ function lastVerifyCycle(events) {
410
414
  continue;
411
415
  }
412
416
  if (!cur || afterRunEnd || cur.tip !== tip || cur.cmdHash !== cmdHash) {
413
- cur = { tip, cmdHash, gates: new Set(), failed: false, forgiven: false };
417
+ cur = { tip, cmdHash, gates: new Set(), failed: false, forgiven: false, capacities: [] };
414
418
  }
415
419
  afterRunEnd = false;
420
+ // T7: EVERY verdict row's own capacity, not just the start row's. The start row is a statement of
421
+ // intent written before a single command ran; the green this cache would carry forward lives on
422
+ // these rows, so a row whose capacity differs from the session's — or which is malformed — has to
423
+ // be able to sink the cycle by itself.
424
+ cur.capacities.push(e.data.capacity);
416
425
  if (e.event === "tip-verify-failed")
417
426
  cur.failed = true;
418
427
  else {
@@ -440,23 +449,31 @@ function lastVerifyCycle(events) {
440
449
  export async function verifyIntegrationTipCached(intWt, commands, journal, opts = {}) {
441
450
  const cmdHash = commandsHash(commands);
442
451
  const tip = await gitHead(intWt);
452
+ // T7: the capacity this session's verify children WOULD run under — the third thing a carried
453
+ // green must match, beside the tip and the command set. A cached verdict is the one place a green
454
+ // crosses a session boundary with nothing re-run, and a session resumed at a different concurrency
455
+ // divides the machine by a different number: that green was established in another world, so it is
456
+ // not carried forward and the commands run again. A pre-T7 cycle records no capacity and keeps
457
+ // exactly the behaviour it has today.
458
+ const capacity = resolvedCapacity();
443
459
  const porcelain = await shGit("git status --porcelain", intWt);
444
460
  const clean = porcelain.code === 0 && porcelain.stdout.trim() === "";
445
461
  const last = lastVerifyCycle(journal.read());
446
462
  const cached = last !== undefined && !last.failed && !last.forgiven && last.tip === tip && last.cmdHash === cmdHash
447
- && Object.keys(commands).every((g) => last.gates.has(g));
463
+ && Object.keys(commands).every((g) => last.gates.has(g))
464
+ && last.capacities.every((recorded) => sameCapacity(recorded, capacity));
448
465
  // A pair can be verified red and then green without either SHA or command hash changing (for
449
466
  // example, an external service or ignored fixture recovers). Delimit attempts explicitly so that
450
467
  // the earlier red cannot remain latched into the later complete green cycle.
451
- journal.append("tip-verify-start", undefined, { tip, cmdHash, gates: Object.keys(commands), cached: clean && cached });
468
+ journal.append("tip-verify-start", undefined, { tip, cmdHash, capacity, gates: Object.keys(commands), cached: clean && cached });
452
469
  if (clean && cached) {
453
- journal.append("tip-verify-cached", undefined, { tip, cmdHash, gates: Object.keys(commands) });
470
+ journal.append("tip-verify-cached", undefined, { tip, cmdHash, capacity, gates: Object.keys(commands) });
454
471
  // The skip must not read as a red. Every surface derives the tip's verdict from this cycle's
455
472
  // `tip-verify` events (cockpit derive.ts tipVerificationPassed: a run-end claiming "passed" with
456
473
  // ZERO events is fail-closed to FALSE), so a carried-forward green still journals its per-gate
457
474
  // pass — `cached: true` keeps it honest about not having re-run the command.
458
475
  for (const gate of Object.keys(commands)) {
459
- journal.append("tip-verify", undefined, { gate, cmd: commands[gate], pass: true, exitCode: 0, cached: true, tip, cmdHash });
476
+ journal.append("tip-verify", undefined, { gate, cmd: commands[gate], pass: true, exitCode: 0, cached: true, tip, cmdHash, capacity });
460
477
  }
461
478
  return false;
462
479
  }
@@ -464,7 +481,7 @@ export async function verifyIntegrationTipCached(intWt, commands, journal, opts
464
481
  for (const r of await verifyIntegrationTip(intWt, commands, journal.dir, opts.baseline)) {
465
482
  if (r.pass) {
466
483
  // Q121s: a forgiven pass journals its fingerprints — honest about what was carried, never a silent green.
467
- journal.append("tip-verify", undefined, { gate: r.gate, cmd: r.cmd, pass: true, exitCode: r.exitCode, details: r.details, ...(r.forgiven ? { forgiven: true, fingerprints: r.fingerprints } : {}), tip, cmdHash });
484
+ journal.append("tip-verify", undefined, { gate: r.gate, cmd: r.cmd, pass: true, exitCode: r.exitCode, details: r.details, ...(r.forgiven ? { forgiven: true, fingerprints: r.fingerprints } : {}), tip, cmdHash, capacity });
468
485
  }
469
486
  else {
470
487
  journal.append("tip-verify-failed", undefined, {
@@ -476,6 +493,7 @@ export async function verifyIntegrationTipCached(intWt, commands, journal, opts
476
493
  lastMergedTask: opts.lastMergedTask,
477
494
  tip,
478
495
  cmdHash,
496
+ capacity,
479
497
  });
480
498
  tipFailed = true;
481
499
  }
@@ -503,6 +521,35 @@ async function gateCommitSubject(base, head, wt) {
503
521
  return head; // fail closed to the exact object id if canonicalization fails
504
522
  return createHash("sha256").update(history.stdout).digest("hex");
505
523
  }
524
+ // A CPU total cannot prove a tree empty: a newly launched live process can still have a measured
525
+ // total of zero at the host's clock resolution. This probe asks the narrower cardinality question
526
+ // OBS-737 needs. A readable process table with no marker root is measured empty; a failed or empty
527
+ // table is unmeasurable. Descendants are closed over PPID because not every child retains the
528
+ // dispatch-script marker in its own argv.
529
+ async function observeWorkerProcessTree(marker, cwd) {
530
+ const snapshot = await shGit("ps -Awwo pid=,ppid=,command=", cwd, 15_000);
531
+ if (snapshot.code !== 0)
532
+ return "unmeasurable";
533
+ const rows = [];
534
+ for (const line of snapshot.stdout.split("\n")) {
535
+ const match = /^\s*(\d+)\s+(\d+)\s+(.*)$/.exec(line);
536
+ if (match)
537
+ rows.push({ pid: match[1], ppid: match[2], command: match[3] });
538
+ }
539
+ if (rows.length === 0)
540
+ return "unmeasurable";
541
+ const tree = new Set(rows.filter((row) => row.command.includes(marker)).map((row) => row.pid));
542
+ for (let grew = true; grew;) {
543
+ grew = false;
544
+ for (const row of rows) {
545
+ if (!tree.has(row.pid) && tree.has(row.ppid)) {
546
+ tree.add(row.pid);
547
+ grew = true;
548
+ }
549
+ }
550
+ }
551
+ return tree.size === 0 ? "empty" : "running";
552
+ }
506
553
  const OBSERVE_CHUNK_BYTES = 64 * 1024;
507
554
  const OBSERVE_BUDGET_BYTES = 256 * 1024 * 1024;
508
555
  let observeBudgetBytes = OBSERVE_BUDGET_BYTES;
@@ -587,6 +634,20 @@ async function boundedGitObservation(command, worktree, budget) {
587
634
  return undefined;
588
635
  return result.stdout;
589
636
  }
637
+ // "No worktree delta" means no staged/unstaged/untracked bytes AND no commits beyond the task's
638
+ // dispatch base. Both reads are bounded and preserve the observer's third state: a failed probe is
639
+ // unreadable, never clean.
640
+ async function observeWorktreeDelta(base, worktree) {
641
+ let budget = observeBudgetBytes;
642
+ const status = await boundedGitObservation("GIT_OPTIONAL_LOCKS=0 git status --porcelain=v1 -z --untracked-files=all", worktree, budget);
643
+ if (status === undefined)
644
+ return "unreadable";
645
+ budget -= Buffer.byteLength(status);
646
+ const ahead = await boundedGitObservation(`GIT_OPTIONAL_LOCKS=0 git rev-list --count ${shq(base)}..HEAD`, worktree, budget);
647
+ if (ahead === undefined || !/^\d+$/.test(ahead.trim()))
648
+ return "unreadable";
649
+ return status.length === 0 && ahead.trim() === "0" ? "unchanged" : "changed";
650
+ }
590
651
  // The signature has four git-owned inputs: HEAD, staged blob/mode/path identity, porcelain path and
591
652
  // state, and the set of worktree paths whose bytes Git cannot supply. Those paths contribute their
592
653
  // filesystem identity, mode, symlink text, and content. A failed or over-budget leg is a third
@@ -898,6 +959,10 @@ export async function runDaemon(repoRoot, opts = {}) {
898
959
  const replayedGateResults = resumeLifecycleOpen
899
960
  ? journal.replayCurrentAttemptGateResults()
900
961
  : new Map();
962
+ // T7: the capacity THIS session resolved — read inside the run's fork budget, so it is the number
963
+ // every shell this run spawns will divide the machine by. Recorded evidence from another session
964
+ // is only reusable against this.
965
+ const sessionCapacity = resolvedCapacity();
901
966
  const replayedExclusions = opts.resume ? journal.replayExcludedChannels() : new Set();
902
967
  if (opts.resume) {
903
968
  // v1.53 T5: a superseded run is dead — resuming it beside its successor is the exact
@@ -1069,10 +1134,10 @@ export async function runDaemon(repoRoot, opts = {}) {
1069
1134
  };
1070
1135
  // gateFails/consults are execTask-scoped counters passed in so a park row is a rich verified-failure
1071
1136
  // observation (e.g. ladder-exhausted + gateFails:4); every task-human row has a closed kind, never prose alone.
1072
- const park = async (t, reason, kind, assignment, attempts, startMs, gateFails = 0, consults = 0, tokens, metered = 0, retryMode = "fresh") => {
1137
+ const park = async (t, reason, kind, assignment, attempts, startMs, gateFails = 0, consults = 0, tokens, metered = 0, retryMode = "fresh", details = {}) => {
1073
1138
  graph = setStatus(graph, t.id, "human");
1074
1139
  saveGraph(repoRoot, graph);
1075
- journal.append("task-human", t.id, { reason, kind });
1140
+ journal.append("task-human", t.id, { ...details, reason, kind });
1076
1141
  if (assignment) {
1077
1142
  // OBS-547: `metered` counts CHARGEABLE metered attempts, so an unchargeable dispatch passes 0 and
1078
1143
  // the count is omitted rather than written as 0 or as `1` beside `attempts: 0` — a row claiming
@@ -1150,6 +1215,22 @@ export async function runDaemon(repoRoot, opts = {}) {
1150
1215
  journal.append("worktree-preserved", t.id, { ref });
1151
1216
  return driver.worktree(repoRoot, taskBranch, taskBase);
1152
1217
  };
1218
+ // A dead worker with a clean checkout still needs a durable recovery handle: there may be no
1219
+ // pane left to identify even the commit it was dispatched from. preserveWorktree deliberately
1220
+ // creates no ref for ordinary clean recreations, so this exceptional terminal path pins HEAD
1221
+ // explicitly under the same recovery namespace. Reconfirm the delta immediately before the
1222
+ // ref write; a change or unreadable recheck withdraws the park.
1223
+ const preserveDeadWorker = async (worktree, taskBase) => {
1224
+ const state = await observeWorktreeDelta(taskBase, worktree);
1225
+ if (state !== "unchanged")
1226
+ return { state };
1227
+ const head = await gitHead(worktree);
1228
+ const ref = `refs/tickmarkr/preserved/${head}`;
1229
+ const updated = await shGit(`git update-ref ${shq(ref)} ${shq(head)}`, worktree);
1230
+ if (updated.code !== 0)
1231
+ throw new Error(`could not preserve dead worker HEAD at ${ref}: ${updated.stderr || updated.stdout}`);
1232
+ return { state, ref };
1233
+ };
1153
1234
  const r = route(t, cfg, channels, profile, undefined, demotedChannels);
1154
1235
  for (const lint of r.lints)
1155
1236
  journal.append("routing-lint", t.id, { lint });
@@ -1221,6 +1302,15 @@ export async function runDaemon(repoRoot, opts = {}) {
1221
1302
  let gateSubject;
1222
1303
  const journalGateResult = (g) => {
1223
1304
  const blocking = gateFailed(g) && (g.gate === "review" || g.gate === "acceptance");
1305
+ // T2: a review that PASSED while DEFERRING a concern still recorded a defect — the prompt
1306
+ // promises the deferral is recorded and never dropped, and a details string is not a record a
1307
+ // later round can match. The blocking projection above is the only writer of structured
1308
+ // findings today, so on this row it writes nothing and every structured reader goes blind.
1309
+ // Same shape, same identity, on the passing row: the verdict is untouched (`pass` stays true),
1310
+ // only the projection widens to the rows the reviewer itself classified as deferred.
1311
+ const deferred = !blocking && g.gate === "review" && g.pass === true
1312
+ ? deferredReviewFindings(g.details)
1313
+ : [];
1224
1314
  // R3 (OBS-186): a gate that DECLINED has no verdict to state, and this row is the ONE seam every
1225
1315
  // fold outside this file shares. Writing `pass: false` for a decline is what turned a skip into
1226
1316
  // a failure at all of them at once — the engagement round budget (reviewRoundsSinceApproval,
@@ -1265,6 +1355,9 @@ export async function runDaemon(repoRoot, opts = {}) {
1265
1355
  ...(blocking ? {
1266
1356
  taskContentDigest: contentDigest,
1267
1357
  findings: structuredFindings(g.gate, g.details),
1358
+ } : deferred.length > 0 ? {
1359
+ taskContentDigest: contentDigest,
1360
+ findings: deferred,
1268
1361
  } : {}),
1269
1362
  // v2.0 T2 (OBS-554): the gate's OWN measurement, lifted verbatim from the meta run-gates
1270
1363
  // stamped WHERE THE GATE RAN. Nothing here re-derives a duration by subtracting journal
@@ -1275,6 +1368,27 @@ export async function runDaemon(repoRoot, opts = {}) {
1275
1368
  // every recalibration this telemetry funds, a gap is honest and a zero is a lie. The
1276
1369
  // seven-gate closed set is asserted end-to-end in tests/run/gate-telemetry.test.ts.
1277
1370
  ...gateMeasurement(g.meta),
1371
+ // T7: the capacity the gate's own command child ran under, lifted verbatim from the result
1372
+ // the battery produced — read where the shell built that child's environment, never
1373
+ // re-derived from the run's own budget, which would answer a different number than the
1374
+ // operator's export did. It is this row's only COMPARABLE identity: the two load samples
1375
+ // beside it are endpoint reads of a gate whose interior neither of them saw, so no reader
1376
+ // compares them, and matching capacity never claims the machine was calm. A gate that ran
1377
+ // no command carries nothing — it divided nothing.
1378
+ //
1379
+ // `dirtiedBy` is the one hole in that lift: run-gates REPLACES a green battery verdict with
1380
+ // a refusal when the command left the worktree dirty (run-gates.ts, three sites: the legacy
1381
+ // batch, the per-command loop and the merge-candidate full suite), and the refusal is a
1382
+ // fresh verdict object carrying nothing off the result it replaced. That command's child DID
1383
+ // run, so its row still owes the world it ran in. `sessionCapacity` is that world and not a
1384
+ // re-derivation of it: it applies the same precedence `shell` does — an operator export
1385
+ // first — and was read inside the same fork budget every gate child of this run is spawned
1386
+ // under, so it is by construction the number that child received. The flag is set only where
1387
+ // a command of THIS gate ran and dirtied the tree, so the round-entry refusal (no command
1388
+ // ran) and the round-end withdrawal (lands on a gate that runs no command) still carry
1389
+ // nothing.
1390
+ ...(g.capacity ? { capacity: g.capacity }
1391
+ : g.meta?.dirtiedBy === g.gate ? { capacity: sessionCapacity } : {}),
1278
1392
  });
1279
1393
  };
1280
1394
  // R3 (OBS-186): judge ‖ review are launched together and publish in COMPLETION order
@@ -1555,7 +1669,21 @@ export async function runDaemon(repoRoot, opts = {}) {
1555
1669
  const canonicalCurrentCommit = currentTaskSubject === replayedGates.commit;
1556
1670
  const recreatedLegacyCommit = priorTaskTip === replayedGates.commit
1557
1671
  && priorTaskSubject === currentTaskSubject;
1558
- let reusable = exactCurrentCommit || canonicalCurrentCommit || recreatedLegacyCommit;
1672
+ // T7: the commit says the gates would inspect the same TREE; it says nothing about the
1673
+ // machine they measured it on. A resume is a new session and may have resolved a different
1674
+ // concurrency, so the rows behind a replayed green must also have been measured under the
1675
+ // capacity this session resolved — otherwise a contiguous green prefix spans two worlds.
1676
+ // Rows from before this stamp carry no capacity and replay exactly as they do today.
1677
+ const replayedCapacities = priorEvents
1678
+ .filter((e) => e.event === "gate-result" && e.taskId === t.id && e.data.commit === replayedGates.commit)
1679
+ .map((e) => e.data.capacity);
1680
+ const sameWorld = replayedCapacities.every((recorded) => sameCapacity(recorded, sessionCapacity));
1681
+ if (!sameWorld) {
1682
+ journal.append("gate-replay-capacity-changed", t.id, {
1683
+ commit: replayedGates.commit, recorded: replayedCapacities, resolved: sessionCapacity,
1684
+ });
1685
+ }
1686
+ let reusable = sameWorld && (exactCurrentCommit || canonicalCurrentCommit || recreatedLegacyCommit);
1559
1687
  const reused = [];
1560
1688
  const declaredGates = GATE_NAMES.filter((gate) => t.gates.includes(gate));
1561
1689
  for (const gate of declaredGates) {
@@ -1768,10 +1896,37 @@ export async function runDaemon(repoRoot, opts = {}) {
1768
1896
  // task (journal.ts `outstandingReviewFindings`). Appended row-wise, because this round's own
1769
1897
  // feedback or a repair brief may already quote a finding and repeating it helps no worker.
1770
1898
  const outstandingFindings = outstandingReviewFindings(journaledSoFar, t.id);
1771
- const unquoted = outstandingFindings.filter((f) => !feedback.includes(f.note));
1772
- if (unquoted.length > 0) {
1773
- const brief = ["## Outstanding review findings — a review has NOT passed on these yet",
1774
- ...unquoted.map((f) => `- ${f.path}: ${f.note}`)].join("\n");
1899
+ // T2: the two are different facts about the work and no one heading is true of both. A finding
1900
+ // the reviewer DEFERRED was accepted with a rationale by a review that did not block on it; a
1901
+ // blocking one is still waiting for a review to pass. Filing the deferral under the blocking
1902
+ // heading tells the next worker a passing review is owed on a concern that already drew one —
1903
+ // the exact falsehood this carry exists to remove, restated in the brief that carries it.
1904
+ //
1905
+ // The de-dup below is BLOCKING-ONLY on purpose. A review round's raw bytes quote every finding
1906
+ // it recorded, deferrals included, and those bytes ride into the very next dispatch under the
1907
+ // repair brief's "fix ONLY what these findings name" — so on the ordinary immediate retry the
1908
+ // deferral is already stated, and stated AS BLOCKING. Suppressing its heading there because it
1909
+ // is "already quoted" leaves exactly the falsehood. A quoted BLOCKING finding is quoted
1910
+ // truthfully, so that one still de-dups; a deferral is instead CUT from the raw bytes and
1911
+ // restated once, under the only heading true of it.
1912
+ const deferredRows = outstandingFindings.filter(isDeferredFinding);
1913
+ const withoutDeferrals = (text) => deferredRows.reduce((brief, finding) => brief.replaceAll(renderStructuredReviewFinding(finding), ""), text).replace(/\n{3,}/g, "\n\n").trim();
1914
+ feedback = withoutDeferrals(feedback);
1915
+ if (repairFindings !== undefined)
1916
+ repairFindings = withoutDeferrals(repairFindings);
1917
+ const briefs = [
1918
+ [OUTSTANDING_FINDINGS_HEADING, outstandingFindings.filter((f) => !isDeferredFinding(f) && !feedback.includes(f.note))],
1919
+ [DEFERRED_FINDINGS_HEADING, deferredRows],
1920
+ ];
1921
+ for (const [heading, rows] of briefs) {
1922
+ if (rows.length === 0)
1923
+ continue;
1924
+ const brief = [heading, ...rows.map((f) => {
1925
+ const rationale = f.rationale === undefined
1926
+ ? ""
1927
+ : `\n Rationale: ${f.rationale}`;
1928
+ return `- ${f.path}: ${f.note}${rationale}`;
1929
+ })].join("\n");
1775
1930
  feedback = feedback ? `${feedback}\n\n${brief}` : brief;
1776
1931
  }
1777
1932
  retryMode = repairFindings
@@ -2088,6 +2243,7 @@ export async function runDaemon(repoRoot, opts = {}) {
2088
2243
  // subprocess tree that REACHED its exit marker, not about what the worker claimed.
2089
2244
  let processExited = false;
2090
2245
  let earlyLaunchDead = false;
2246
+ let deadWorkerPark;
2091
2247
  let settleParsed;
2092
2248
  let seedResult;
2093
2249
  // v1.22 T5 / OBS-19: auto-answer a fingerprint-matched trust dialog exactly once per slot.
@@ -2227,6 +2383,8 @@ export async function runDaemon(repoRoot, opts = {}) {
2227
2383
  let quotaStreak = 0;
2228
2384
  let rowSaturationHeld = false; // journaled once per attempt when the kill stands down
2229
2385
  let cpuHeld = false; // likewise for the CPU leg's stand-down (OBS-548)
2386
+ let paneReadHeld = false;
2387
+ let paneStatusHeld = false;
2230
2388
  while (Date.now() - lastProgressAt < stallWindowMs) {
2231
2389
  const sliceStart = Date.now();
2232
2390
  const remaining = stallWindowMs - (sliceStart - lastProgressAt);
@@ -2254,7 +2412,39 @@ export async function runDaemon(repoRoot, opts = {}) {
2254
2412
  break;
2255
2413
  }
2256
2414
  }
2257
- const paneText = await driver.read(slot, PANE_READ_ROWS);
2415
+ // A failed pane read is absence of evidence, never evidence of an absent pane. Keep the
2416
+ // rolling taskTimeoutMinutes window as the backstop and name the held probe once.
2417
+ let paneText;
2418
+ try {
2419
+ paneText = await driver.read(slot, PANE_READ_ROWS);
2420
+ }
2421
+ catch (error) {
2422
+ // Preserve the pre-existing exception path when the independent status probe still
2423
+ // sees a pane. The outer attempt finally owns accountant cleanup on that path. Only
2424
+ // an undetectable status makes the read failure relevant to the death detector, and
2425
+ // that genuinely unmeasurable pair fails open to the rolling timeout.
2426
+ let paneUndetectable = true;
2427
+ try {
2428
+ paneUndetectable = await driver.status(slot) === "unknown";
2429
+ }
2430
+ catch {
2431
+ // Two unreadable pane probes are still unmeasurable, never proof of death.
2432
+ }
2433
+ if (!paneUndetectable)
2434
+ throw error;
2435
+ if (!paneReadHeld) {
2436
+ paneReadHeld = true;
2437
+ journal.append("worker-dead-held", t.id, {
2438
+ slot: slot.name, attempt, reason: "pane-read-unreadable",
2439
+ error: error instanceof Error ? error.message : String(error),
2440
+ });
2441
+ }
2442
+ await armCpuLeg(false);
2443
+ const spent = Date.now() - sliceStart;
2444
+ if (spent < slice)
2445
+ await new Promise((r) => setTimeout(r, Math.min(slice - spent, 1_000)));
2446
+ continue;
2447
+ }
2258
2448
  if (paneText.length > 0)
2259
2449
  everHadOutput = true;
2260
2450
  // OBS-117 (v1.71 T6): zero raw output by the early-launch deadline is a dead channel now.
@@ -2323,7 +2513,24 @@ export async function runDaemon(repoRoot, opts = {}) {
2323
2513
  // gate held. page on "idle" too: herdr's blocked-scrape is strict and proved flaky
2324
2514
  // for TUI dialogs (live check: cursor's trust dialog scraped as idle).
2325
2515
  // "unknown"/"working" never page.
2326
- const st = await driver.status(slot);
2516
+ let st;
2517
+ try {
2518
+ st = await driver.status(slot);
2519
+ }
2520
+ catch (error) {
2521
+ if (!paneStatusHeld) {
2522
+ paneStatusHeld = true;
2523
+ journal.append("worker-dead-held", t.id, {
2524
+ slot: slot.name, attempt, reason: "pane-status-unreadable",
2525
+ error: error instanceof Error ? error.message : String(error),
2526
+ });
2527
+ }
2528
+ await armCpuLeg(false);
2529
+ const spent = Date.now() - sliceStart;
2530
+ if (spent < slice)
2531
+ await new Promise((r) => setTimeout(r, Math.min(slice - spent, 1_000)));
2532
+ continue;
2533
+ }
2327
2534
  if (st !== lastStatus) {
2328
2535
  lastStatus = st;
2329
2536
  journal.append("worker-status", t.id, { slot: slot.name, status: st, attempt });
@@ -2363,6 +2570,95 @@ export async function runDaemon(repoRoot, opts = {}) {
2363
2570
  // daemon has nothing left to do — the pane falls back under the fast-kill and page
2364
2571
  // watchdogs like any other, instead of riding the whole rolling window untended.
2365
2572
  const nudgePending = nudgeable && (!nudged || nudgeDeadline !== undefined);
2573
+ // OBS-737: disposition for the one fully measured death state. `unknown` alone is not
2574
+ // absence (status parsing can fail), and an empty read alone is not absence (a live pane
2575
+ // can be quiet); together they are the existing driver-level pane absence witness. The
2576
+ // process probe and both worktree observations retain their own third states. Only the
2577
+ // explicit conjunction parks; every other state falls through to the unchanged rolling
2578
+ // timeout below. Seeded launches are excluded because their process marker is knowingly
2579
+ // unmeasurable (armCpuLeg documents that contract above).
2580
+ const paneAbsentCandidate = st === "unknown" && paneText.trim().length === 0;
2581
+ // A subprocess can exit between waitOutput and read while its stdout is still draining.
2582
+ // Confirm emptiness in this same poll before paying for `ps`; then confirm once more
2583
+ // after the process probe yielded the event loop. Any bytes or read error withdraw the
2584
+ // absence witness, so a fast completed worker cannot be parked in that drain race.
2585
+ let paneAbsent = paneAbsentCandidate;
2586
+ if (paneAbsent) {
2587
+ try {
2588
+ const confirmation = await driver.read(slot, PANE_READ_ROWS);
2589
+ paneAbsent = confirmation.trim().length === 0;
2590
+ if (!paneAbsent) {
2591
+ everHadOutput = true;
2592
+ if (stallProgress.observe({ paneText: confirmation, contextTokens }))
2593
+ lastProgressAt = Date.now();
2594
+ }
2595
+ }
2596
+ catch (error) {
2597
+ paneAbsent = false;
2598
+ if (!paneReadHeld) {
2599
+ paneReadHeld = true;
2600
+ journal.append("worker-dead-held", t.id, {
2601
+ slot: slot.name, attempt, reason: "pane-read-unreadable",
2602
+ error: error instanceof Error ? error.message : String(error),
2603
+ });
2604
+ }
2605
+ }
2606
+ }
2607
+ const processTree = paneAbsent && worktreeSinceLaunch === "unchanged" && !hasSeed
2608
+ ? await observeWorkerProcessTree(dispatchScript, wt)
2609
+ : "unmeasurable";
2610
+ if (processTree === "empty") {
2611
+ try {
2612
+ const confirmation = await driver.read(slot, PANE_READ_ROWS);
2613
+ paneAbsent = confirmation.trim().length === 0;
2614
+ if (!paneAbsent) {
2615
+ everHadOutput = true;
2616
+ if (stallProgress.observe({ paneText: confirmation, contextTokens }))
2617
+ lastProgressAt = Date.now();
2618
+ }
2619
+ }
2620
+ catch (error) {
2621
+ paneAbsent = false;
2622
+ if (!paneReadHeld) {
2623
+ paneReadHeld = true;
2624
+ journal.append("worker-dead-held", t.id, {
2625
+ slot: slot.name, attempt, reason: "pane-read-unreadable",
2626
+ error: error instanceof Error ? error.message : String(error),
2627
+ });
2628
+ }
2629
+ }
2630
+ }
2631
+ const worktreeDelta = processTree === "empty"
2632
+ && paneAbsent ? await observeWorktreeDelta(taskBase, wt)
2633
+ : "unreadable";
2634
+ // The first process snapshot can race a just-starting child after the dispatch pane
2635
+ // disappeared. Re-read it after the pane and worktree legs have both held: preservation
2636
+ // is terminal, so a process appearing in that interval must withdraw the park rather
2637
+ // than be orphaned by it. The final worktree recheck remains inside preserveDeadWorker.
2638
+ const confirmedProcessTree = worktreeDelta === "unchanged"
2639
+ ? await observeWorkerProcessTree(dispatchScript, wt)
2640
+ : "unmeasurable";
2641
+ const deathCertain = paneAbsent
2642
+ && processTree === "empty"
2643
+ && confirmedProcessTree === "empty"
2644
+ && worktreeDelta === "unchanged";
2645
+ if (deathCertain) {
2646
+ const preservation = await preserveDeadWorker(wt, taskBase);
2647
+ if (!preservation.ref) {
2648
+ journal.append("worker-dead-held", t.id, {
2649
+ slot: slot.name, attempt, reason: `worktree-${preservation.state}`,
2650
+ });
2651
+ continue;
2652
+ }
2653
+ const ref = preservation.ref;
2654
+ const reason = `worker is unambiguously dead: pane absent, process tree empty, and worktree unchanged; preserved at ${ref}`;
2655
+ deadWorkerPark = { ref, reason };
2656
+ journal.append("worktree-preserved", t.id, { ref });
2657
+ journal.append("worker-dead-held", t.id, {
2658
+ slot: slot.name, attempt, reason: "unambiguous-worker-death", ref,
2659
+ });
2660
+ break;
2661
+ }
2366
2662
  // T1 review fix: the kill's "no output growth" leg clocks off the RAW growth signals,
2367
2663
  // never lastProgressAt alone — the flat-token rule (stall.ts) deliberately suppresses
2368
2664
  // the re-arm report on row growth once tokens stick, and contextTokens is sticky across
@@ -2536,7 +2832,13 @@ export async function runDaemon(repoRoot, opts = {}) {
2536
2832
  if (!finished && exitCode === null) {
2537
2833
  // timed out (or only ever saw false positives): harvest whatever the pane holds now
2538
2834
  timedOut = Date.now() - lastProgressAt >= stallWindowMs;
2539
- output = await driver.read(slot, PANE_READ_ROWS);
2835
+ try {
2836
+ output = await driver.read(slot, PANE_READ_ROWS);
2837
+ }
2838
+ catch {
2839
+ // The poll loop already recorded the unreadable pane. Retain the last readable bytes
2840
+ // so this ambiguous path still reaches the ordinary timeout/consult backstop.
2841
+ }
2540
2842
  finished = new RegExp(trailerPattern(nonce)).test(output);
2541
2843
  const exit = exitRe.exec(output);
2542
2844
  exitCode = exit ? Number(exit[1]) : null;
@@ -2670,6 +2972,10 @@ export async function runDaemon(repoRoot, opts = {}) {
2670
2972
  tokens = addUsage(tokens, attemptUsage);
2671
2973
  metered++;
2672
2974
  }
2975
+ if (deadWorkerPark) {
2976
+ await park(t, deadWorkerPark.reason, "stall", assignment, attempt + 1, startMs, gateFails, consults, tokens, metered, retryMode, { ref: deadWorkerPark.ref });
2977
+ return;
2978
+ }
2673
2979
  let result = settleParsed ?? adapter.parse(output, nonce);
2674
2980
  const workerFinished = finished;
2675
2981
  const workerCause = classifyWorkerResultCause({ output, ok: result.ok, finished, exitCode, summary: result.summary, timedOut, deadChannel: deadChannelKilled });
package/dist/run/git.d.ts CHANGED
@@ -34,6 +34,54 @@ export declare const deriveForkCap: (concurrency: number, cores?: number) => num
34
34
  export declare const runWithForkBudget: <T>(concurrency: number, fn: () => Promise<T>) => Promise<T>;
35
35
  /** The cap owned by the run on this async context; the standalone default outside one. */
36
36
  export declare const resolvedForkCap: () => string;
37
+ /**
38
+ * T7: the CAPACITY a suite verdict was measured under — the fork cap the command's child actually
39
+ * received, and the core count that cap was divided from. Two verdicts are comparable only when both
40
+ * numbers match: a run resumed at a different concurrency divides the same machine by a different
41
+ * number, so a green measured in that other world is not evidence about this one.
42
+ *
43
+ * This pair is the WHOLE comparable identity, and the load averages a gate row already carries beside
44
+ * it are deliberately NOT part of it — no reader below ever feeds a load sample into the comparison.
45
+ * The capacity is deterministic and resolved here, where the child's environment is built. The load
46
+ * endpoints are neither: they are two samples taken at a gate's boundaries, and a gate's INTERIOR is
47
+ * invisible to them — on this milestone's own run a gate's interior reached well over twice what
48
+ * either of its own endpoints saw. So matching capacity establishes only that two measurements
49
+ * divided the same machine by the same number. It says nothing about whether the machine was calm.
50
+ */
51
+ export interface RunCapacity {
52
+ forkCap: number;
53
+ cores: number;
54
+ }
55
+ export type CapacityRead = {
56
+ state: "present";
57
+ capacity: RunCapacity;
58
+ } | {
59
+ state: "absent";
60
+ } | {
61
+ state: "malformed";
62
+ };
63
+ /**
64
+ * Three states, never two. A record carrying NO capacity is an older record from before this stamp
65
+ * existed: it keeps exactly the verdict it has today. A record carrying a capacity it cannot state —
66
+ * half the pair, an empty container, a zero, a negative, an unparseable value — is a NEWER record
67
+ * that is malformed, and reading it as an older one is how a fail-closed guard stops firing silently.
68
+ */
69
+ export declare function readCapacity(value: unknown): CapacityRead;
70
+ /**
71
+ * May a verdict recorded under `recorded` be reused — forgiven, cached, replayed — by a session
72
+ * running under `current`? Absent → yes, unchanged. Present and identical → yes. Malformed, a
73
+ * different capacity, or a current capacity the caller could not state → no.
74
+ */
75
+ export declare function sameCapacity(recorded: unknown, current: RunCapacity | undefined): boolean;
76
+ export declare const describeCapacity: (value: unknown) => string;
77
+ /**
78
+ * The capacity a child spawned on THIS async context would receive: the same precedence `shell`
79
+ * applies below — an operator export of the cap wins over the run's own derived value — beside the
80
+ * cores it was divided from. A caller holding a command's own result reads the capacity off THAT
81
+ * result (`ShResult.capacity`, stamped where the child's environment was built); this is for the
82
+ * decisions taken BEFORE any child exists — a cache hit, a reuse predicate.
83
+ */
84
+ export declare const resolvedCapacity: () => RunCapacity;
37
85
  /** The shipped shell ceiling: the fallback every caller gets when nothing measured a better one. */
38
86
  export declare const DEFAULT_SHELL_TIMEOUT_MS = 600000;
39
87
  export interface ShResult {
@@ -42,6 +90,8 @@ export interface ShResult {
42
90
  stderr: string;
43
91
  timedOut?: boolean;
44
92
  durationMs?: number;
93
+ /** T7: the capacity THIS child ran under, stamped where its environment was built (see `shell`). */
94
+ capacity?: RunCapacity;
45
95
  }
46
96
  export declare const setSpawnForTests: (fn: typeof spawn) => void;
47
97
  export declare const resetSpawnForTests: () => void;