tickmarkr 1.97.0 → 2.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (47) hide show
  1. package/README.md +19 -1
  2. package/dist/brand.d.ts +28 -0
  3. package/dist/brand.js +41 -0
  4. package/dist/cli/commands/compile.js +32 -1
  5. package/dist/cli/commands/doctor.d.ts +19 -0
  6. package/dist/cli/commands/doctor.js +59 -0
  7. package/dist/cli/commands/init.js +5 -2
  8. package/dist/cli/commands/resume.js +25 -8
  9. package/dist/cli/commands/run.d.ts +61 -1
  10. package/dist/cli/commands/run.js +372 -18
  11. package/dist/cli/commands/status.js +145 -28
  12. package/dist/cli/index.d.ts +1 -1
  13. package/dist/cli/index.js +1 -1
  14. package/dist/compile/collateral.d.ts +25 -0
  15. package/dist/compile/collateral.js +46 -11
  16. package/dist/compile/native.js +10 -0
  17. package/dist/config/config.d.ts +1 -0
  18. package/dist/config/config.js +2 -2
  19. package/dist/drivers/herdr.d.ts +19 -13
  20. package/dist/drivers/herdr.js +90 -26
  21. package/dist/drivers/index.d.ts +5 -1
  22. package/dist/drivers/index.js +16 -1
  23. package/dist/drivers/orca.d.ts +189 -0
  24. package/dist/drivers/orca.js +879 -0
  25. package/dist/drivers/types.d.ts +2 -0
  26. package/dist/gates/acceptance.js +17 -7
  27. package/dist/gates/llm.d.ts +19 -0
  28. package/dist/gates/llm.js +104 -6
  29. package/dist/gates/run-gates.d.ts +18 -0
  30. package/dist/gates/run-gates.js +195 -29
  31. package/dist/gates/scope.d.ts +9 -1
  32. package/dist/gates/scope.js +22 -2
  33. package/dist/graph/graph.d.ts +1 -0
  34. package/dist/graph/graph.js +19 -2
  35. package/dist/report/compare.js +17 -2
  36. package/dist/run/daemon.d.ts +1 -8
  37. package/dist/run/daemon.js +231 -246
  38. package/dist/run/environment.d.ts +18 -1
  39. package/dist/run/environment.js +19 -2
  40. package/dist/run/journal.d.ts +50 -3
  41. package/dist/run/journal.js +181 -5
  42. package/dist/run/protocol.d.ts +4 -4
  43. package/dist/run/stall.d.ts +30 -0
  44. package/dist/run/stall.js +173 -0
  45. package/dist/tui/ink/init-app.js +14 -4
  46. package/package.json +1 -1
  47. package/skills/tickmarkr-overseer/SKILL.md +27 -15
@@ -8,26 +8,30 @@ import { classifyDeadChannel, NO_TRAILER_SUMMARY, trailerPattern, UNPARSEABLE_TR
8
8
  import { allAdapters, getAdapter, probeAll, readDoctor, rolePools } from "../adapters/registry.js";
9
9
  import { addUsage, channelKey, matchesTrustDialog, QUOTA_RE } from "../adapters/types.js";
10
10
  import { bannerShell, paneDispatchCommand } from "../brand.js";
11
+ import { collateralHits } from "../compile/collateral.js";
11
12
  import { DEFAULT_DIFF_CAP, globalConfigDir, loadConfigWithMode, readOverlayFile, repoOverlayPath, } from "../config/config.js";
12
13
  import { DeliveryReadinessError } from "../drivers/herdr.js";
13
14
  import { herdrSealShellPrefix, SubprocessDriver } from "../drivers/subprocess.js";
14
15
  import { formatOwnedName } from "../drivers/types.js";
15
16
  import { captureBaseline, detectGateCommands, detectVacuousOracles } from "../gates/baseline.js";
16
17
  import { runGates } from "../gates/run-gates.js";
17
- import { addEvidence, attributeBlocked, blockedTasks, getTask, graphDefinitionHash, loadGraph, pendingTasks, readyTasks, saveGraph, setStatus } from "../graph/graph.js";
18
+ import { addEvidence, attributeBlocked, blockedTasks, getTask, graphDefinitionHash, loadGraph, pendingTasks, readyTasks, saveGraph, setStatus, taskContentDigest } from "../graph/graph.js";
18
19
  import { GATE_NAMES } from "../graph/schema.js";
19
20
  import { augmentRetryBrief, consult, renderRetryGuidance } from "./consult.js";
20
21
  import { runEnvironment } from "./environment.js";
21
22
  import { cleanupRunWorktrees, gitHead, linkNodeModules, npmDependencyInstallCommand, npmDependencyManifestChanged, runWithForkBudget, sh, shGit, WORKTREE_LAYOUT_CONTRACT, worktreePath } from "./git.js";
22
23
  import { runInteractiveSeed } from "./interactive-seed.js";
23
- import { activeRetryBan, classifyTaskFailure, classifyWorkerResultCause, engagementComparable, GATE_FINGERPRINT_CAP, GATE_SATISFIED_RELEASE, identicalGateFailures, journaledFailureBrief, Journal, loadRoutingProfile, newRunId, normalizeGateFailure, pendingRepairFindings, phaseForGate, recordedTaskFailureKind, repairsSinceApproval, reviewRoundsSinceApproval, runHasEnded, structuredFindings, upheldFeedbackByTask } from "./journal.js";
24
+ import { activeRetryBan, classifyTaskFailure, classifyWorkerResultCause, engagementComparable, formatPriorFindingEvidence, GATE_FINGERPRINT_CAP, GATE_SATISFIED_RELEASE, identicalGateFailures, journaledFailureBrief, Journal, loadRoutingProfile, newRunId, normalizeGateFailure, pendingRepairFindings, phaseForGate, readPriorRunEvidence, recordedTaskFailureKind, repairsSinceApproval, reviewRoundsSinceApproval, runHasEnded, structuredFindings, upheldFeedbackByTask } from "./journal.js";
24
25
  import { isDiffCapPark } from "../gates/review.js";
25
26
  import { acquireApprovalSerialization, acquireRunLock, releaseRunLock } from "./lock.js";
26
27
  import { ensureIntegration, integrationBranch, integrationHead, mergeTask, verifyIntegrationTip } from "./merge.js";
27
28
  import { nextChannel, route } from "../route/router.js";
28
29
  import { desiredPanes } from "./reconcile.js";
29
- import { NUDGEABLE_ADAPTERS, PANE_READ_ROWS, StallProgressTracker, stallSnapshotBannerRows } from "./stall.js";
30
+ import { harvestCpuFlatWindowMs, NUDGEABLE_ADAPTERS, PANE_READ_ROWS, StallProgressTracker, stallSnapshotBannerRows, WorkerTreeCpuAccountant, } from "./stall.js";
30
31
  import { armSupervision } from "./supervision.js";
32
+ // Compatibility exports for the daemon liveness tests and existing consumers. The implementation
33
+ // lives in stall.ts so gate dispatch can depend on it without importing the daemon.
34
+ export { harvestCpuFlatWindowMs, resetHarvestCpuFlatMsForTests, setHarvestCpuFlatMsForTests, workerTreeCpuMs, } from "./stall.js";
31
35
  const MODE_RANK = { "staff-led": 0, "risk-based": 1, "partner-led": 2 };
32
36
  // An override (flag/spec) re-resolves through loadConfigWithMode itself, via a synthesized repo overlay
33
37
  // carrying routing.mode — floors, explore, lints, and provenance all come from config.ts's preset
@@ -182,6 +186,10 @@ function narrowRepairBattery(failing) {
182
186
  return isOracleFailure(g) && g.meta?.unparseable !== true;
183
187
  return false;
184
188
  }
189
+ // v2.0 T2 (OBS-554): the measurement keys run-gates stamps on a GateResult, lifted verbatim onto the
190
+ // gate row. One list, one lift — both onGate sites record through the same helper.
191
+ const GATE_TELEMETRY_KEYS = ["durationMs", "load1Start", "load1End", "selectedDurationMs", "fullDurationMs", "invocations"];
192
+ const gateMeasurement = (meta = {}) => Object.fromEntries(GATE_TELEMETRY_KEYS.filter((k) => meta[k] !== undefined).map((k) => [k, meta[k]]));
185
193
  /**
186
194
  * T4 (OBS-265): the journal with the review objections a round did NOT hinge on removed. Judge and
187
195
  * review are now launched together, so a round can journal a failed review that the serial walk would
@@ -350,40 +358,6 @@ export function setHarvestSilentMsForTests(ms) {
350
358
  export function resetHarvestSilentMsForTests() {
351
359
  harvestSilentMs = HARVEST_SILENT_MS;
352
360
  }
353
- // A CPU delta needs two samples separated in WALL CLOCK, and the CPU clock is QUANTIZED: darwin's
354
- // `ps` prints hundredths ("0:00.03"), linux's prints whole seconds ("00:00:01"). Equality across a
355
- // window shorter than the quantum is not evidence of anything — a worker throttled to a low duty
356
- // cycle accrues less than one tick per sample and reads flat while genuinely working. So the flat
357
- // observation must span the LARGER of a floor and this many ticks of the clock actually in use:
358
- // crossing 30 ticks means the tree burned <1 tick in 30, i.e. under ~3% of one core. On a
359
- // hundredths host that is a 3s window; on a whole-second host it is 30s — still nothing against the
360
- // ~15m redispatch it replaces. Resolution is read off the sampled rows, never assumed.
361
- const HARVEST_CPU_FLAT_MS = 3_000;
362
- const HARVEST_CPU_FLAT_TICKS = 30;
363
- // Once the flat window opens, retain descendants often enough to observe brief tool processes that
364
- // can start and exit between the daemon's ordinary wait slices. This sampler exists only during an
365
- // eligible silence window; it is stopped on progress or as soon as the worker wait concludes.
366
- const HARVEST_CPU_ACCOUNTING_POLL_MS = 100;
367
- // T2 review (material): that 100ms cadence forks a shell plus `ps` ten times a second, and on a host
368
- // where `ps` is unsupported or denied (the managed-sandbox class) EVERY sample fails — tens of
369
- // thousands of processes per silent attempt, multiplied by daemon concurrency, for a probe that can
370
- // never conclude anything. Persistent failure is structural, not transient, so the sampler STOPS
371
- // after this many consecutive unreadable snapshots. It stays stopped for the silence window it was
372
- // started for: read() then reports no CPU, the triad refuses to conclude and journals the gap, and a
373
- // later window (after real progress clears the accountant) starts a fresh one that pays the same
374
- // bounded probe again.
375
- const HARVEST_CPU_UNMEASURABLE_SAMPLE_CAP = 20;
376
- let harvestCpuFlatMs;
377
- export function harvestCpuFlatWindowMs(resolutionMs) {
378
- return harvestCpuFlatMs ?? Math.max(HARVEST_CPU_FLAT_MS, resolutionMs * HARVEST_CPU_FLAT_TICKS);
379
- }
380
- /** Test seam — pin the flat window so a probe case need not sit through a real one. */
381
- export function setHarvestCpuFlatMsForTests(ms) {
382
- harvestCpuFlatMs = ms;
383
- }
384
- export function resetHarvestCpuFlatMsForTests() {
385
- harvestCpuFlatMs = undefined;
386
- }
387
361
  // Once the silence gate is met the CPU probe owns the poll cadence: the trailer-wait slice is 30s,
388
362
  // so two samples would otherwise cost a minute of wall clock apiece. Below the gate the only rule
389
363
  // is not to sleep PAST it — at the shipped 5m gate that changes no slice a worker sees today.
@@ -509,167 +483,6 @@ export async function verifyIntegrationTipCached(intWt, commands, journal, opts
509
483
  }
510
484
  return tipFailed;
511
485
  }
512
- // `ps` CPU time: "[[dd-]hh:]mm:ss[.frac]" (darwin prints "0:00.03", linux "00:00:01", both print
513
- // "1-02:03:04" past a day). Anything else is a header or a row this parser must not guess at.
514
- // `frac` reports whether THIS host prints sub-second digits — the quantum the flat window is sized
515
- // against, measured rather than assumed (a darwin sample is 10ms, a linux one 1000ms).
516
- function parsePsCpu(raw) {
517
- const m = /^(?:(\d+)-)?(?:(\d+):)?(\d+):(\d+(?:\.\d+)?)$/.exec(raw);
518
- if (!m)
519
- return undefined;
520
- const ms = ((Number(m[1] ?? 0) * 24 + Number(m[2] ?? 0)) * 60 + Number(m[3])) * 60_000 + Math.round(Number(m[4]) * 1000);
521
- return { ms, frac: m[4].includes(".") };
522
- }
523
- let linuxClockTickMs;
524
- function linuxProcessCpuMs(pid, cwd) {
525
- if (!existsSync("/proc/self/stat"))
526
- return Promise.resolve(undefined);
527
- // shGit, not sh: the accountant samples this path at a 100ms cadence, and a LOGIN shell would
528
- // re-run the operator's profile (nvm/pyenv/direnv side effects included) on every sample.
529
- linuxClockTickMs ??= shGit("getconf CLK_TCK", cwd, 15_000).then((r) => {
530
- const ticks = r.code === 0 ? Number(r.stdout.trim()) : Number.NaN;
531
- return Number.isFinite(ticks) && ticks > 0 ? 1_000 / ticks : undefined;
532
- });
533
- return linuxClockTickMs.then((resolutionMs) => {
534
- if (resolutionMs === undefined)
535
- return undefined;
536
- try {
537
- // `/proc/<pid>/stat` fields 14-17 are user/system jiffies for the process and its waited-for
538
- // children. The child totals retain tools that start and exit wholly between live-tree polls.
539
- // Split after the LAST ')' because comm may contain spaces or parentheses; field 3 is rest[0].
540
- const stat = readFileSync(`/proc/${pid}/stat`, "utf8");
541
- const fields = stat.slice(stat.lastIndexOf(")") + 2).trim().split(/\s+/);
542
- const ticks = Number(fields[11]) + Number(fields[12]) + Number(fields[13]) + Number(fields[14]);
543
- return Number.isFinite(ticks) ? { ms: ticks * resolutionMs, resolutionMs } : undefined;
544
- }
545
- catch {
546
- return undefined; // process exited between ps ancestry capture and the precise CPU read
547
- }
548
- });
549
- }
550
- // T2 (OBS-264): the triad's CPU leg. Every non-seeded process of an attempt descends from that
551
- // attempt's own dispatch script, whose path is unique — print, argv-interactive and resume launches
552
- // all start there. (interactive-seed is intentionally fail-open below because its adapter-owned
553
- // launch bypasses this script.) ONE `ps` snapshot finds the root and all current descendants:
554
- // the agent CLI is a CHILD of the script's shell, so the root's own TIME never moves while the CLI
555
- // thinks. `resolutionMs` is the sampled clock's quantum, which sizes the caller's flat window.
556
- // Returns 0 when nothing matches: a worker whose process tree is gone is the strongest possible
557
- // "not working". Returns undefined when the snapshot itself failed or parsed to nothing —
558
- // unmeasurable CPU is never evidence a worker stopped, and the caller refuses to conclude on it.
559
- async function workerTreeCpuSnapshot(marker, cwd) {
560
- // shGit, not sh: same login-shell cost as the CLK_TCK probe above — `ps` needs no profile.
561
- const snapshot = await shGit("ps -Awwo pid=,ppid=,time=,command=", cwd, 15_000);
562
- if (snapshot.code !== 0)
563
- return undefined;
564
- const rows = [];
565
- for (const line of snapshot.stdout.split("\n")) {
566
- const m = /^\s*(\d+)\s+(\d+)\s+(\S+)\s+(.*)$/.exec(line);
567
- if (!m)
568
- continue;
569
- const cpu = parsePsCpu(m[3]);
570
- if (cpu !== undefined)
571
- rows.push({ pid: m[1], ppid: m[2], cpuMs: cpu.ms, frac: cpu.frac, cmd: m[4] });
572
- }
573
- if (rows.length === 0)
574
- return undefined;
575
- const tree = new Set(rows.filter((p) => p.cmd.includes(marker)).map((p) => p.pid));
576
- // `ps` output is not topologically ordered — relax the parent→child closure until it stops growing.
577
- for (let grew = true; grew;) {
578
- grew = false;
579
- for (const p of rows) {
580
- if (!tree.has(p.pid) && tree.has(p.ppid)) {
581
- tree.add(p.pid);
582
- grew = true;
583
- }
584
- }
585
- }
586
- const precise = new Map();
587
- let preciseResolutionMs;
588
- for (const p of rows) {
589
- if (!tree.has(p.pid))
590
- continue;
591
- const cpu = await linuxProcessCpuMs(p.pid, cwd);
592
- precise.set(p.pid, cpu?.ms ?? p.cpuMs);
593
- if (cpu !== undefined)
594
- preciseResolutionMs = cpu.resolutionMs;
595
- }
596
- // Even an empty worker tree needs the host's actual measurement quantum: on Linux the /proc
597
- // jiffy clock remains available after the worker exits, while `ps time` only prints whole seconds.
598
- if (preciseResolutionMs === undefined && existsSync("/proc/self/stat")) {
599
- preciseResolutionMs = (await linuxProcessCpuMs(String(process.pid), cwd))?.resolutionMs;
600
- }
601
- return {
602
- processes: precise,
603
- resolutionMs: preciseResolutionMs ?? (rows.some((p) => p.frac) ? 10 : 1_000),
604
- };
605
- }
606
- export async function workerTreeCpuMs(marker, cwd) {
607
- const snapshot = await workerTreeCpuSnapshot(marker, cwd);
608
- if (snapshot === undefined)
609
- return undefined;
610
- return {
611
- ms: [...snapshot.processes.values()].reduce((sum, cpuMs) => sum + cpuMs, 0),
612
- resolutionMs: snapshot.resolutionMs,
613
- };
614
- }
615
- // Sparse live-tree totals forget a tool's CPU as soon as that tool exits. This attempt-local
616
- // accountant instead adds each observed process's CPU DELTA to a monotonic total and replaces only
617
- // the live-PID cursor on each sample. When a PID disappears, its contribution stays in `totalMs`;
618
- // if that PID is later reused, its fresh total is added from zero because it left `live` in between.
619
- class WorkerTreeCpuAccountant {
620
- marker;
621
- cwd;
622
- active = false;
623
- loop;
624
- live = new Map();
625
- totalMs = 0;
626
- gaps = 0;
627
- consecutiveGaps = 0;
628
- latest;
629
- constructor(marker, cwd) {
630
- this.marker = marker;
631
- this.cwd = cwd;
632
- }
633
- async sample() {
634
- const snapshot = await workerTreeCpuSnapshot(this.marker, this.cwd);
635
- if (snapshot === undefined) {
636
- this.gaps++;
637
- this.live.clear();
638
- this.latest = undefined;
639
- // Stop forking `ps` at 10Hz once the host has proved it cannot answer — see the cap's comment.
640
- if (++this.consecutiveGaps >= HARVEST_CPU_UNMEASURABLE_SAMPLE_CAP)
641
- this.active = false;
642
- return;
643
- }
644
- this.consecutiveGaps = 0;
645
- for (const [pid, cpuMs] of snapshot.processes) {
646
- const prior = this.live.get(pid);
647
- this.totalMs += prior === undefined || cpuMs < prior ? cpuMs : cpuMs - prior;
648
- }
649
- this.live = snapshot.processes;
650
- this.latest = { ms: this.totalMs, resolutionMs: snapshot.resolutionMs };
651
- }
652
- async start() {
653
- if (this.active)
654
- return;
655
- this.active = true;
656
- await this.sample();
657
- this.loop = (async () => {
658
- while (this.active) {
659
- await new Promise((resolve) => setTimeout(resolve, HARVEST_CPU_ACCOUNTING_POLL_MS));
660
- if (this.active)
661
- await this.sample();
662
- }
663
- })();
664
- }
665
- read() {
666
- return { cpu: this.latest, gaps: this.gaps };
667
- }
668
- async stop() {
669
- this.active = false;
670
- await this.loop;
671
- }
672
- }
673
486
  async function commitsAheadOf(base, wt) {
674
487
  const head = await gitHead(wt);
675
488
  if (head === base)
@@ -925,6 +738,11 @@ export async function runDaemon(repoRoot, opts = {}) {
925
738
  let releaseApprovalSerialization;
926
739
  try {
927
740
  let graph = loadGraph(repoRoot);
741
+ // One bounded snapshot supplies every dispatch. On resume the current journal still participates
742
+ // in the fold so its task-done/ordinary-approval events can retire older evidence, but findings it
743
+ // produced are suppressed: same-run feedback already replays those bytes and must not deliver them
744
+ // twice. Evidence from genuinely earlier runs remains available to a resumed fresh dispatch.
745
+ const priorRunEvidence = readPriorRunEvidence(repoRoot, graph.tasks, { suppressRunId: runId });
928
746
  // GATE-FIX-4 defect 4 (no-op run refusal): a fresh run on a graph with nothing dispatchable used
929
747
  // to journal {run-start, run-end} with zero dispatches — and downstream readers (greenness exit,
930
748
  // status, notify) treat that run-end as completion, so an all-terminal graph "went green" having
@@ -1159,6 +977,19 @@ export async function runDaemon(repoRoot, opts = {}) {
1159
977
  for (const w of await detectVacuousOracles(repoRoot, graph.tasks))
1160
978
  journal.append("baseline-warning", w.taskId, { ...w });
1161
979
  }
980
+ // OBS-547: ONE full per-task collateral prediction for the whole run, computed here — uncapped, and
981
+ // handed whole to every scope gate below (never recomputed at a red, never the plan's 20-item view).
982
+ // Persisted beside baseline.json and RELOADED on resume (same reason baseline is): the prediction is
983
+ // pre-dispatch state, and rescanning a repository the run has since edited would let an offender flip
984
+ // between predicted and missed across stop/amend/resume. Missing file (pre-OBS-547 journal) ⇒ scan and
985
+ // pin it now, so every later resume of this run agrees with this one.
986
+ const collateralPath = join(journal.dir, "collateral.json");
987
+ const pinnedCollateral = opts.resume && existsSync(collateralPath)
988
+ ? new Map(Object.entries(JSON.parse(readFileSync(collateralPath, "utf8"))))
989
+ : null;
990
+ const collateral = pinnedCollateral ?? collateralHits(graph.tasks, repoRoot);
991
+ if (!pinnedCollateral)
992
+ writeFileSync(collateralPath, JSON.stringify(Object.fromEntries(collateral), null, 2));
1162
993
  // T6: open the narrator AFTER run-start/run-resume is journaled so the watch surface has a run to
1163
994
  // show. driver.narrator is undefined on subprocess → no-op (subprocess spawns nothing). Swallowed:
1164
995
  // a failed-to-open or later-dead watch pane never affects the run.
@@ -1249,11 +1080,60 @@ export async function runDaemon(repoRoot, opts = {}) {
1249
1080
  saveGraph(repoRoot, graph);
1250
1081
  journal.append("task-human", t.id, { reason, kind });
1251
1082
  if (assignment) {
1252
- journal.telemetry({ taskId: t.id, shape: t.shape, adapter: assignment.adapter, model: assignment.model, channel: assignment.channel, attempts, outcome: "human", durationMs: Date.now() - startMs, parkKind: kind, gateFails, consults, tokens, meteredAttempts: tokens ? metered : undefined, retryMode });
1083
+ // OBS-547: `metered` counts CHARGEABLE metered attempts, so an unchargeable dispatch passes 0 and
1084
+ // the count is omitted rather than written as 0 or as `1` beside `attempts: 0` — a row claiming
1085
+ // more metered attempts than it charges reports as "floor: 1/0 attempts metered". tokens stay
1086
+ // (the spend was real); tokens-without-a-count is the already-modelled degraded ⇒ floor row.
1087
+ journal.telemetry({ taskId: t.id, shape: t.shape, adapter: assignment.adapter, model: assignment.model, channel: assignment.channel, attempts, outcome: "human", durationMs: Date.now() - startMs, parkKind: kind, gateFails, consults, tokens, meteredAttempts: tokens && metered ? metered : undefined, retryMode });
1253
1088
  }
1254
1089
  await reconcile({ spareLiveLlm: true }); // task-human is a terminal event — sweep, sparing sibling tasks' live LLM panes
1255
1090
  await driver.notify(`tickmarkr ${runId}: ${t.id} needs a human — ${reason}`, { tier: "attention" });
1256
1091
  };
1092
+ // OBS-547: cross-reference a scope red against the prediction this run already computed. Every hard
1093
+ // offender predicted ⇒ an AUTHORING defect whose repair is pre-written: journal the classification
1094
+ // with the verbatim files[] lines and park unchargeable. It owes nothing to the quality machinery —
1095
+ // no chargeable attempt, no gateFails, no escalation, no ladder rung — and `scope-authoring` is what
1096
+ // makes a resume replay agree (journal.ts replayResumeState). One unpredicted offender keeps today's
1097
+ // chargeable behaviour and records the miss, so the lint's blind spots accumulate as evidence rather
1098
+ // than folklore. ONE disposition for BOTH paths that observe a red — the ordinary attempt and the
1099
+ // resume gate replay: which code path noticed the red must never decide who pays for it. Returns
1100
+ // true when it parked; the caller then returns without charging anything.
1101
+ // `attempts` is the count of CHARGEABLE attempts before this dispatch — never this dispatch's own
1102
+ // count. The two callers arrive at it differently (the attempt loop's index already excludes the
1103
+ // dispatch in flight; a gate replay's rs.attempts already includes it), and the ordinal journaled
1104
+ // below is derived from it here so both paths number the same dispatch identically.
1105
+ const dispositionScopeRed = async (t, results, assignment, attempts, startMs, gateFails, consults, tokens, metered, retryMode) => {
1106
+ const scopeRed = results.find((g) => g.gate === "scope" && gateFailed(g));
1107
+ const verdict = scopeRed?.meta?.collateral;
1108
+ if (!verdict)
1109
+ return false;
1110
+ if (verdict.authoring) {
1111
+ journal.append("scope-authoring", t.id, {
1112
+ gate: "scope",
1113
+ predicted: verdict.predicted,
1114
+ repair: verdict.repair,
1115
+ attempt: attempts + 1,
1116
+ chargeable: false,
1117
+ // The dispatch was PHYSICALLY metered even though nobody is charged for it. Physical metering
1118
+ // is recorded here, separately, so the telemetry row can stay chargeable-consistent: passing
1119
+ // this count on would print `meteredAttempts: 1` beside `attempts: 0`.
1120
+ ...(tokens ? { tokens, meteredDispatches: metered } : {}),
1121
+ ...(assignment ? { channel: channelKey(assignment) } : {}),
1122
+ });
1123
+ await driver.notify(`tickmarkr ${runId}: ${t.id} scope red was PREDICTED — authoring defect, no attempt charged`, { tier: "attention" });
1124
+ await park(t, `scope: every out-of-scope path was predicted by the collateral lint before dispatch — authoring defect, not a worker failure. Repair:\n${verdict.repair}`, "authoring", assignment, attempts, startMs, gateFails, consults, tokens, 0, retryMode);
1125
+ return true;
1126
+ }
1127
+ if (verdict.missed.length) {
1128
+ journal.append("collateral-miss", t.id, {
1129
+ gate: "scope",
1130
+ unpredicted: verdict.missed,
1131
+ predicted: verdict.predicted,
1132
+ attempt: attempts + 1,
1133
+ });
1134
+ }
1135
+ return false;
1136
+ };
1257
1137
  const execTask = async (t) => {
1258
1138
  const startMs = Date.now();
1259
1139
  const taskTimeoutMinutes = t.timeoutMinutes ?? cfg.taskTimeoutMinutes;
@@ -1283,6 +1163,7 @@ export async function runDaemon(repoRoot, opts = {}) {
1283
1163
  // fresh-budget release, prefer nextChannel over the surviving tried-list so burned channels are
1284
1164
  // not re-tried first (consult bans / prior failovers survive the release).
1285
1165
  const rs = resume.get(t.id);
1166
+ const contentDigest = taskContentDigest(t);
1286
1167
  if (rs?.lastAssignment && rs.attempts > 0
1287
1168
  && channels.some((c) => channelKey(c) === channelKey(rs.lastAssignment))
1288
1169
  && !demotedChannels.has(channelKey(rs.lastAssignment))) {
@@ -1376,7 +1257,19 @@ export async function runDaemon(repoRoot, opts = {}) {
1376
1257
  ...(g.meta?.fullSuite === true ? { fullSuite: true } : {}),
1377
1258
  // A finding's path is its own evidence path. Do not pass task scope here: a declaration says
1378
1259
  // where work is allowed, not where this verdict found the defect.
1379
- ...(blocking ? { findings: structuredFindings(g.gate, g.details) } : {}),
1260
+ ...(blocking ? {
1261
+ taskContentDigest: contentDigest,
1262
+ findings: structuredFindings(g.gate, g.details),
1263
+ } : {}),
1264
+ // v2.0 T2 (OBS-554): the gate's OWN measurement, lifted verbatim from the meta run-gates
1265
+ // stamped WHERE THE GATE RAN. Nothing here re-derives a duration by subtracting journal
1266
+ // timestamps — that would measure this row's queue as well as its work. It rides the
1267
+ // gate-result row itself because that IS the row a recalibration reads: a measurement kept
1268
+ // in a side stream is one join away from the verdict it explains, and the two can drift.
1269
+ // A gate run-gates did not measure contributes NO field rather than a fabricated zero — for
1270
+ // every recalibration this telemetry funds, a gap is honest and a zero is a lie. The
1271
+ // seven-gate closed set is asserted end-to-end in tests/run/gate-telemetry.test.ts.
1272
+ ...gateMeasurement(g.meta),
1380
1273
  });
1381
1274
  };
1382
1275
  // R3 (OBS-186): judge ‖ review are launched together and publish in COMPLETION order
@@ -1418,9 +1311,18 @@ export async function runDaemon(repoRoot, opts = {}) {
1418
1311
  // in resume state. The journal already holds the upheld review's bytes; no reset of attempt or
1419
1312
  // channel state can take them away, on any path, including `resume --retry-failed`.
1420
1313
  const upheldFeedback = upheldFeedbackByTask(journal.read()).get(t.id) ?? rs?.upheldFeedback;
1421
- let feedback = upheldFeedback
1314
+ const carriedEvidence = priorRunEvidence.findings
1315
+ .filter((finding) => finding.taskId === t.id)
1316
+ .map(formatPriorFindingEvidence)
1317
+ .join("\n\n");
1318
+ const withCarriedEvidence = (brief) => {
1319
+ if (!carriedEvidence || brief.includes(carriedEvidence))
1320
+ return brief;
1321
+ return brief ? `${brief}\n\n${carriedEvidence}` : carriedEvidence;
1322
+ };
1323
+ let feedback = withCarriedEvidence(upheldFeedback
1422
1324
  ? `The operator UPHELD the reviewer's findings — address them without discarding landed work.\nreview: ${upheldFeedback}`
1423
- : "";
1325
+ : "");
1424
1326
  let ladderIdx = 0;
1425
1327
  let modeFallbackNoted = false; // v1.2: journal the interactive→print fallback once per task, not per attempt
1426
1328
  let gateFails = 0; // TEL-02: incremented ONLY where feedback is built from failing gates — never derived from attempts (quota failovers bump attempts too, Pitfall 6)
@@ -1677,6 +1579,7 @@ export async function runDaemon(repoRoot, opts = {}) {
1677
1579
  const { results } = await runGates(resumedTask, {
1678
1580
  worktree: wt, baseRef: taskBase, result: priorResult, author: gateAuthor,
1679
1581
  commands, baseline, channels: pools.review, judgeChannels: pools.judge, adapters, cfg, artifactDir: journal.dir,
1582
+ collateral: collateral.get(t.id) ?? [],
1680
1583
  // a recheck re-verifies a human's release: it never selects tests down, it runs the suite.
1681
1584
  pipeline: "v185",
1682
1585
  via: cfg.visibility.llm === "pane"
@@ -1710,6 +1613,20 @@ export async function runDaemon(repoRoot, opts = {}) {
1710
1613
  graph = addEvidence(graph, t.id, { commits: approvedCommits, gateResults: results });
1711
1614
  saveGraph(repoRoot, graph);
1712
1615
  if (!results.every(gateSatisfied)) {
1616
+ // OBS-547: disposition FIRST, and through the same helper the ordinary attempt path uses. A
1617
+ // crash between a journaled scope red and its classification must not turn a predicted red
1618
+ // into a charged one just because a resume is what observed it.
1619
+ // rs.attempts COUNTS the interrupted dispatch this replay is judging; the chargeable
1620
+ // attempts behind it are one fewer. Passing the count itself would number the same dispatch
1621
+ // one higher here than on the ordinary path and bill an attempt the fresh path forgives.
1622
+ // UNLESS an earlier resume already journaled the classification and died before its park:
1623
+ // replayResumeState() has then ALREADY taken that dispatch back, so subtracting again would
1624
+ // erase an EARLIER chargeable attempt and attribute the park to the assignment the rewind
1625
+ // restored. Ask the journal which state this is, and take the classified dispatch's own
1626
+ // assignment back with it.
1627
+ const classified = journal.classifiedDispatch(t.id);
1628
+ if (await dispositionScopeRed(t, results, classified?.assignment ?? gateAuthor, classified ? (rs?.attempts ?? 0) : Math.max(0, (rs?.attempts ?? 0) - 1), startMs, gateFails, consults, tokens, metered, retryMode))
1629
+ return;
1713
1630
  gateFails++;
1714
1631
  // Observed green gates are only measurements. If the resumed suffix is red, preserve that
1715
1632
  // result in the journal and return to the ordinary attempt/consult ladder, which rebuilds
@@ -1734,7 +1651,9 @@ export async function runDaemon(repoRoot, opts = {}) {
1734
1651
  }
1735
1652
  graph = setStatus(graph, t.id, "done");
1736
1653
  saveGraph(repoRoot, graph);
1737
- journal.append("task-done", t.id, { attempts: rs?.attempts ?? 0, assignment: gateAuthor });
1654
+ journal.append("task-done", t.id, {
1655
+ attempts: rs?.attempts ?? 0, assignment: gateAuthor, taskContentDigest: contentDigest,
1656
+ });
1738
1657
  journal.append("merge", t.id, { branch: taskBranch, commit: await integrationHead(intWt) });
1739
1658
  journal.telemetry({
1740
1659
  taskId: t.id, shape: t.shape, adapter: gateAuthor.adapter, model: gateAuthor.model,
@@ -1956,7 +1875,7 @@ export async function runDaemon(repoRoot, opts = {}) {
1956
1875
  const promptFile = writePrompt(journal.dir, t, attempt, feedback, nonce);
1957
1876
  // OBS-56: state the non-interactive, one-pass finish contract and the OBS-54 stall budget in every
1958
1877
  // worker prompt, not only consult retry guidance. Prepended so prompt.ts's completion trailer stays last.
1959
- const workerContract = `## Harness contract\n- This harness is non-interactive: make one continuous pass; do not stop for questions or follow-up input.\n- You have a ${taskTimeoutMinutes} minute stall window. Budget the full suite once, then commit and emit the completion trailer before it expires.\n- Each test: acceptance criterion must exist as a vitest test whose OWN title (the leaf, not counting enclosing describe titles) is the criterion string verbatim — never shortened, never decorated. Nesting under describe() is allowed.`; // OBS-64; OBS-511: leaf-title rule stated where the worker reads it
1878
+ const workerContract = `## Harness contract\n- This harness is non-interactive: make one continuous pass; do not stop for questions or follow-up input.\n- You have a ${taskTimeoutMinutes} minute stall window. The gates run the full suite for you — never spend this window on one; commit your work and emit the completion trailer inside it.\n- Each test: acceptance criterion must exist as a vitest test whose OWN title (the leaf, not counting enclosing describe titles) is the criterion string verbatim — never shortened, never decorated. Nesting under describe() is allowed.`; // OBS-64; OBS-511: leaf-title rule stated where the worker reads it; OBS-548: the suite is the GATES' job — this repo's own suite outlasts the output-silence windows the daemon polices, so a worker obeying "budget the full suite" was killed by construction
1960
1879
  // OBS-47: state the worktree layout contract in the worker prompt (cheap-tier workers were
1961
1880
  // committing/deleting node_modules and tripping the scope gate). The harness re-asserts the link
1962
1881
  // itself before gates regardless of what the worker does with it.
@@ -2023,30 +1942,34 @@ export async function runDaemon(repoRoot, opts = {}) {
2023
1942
  unmeasurableNoted = true;
2024
1943
  journal.append("worker-harvest-unmeasurable", t.id, { slot: slot.name, attempt, reason });
2025
1944
  };
2026
- const harvestConcludes = async (silentMs) => {
2027
- if (silentMs < harvestSilentMs) {
1945
+ // The CPU leg needs a marker in the worker's own argv, and every launch path puts this
1946
+ // attempt's dispatch script there EXCEPT interactiveSeed: runInteractiveSeed launches the
1947
+ // TUI directly, by a command the ADAPTER owns (seed.launch(model)) which tickmarkr cannot
1948
+ // make attempt-unique and must deliver verbatim. A marker that matches nothing reads as
1949
+ // zero CPU — precisely the false "flat" that would harvest a worker mid-turn — so a seeded
1950
+ // attempt has no measurable CPU leg and neither reader ever concludes it. The other half of
1951
+ // OBS-264 is untouched there: when its window does expire with commits on the worktree, the
1952
+ // no-trailer tail gates them instead of buying a fresh worker to re-produce them.
1953
+ const armCpuLeg = async (armed) => {
1954
+ if (!armed) {
2028
1955
  await cpuAccountant?.stop();
2029
1956
  cpuAccountant = undefined;
2030
1957
  cpuFlat = undefined;
2031
1958
  cpuGapCount = 0;
2032
- return false;
1959
+ return;
2033
1960
  }
2034
- // The CPU leg needs a marker in the worker's own argv, and every launch path puts this
2035
- // attempt's dispatch script there EXCEPT interactiveSeed: runInteractiveSeed launches the
2036
- // TUI directly, by a command the ADAPTER owns (seed.launch(model)) which tickmarkr cannot
2037
- // make attempt-unique and must deliver verbatim. A marker that matches nothing reads as
2038
- // zero CPU — precisely the false "flat" that would harvest a worker mid-turn — so a seeded
2039
- // attempt has no measurable CPU leg and the triad never concludes it. The other half of
2040
- // OBS-264 is untouched there: when its window does expire with commits on the worktree, the
2041
- // no-trailer tail gates them instead of buying a fresh worker to re-produce them.
1961
+ if (hasSeed || cpuAccountant !== undefined)
1962
+ return;
1963
+ cpuAccountant = new WorkerTreeCpuAccountant(dispatchScript, wt);
1964
+ await cpuAccountant.start();
1965
+ };
1966
+ const readCpuLeg = () => {
2042
1967
  if (hasSeed) {
2043
1968
  noteUnmeasurable("interactive-seed launch is not in the probed process tree");
2044
- return false;
2045
- }
2046
- if (cpuAccountant === undefined) {
2047
- cpuAccountant = new WorkerTreeCpuAccountant(dispatchScript, wt);
2048
- await cpuAccountant.start();
1969
+ return { state: "unmeasurable" };
2049
1970
  }
1971
+ if (cpuAccountant === undefined)
1972
+ return { state: "unmeasurable" }; // no window open: nothing measured yet
2050
1973
  const observation = cpuAccountant.read();
2051
1974
  if (observation.gaps !== cpuGapCount) {
2052
1975
  cpuGapCount = observation.gaps;
@@ -2060,21 +1983,29 @@ export async function runDaemon(repoRoot, opts = {}) {
2060
1983
  // structural hole as the seeded launch, and must not be the one that stays silent.
2061
1984
  cpuFlat = undefined;
2062
1985
  noteUnmeasurable("the worker process snapshot could not be read");
2063
- return false;
1986
+ return { state: "unmeasurable" };
2064
1987
  }
2065
1988
  const now = Date.now();
2066
1989
  if (cpu.ms !== cpuFlat?.ms) {
2067
1990
  cpuFlat = { ms: cpu.ms, since: now };
2068
- return false;
1991
+ return { state: "accruing" };
2069
1992
  }
2070
- if (now - cpuFlat.since < harvestCpuFlatWindowMs(cpu.resolutionMs))
1993
+ return now - cpuFlat.since < harvestCpuFlatWindowMs(cpu.resolutionMs)
1994
+ ? { state: "accruing" }
1995
+ : { state: "flat", cpu };
1996
+ };
1997
+ const harvestConcludes = async (silentMs) => {
1998
+ if (silentMs < harvestSilentMs)
1999
+ return false;
2000
+ const leg = readCpuLeg();
2001
+ if (leg.state !== "flat")
2071
2002
  return false;
2072
2003
  const carried = await commitsAheadOf(taskBase, wt);
2073
2004
  if (carried.length === 0)
2074
2005
  return false; // nothing landed: not this branch's population
2075
2006
  journal.append("worker-harvest", t.id, {
2076
2007
  slot: slot.name, attempt, commits: carried.length,
2077
- silentMs, cpuMs: cpu.ms, cpuResolutionMs: cpu.resolutionMs,
2008
+ silentMs, cpuMs: leg.cpu.ms, cpuResolutionMs: leg.cpu.resolutionMs,
2078
2009
  });
2079
2010
  return true;
2080
2011
  };
@@ -2117,6 +2048,12 @@ export async function runDaemon(repoRoot, opts = {}) {
2117
2048
  let output;
2118
2049
  let exitCode;
2119
2050
  let timedOut = false;
2051
+ // OBS-548 addendum: the mechanism that fired is what the repair brief and the consult read.
2052
+ // A fast-kill lands with the rolling window nowhere near expiry (measured: 693 s of silence
2053
+ // inside a 1,800,000 ms window), so the trailer-less tail's stall-timeout fallback named a
2054
+ // window that could not have killed anything and briefed the next worker to fix a stall that
2055
+ // never happened. Fourth conflation in the taxonomy OBS-53 opened.
2056
+ let deadChannelKilled = false;
2120
2057
  // T2 review: print mode's "the exit marker appeared". Kept apart from `finished` (the
2121
2058
  // trailer) but still needed by the keepPanes decision below, whose contract is about a
2122
2059
  // subprocess tree that REACHED its exit marker, not about what the worker claimed.
@@ -2152,7 +2089,7 @@ export async function runDaemon(repoRoot, opts = {}) {
2152
2089
  keptSlots.push(slot);
2153
2090
  else
2154
2091
  await closeSlot(slot);
2155
- feedback = `delivery readiness failed after ${error.waitedMs}ms; pane transcript:\n${error.transcript}`;
2092
+ feedback = withCarriedEvidence(`delivery readiness failed after ${error.waitedMs}ms; pane transcript:\n${error.transcript}`);
2156
2093
  const step = r.ladder[Math.min(ladderIdx++, r.ladder.length - 1)];
2157
2094
  journal.append("escalation", t.id, { step, attempt: attempt + 1 });
2158
2095
  await driver.notify(`tickmarkr ${runId}: ${t.id} escalation: ${step}`, { tier: "attention" });
@@ -2260,6 +2197,7 @@ export async function runDaemon(repoRoot, opts = {}) {
2260
2197
  // lastProgressAt — see the kill below.
2261
2198
  let quotaStreak = 0;
2262
2199
  let rowSaturationHeld = false; // journaled once per attempt when the kill stands down
2200
+ let cpuHeld = false; // likewise for the CPU leg's stand-down (OBS-548)
2263
2201
  while (Date.now() - lastProgressAt < stallWindowMs) {
2264
2202
  const sliceStart = Date.now();
2265
2203
  const remaining = stallWindowMs - (sliceStart - lastProgressAt);
@@ -2396,6 +2334,30 @@ export async function runDaemon(repoRoot, opts = {}) {
2396
2334
  // daemon has nothing left to do — the pane falls back under the fast-kill and page
2397
2335
  // watchdogs like any other, instead of riding the whole rolling window untended.
2398
2336
  const nudgePending = nudgeable && (!nudged || nudgeDeadline !== undefined);
2337
+ // T1 review fix: the kill's "no output growth" leg clocks off the RAW growth signals,
2338
+ // never lastProgressAt alone — the flat-token rule (stall.ts) deliberately suppresses
2339
+ // the re-arm report on row growth once tokens stick, and contextTokens is sticky across
2340
+ // read misses, so a metered non-nudgeable adapter (pi) streaming rows under a stale
2341
+ // counter presented a frozen lastProgressAt and was killed mid-work. lastRowGrowthAt is
2342
+ // recorded on every high-water advance, suppressed or not; token growth already rides
2343
+ // lastProgressAt. Either one advancing is output growth.
2344
+ const lastOutputGrowthAt = Math.max(stallProgress.lastRowGrowthAt ?? 0, lastProgressAt);
2345
+ // Everything the fast-kill can decide from the CHANNEL, evaluated before the CPU leg is
2346
+ // asked for anything: the kill's own probe costs a `ps` every 100ms, so an ineligible
2347
+ // slice must not pay for it. The CPU reading is consulted at the kill itself, below.
2348
+ const fastKillEligible = !stallProgress.rowSignalSaturated
2349
+ && !nudgePending && !nudgeFailed
2350
+ && sliceNow - lastOutputGrowthAt >= deadChannelFastKillMs
2351
+ && worktreeSinceLaunch === "unchanged";
2352
+ const harvestCpuEligible = !nudgePending && !nudgeFailed
2353
+ && sliceNow - lastProgressAt >= harvestSilentMs;
2354
+ // OBS-548: ONE accountant serves both readers, so its lifecycle is decided once per
2355
+ // slice from BOTH windows. Arming per-reader let the harvest's own "not silent enough
2356
+ // yet" branch stop the sampler the fast-kill had just started, and a restarted
2357
+ // accountant re-bases its monotonic total — the leg could then never read flat. The
2358
+ // accountant also carries the readers' nudge holds: when neither reader may conclude,
2359
+ // there is no reason to fork its 10 Hz process snapshots through the remaining window.
2360
+ await armCpuLeg(harvestCpuEligible || fastKillEligible);
2399
2361
  // T2 (OBS-264): the liveness triad CONCLUDES the wait on finished work — commits ahead
2400
2362
  // of the task base (this attempt's own AND any carried forward — see the eligibility
2401
2363
  // comment at the dispatch site), a flat worker-tree CPU delta, and >= harvestSilentMs
@@ -2408,10 +2370,15 @@ export async function runDaemon(repoRoot, opts = {}) {
2408
2370
  // claude-code worker before the rescue could fire, leaving T1's nudge dead code for
2409
2371
  // exactly the committed-and-stalled population OBS-264 is about. Holding concludes at
2410
2372
  // ~14m (nudge + grace) rather than ~36m — nearly all of the OBS-264 win, and a worker
2411
- // that only needed a submit answers with a full trailer instead of partial work. An
2412
- // ANSWERED or twice-undeliverable nudge leaves nothing pending, so the triad governs
2413
- // again; the hold is on a pending daemon ACTION, never on the adapter being nudgeable.
2414
- if ((!nudgePending || nudgeFailed) && await harvestConcludes(sliceNow - lastProgressAt))
2373
+ // that only needed a submit answers with a full trailer instead of partial work.
2374
+ // An ANSWERED nudge leaves nothing pending, so the triad governs again; the hold is on a
2375
+ // pending daemon ACTION, never on the adapter being nudgeable.
2376
+ // OBS-548: an UNDELIVERABLE nudge holds too. A worker inside one long foreground command
2377
+ // has no input box, so it is the population that CANNOT be nudged and is also the one
2378
+ // most likely to be legitimately silent — letting its delivery failure lift the hold made
2379
+ // the failure itself the trigger. `:357-359` already reads a false driver.nudge return as
2380
+ // a delivery outcome, "not proof of an unreachable channel"; unreachable is not at rest.
2381
+ if (!nudgePending && !nudgeFailed && await harvestConcludes(sliceNow - lastProgressAt))
2415
2382
  break;
2416
2383
  // T1 (R1 dead-channel fast-kill): no trailer, an unchanged launch tree, and no output growth
2417
2384
  // for the fast-kill window — the channel is dead, so conclude NOW
@@ -2443,20 +2410,30 @@ export async function runDaemon(repoRoot, opts = {}) {
2443
2410
  rowSaturationHeld = true;
2444
2411
  journal.append("worker-dead-held", t.id, { slot: slot.name, attempt, reason: "row-signal-saturated" });
2445
2412
  }
2446
- // T1 review fix: the kill's "no output growth" leg clocks off the RAW growth signals,
2447
- // never lastProgressAt alone — the flat-token rule (stall.ts) deliberately suppresses
2448
- // the re-arm report on row growth once tokens stick, and contextTokens is sticky across
2449
- // read misses, so a metered non-nudgeable adapter (pi) streaming rows under a stale
2450
- // counter presented a frozen lastProgressAt and was killed mid-work. lastRowGrowthAt is
2451
- // recorded on every high-water advance, suppressed or not; token growth already rides
2452
- // lastProgressAt. Either one advancing is output growth.
2453
- const lastOutputGrowthAt = Math.max(stallProgress.lastRowGrowthAt ?? 0, lastProgressAt);
2454
- if (!stallProgress.rowSignalSaturated
2455
- && (!nudgePending || nudgeFailed)
2456
- && sliceNow - lastOutputGrowthAt >= deadChannelFastKillMs
2457
- && worktreeSinceLaunch === "unchanged") {
2458
- journal.append("worker-dead", t.id, { slot: slot.name, attempt, silentMs: sliceNow - lastOutputGrowthAt });
2459
- break;
2413
+ // OBS-548: the FOURTH leg, and the one the daemon already measured. The three legs above
2414
+ // are all channel-side: they say a pane stopped talking. The tree's CPU says whether
2415
+ // anything is still WORKING, and the harvest triad forty lines up refuses exactly this
2416
+ // conclusion without it — "a worker that is merely thinking still burns CPU and is never
2417
+ // concluded here". A live CPU delta is a live worker, by construction; an UNMEASURABLE
2418
+ // reading keeps the fail-open contract the probe already states (undefined is never
2419
+ // evidence a worker stopped), so on a host whose `ps` cannot be read the fast-kill stands
2420
+ // down and the rolling window owns the pane, exactly as pre-T1.
2421
+ if (fastKillEligible) {
2422
+ const leg = readCpuLeg();
2423
+ if (leg.state === "flat") {
2424
+ deadChannelKilled = true;
2425
+ journal.append("worker-dead", t.id, {
2426
+ slot: slot.name, attempt, silentMs: sliceNow - lastOutputGrowthAt,
2427
+ cpuMs: leg.cpu.ms, cpuResolutionMs: leg.cpu.resolutionMs,
2428
+ });
2429
+ break;
2430
+ }
2431
+ if (!cpuHeld) {
2432
+ cpuHeld = true;
2433
+ journal.append("worker-dead-held", t.id, {
2434
+ slot: slot.name, attempt, reason: leg.state === "accruing" ? "cpu-accruing" : "cpu-unmeasurable",
2435
+ });
2436
+ }
2460
2437
  }
2461
2438
  if (nudgeable && !nudged && sliceNow - lastProgressAt >= nudgeAfterSilentMs) {
2462
2439
  nudged = true;
@@ -2623,6 +2600,7 @@ export async function runDaemon(repoRoot, opts = {}) {
2623
2600
  // steer, driver.nudge is never consulted on this path), so there is no pending daemon
2624
2601
  // action for the triad to preempt — the asymmetry with the interactive call site above is
2625
2602
  // the absence of the thing being held for, not an oversight.
2603
+ await armCpuLeg(Date.now() - lastProgressAt >= harvestSilentMs);
2626
2604
  if (await harvestConcludes(Date.now() - lastProgressAt))
2627
2605
  break;
2628
2606
  }
@@ -2665,7 +2643,7 @@ export async function runDaemon(repoRoot, opts = {}) {
2665
2643
  }
2666
2644
  let result = settleParsed ?? adapter.parse(output, nonce);
2667
2645
  const workerFinished = finished;
2668
- const workerCause = classifyWorkerResultCause({ output, ok: result.ok, finished, exitCode, summary: result.summary, timedOut });
2646
+ const workerCause = classifyWorkerResultCause({ output, ok: result.ok, finished, exitCode, summary: result.summary, timedOut, deadChannel: deadChannelKilled });
2669
2647
  journal.append("worker-result", t.id, {
2670
2648
  ok: result.ok, summary: result.summary, deviations: result.deviations, finished: workerFinished, exitCode,
2671
2649
  mode: interactive ? "interactive" : "print", ...(workerCause ? { cause: workerCause } : {}),
@@ -2888,6 +2866,7 @@ export async function runDaemon(repoRoot, opts = {}) {
2888
2866
  ({ results, commits } = await runGates(t, {
2889
2867
  worktree: wt, baseRef: taskBase, result, author: assignment,
2890
2868
  commands, baseline, channels: pools.review, judgeChannels: pools.judge, adapters, cfg, artifactDir: journal.dir,
2869
+ collateral: collateral.get(t.id) ?? [],
2891
2870
  pipeline: "v185", selectTests: !testGateFailed,
2892
2871
  via: cfg.visibility.llm === "pane"
2893
2872
  ? {
@@ -2930,7 +2909,9 @@ export async function runDaemon(repoRoot, opts = {}) {
2930
2909
  }
2931
2910
  graph = setStatus(graph, t.id, "done");
2932
2911
  saveGraph(repoRoot, graph);
2933
- journal.append("task-done", t.id, { attempts: attempt + 1, assignment });
2912
+ journal.append("task-done", t.id, {
2913
+ attempts: attempt + 1, assignment, taskContentDigest: contentDigest,
2914
+ });
2934
2915
  journal.append("merge", t.id, { branch: taskBranch, commit: await integrationHead(intWt) });
2935
2916
  // firstAttemptOk/gateFails/consults are recorded FACTS, not policy — a parkKind:"stall" row is
2936
2917
  // recorded but NOT quality-negative in v1.6; Phase 12 owns reward policy, so flipping it later needs zero data migration.
@@ -2962,11 +2943,15 @@ export async function runDaemon(repoRoot, opts = {}) {
2962
2943
  await park(t, `${infraFailure.gate}: ${infraFailure.details}`, "infra", assignment, attempt + 1, startMs, gateFails, consults, tokens, metered, retryMode);
2963
2944
  return;
2964
2945
  }
2946
+ // OBS-547: who pays for this red is decided by the run's collateral prediction (see
2947
+ // dispositionScopeRed) — an authoring defect parks unchargeable before any accounting below.
2948
+ if (await dispositionScopeRed(t, results, assignment, attempt, startMs, gateFails, consults, tokens, metered, retryMode))
2949
+ return;
2965
2950
  gateFails++; // this attempt's gates failed — the one place quality degradation is verified (never inferred from attempts)
2966
2951
  // v1.53 T3: prefer the CLI's own session id captured from this attempt's output (kimi's resume
2967
2952
  // trailer) over the harness slot name; absent hook or no capture keeps today's slot-name id.
2968
2953
  retrySession = { channel: channelKey(assignment), id: adapter.sessionIdFrom?.(output) ?? sessionId, contextTokens };
2969
- feedback = results.filter(gateFailed).map((g) => `${g.gate}: ${g.details}`).join("\n\n");
2954
+ feedback = withCarriedEvidence(results.filter(gateFailed).map((g) => `${g.gate}: ${g.details}`).join("\n\n"));
2970
2955
  // OBS-189/G3 (park-economics patch): a request-changes review is a findings brief, not a worker
2971
2956
  // defect — the fix attempt stays on the same channel with the findings as feedback and consumes
2972
2957
  // no escalation-ladder rung. Bounded by the engagement round cap at the top of this loop.