tickmarkr 2.1.3 → 2.1.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,5 +1,5 @@
1
1
  import { type Assignment, type BillingChannel, type WorkerAdapter } from "../adapters/types.js";
2
- import { type TickmarkrConfig } from "../config/config.js";
2
+ import { type TickmarkrConfig, type Tier } from "../config/config.js";
3
3
  import { type Task } from "../graph/schema.js";
4
4
  import { type GateVia } from "./llm.js";
5
5
  import type { GateResult } from "./types.js";
@@ -49,6 +49,13 @@ export declare function isDiffCapPark(result: GateResult): boolean;
49
49
  export declare function diffCapParkReason(results: GateResult[]): string | null;
50
50
  export declare function modelId(model: string): string;
51
51
  export declare function pickReviewer(author: Assignment, channels: BillingChannel[], exclude?: string[], // v1.1 failover: reviewer channels that already produced garbage for this task
52
- prefer?: string[]): BillingChannel | null;
52
+ prefer?: string[], // v1.53 T2: review.prefer — reorders eligible channels, never changes eligibility
53
+ floor?: Tier): BillingChannel | null;
53
54
  export type ReviewUnparseableCause = VerdictUnparseableCause;
55
+ /**
56
+ * This shows the reviewer what the task DECLARED, never what the diff may actually reach. The diff
57
+ * remains a separate stated input, so whether its touched paths fit the declaration stays a reviewer
58
+ * judgement rather than a guarantee made by this renderer.
59
+ */
60
+ export declare function renderDeclaredWriteScope(files: ReadonlyArray<string>): string;
54
61
  export declare function reviewGate(task: Task, worktree: string, baseRef: string, author: Assignment, channels: BillingChannel[], adapters: WorkerAdapter[], cfg: TickmarkrConfig, via?: GateVia, excludeReviewers?: string[], artifactDir?: string): Promise<GateResult>;
@@ -170,7 +170,8 @@ function reviewPreferIndex(c, prefer) {
170
170
  return i === -1 ? prefer.length : i;
171
171
  }
172
172
  export function pickReviewer(author, channels, exclude = [], // v1.1 failover: reviewer channels that already produced garbage for this task
173
- prefer = []) {
173
+ prefer = [], // v1.53 T2: review.prefer — reorders eligible channels, never changes eligibility
174
+ floor) {
174
175
  // FLEET-05 success criterion 2: an author not resolvable in the channel list yields NO reviewer.
175
176
  // The old `?? author.adapter` fallback compared an adapter id to vendor names, matched nothing, and
176
177
  // admitted every reviewer — including the author's own channel (fail-OPEN). null lands on reviewGate's
@@ -182,9 +183,25 @@ prefer = []) {
182
183
  // two independent axes: different vendor AND different base-model identity (ADDED TO the vendor
183
184
  // rule, never replacing it — a future edit can't silently drop either). The diversity filter runs
184
185
  // BEFORE preference ranking: prefer sorts survivors only, so no entry can resurrect an excluded channel.
185
- .filter((c) => c.vendor !== authorChannel.vendor && modelId(c.model) !== modelId(author.model) && !exclude.includes(channelKey(c)))
186
+ .filter((c) => c.vendor !== authorChannel.vendor
187
+ && modelId(c.model) !== modelId(author.model)
188
+ && !exclude.includes(channelKey(c))
189
+ && (floor === undefined || TIER_RANK[c.tier] >= TIER_RANK[floor]))
186
190
  .sort((a, b) => reviewPreferIndex(a, prefer) - reviewPreferIndex(b, prefer) || TIER_RANK[b.tier] - TIER_RANK[a.tier] || marginalCostRank(a) - marginalCostRank(b))[0] ?? null);
187
191
  }
192
+ /**
193
+ * This shows the reviewer what the task DECLARED, never what the diff may actually reach. The diff
194
+ * remains a separate stated input, so whether its touched paths fit the declaration stays a reviewer
195
+ * judgement rather than a guarantee made by this renderer.
196
+ */
197
+ export function renderDeclaredWriteScope(files) {
198
+ if (files.length === 0) {
199
+ return "## Declared write scope\nUnrestricted: this task declared no write-scope patterns.";
200
+ }
201
+ return `## Declared write scope
202
+ The task DECLARED these write-scope patterns:
203
+ ${files.map((path) => `- ${path}`).join("\n")}`;
204
+ }
188
205
  export async function reviewGate(task, worktree, baseRef, author, channels, adapters, cfg, via, excludeReviewers,
189
206
  // OBS-196: run dir for raw-output persistence on an unparseable verdict; absent (older callers,
190
207
  // direct tests) skips persistence and changes nothing else.
@@ -257,13 +274,19 @@ artifactDir) {
257
274
  policy: "full",
258
275
  ...(promotedBy ? { promotedFrom: declaredPolicy, promotedBy } : {}),
259
276
  };
260
- const reviewer = pickReviewer(author, channels, excludeReviewers ?? [], cfg.review.prefer ?? []);
277
+ // A reviewer floor is opt-in at the task. Applying cfg.routing.floors here would change the
278
+ // historical seat for every task that never asked for review-tier coupling.
279
+ const reviewerFloor = task.routingHints?.floor;
280
+ const reviewer = pickReviewer(author, channels, excludeReviewers ?? [], cfg.review.prefer ?? [], reviewerFloor);
261
281
  if (!reviewer) {
262
282
  // meta.noEligibleReviewer lets run-gates' review-retry keep the ORIGINAL unparseable result when
263
283
  // the retry finds no second seat — a truthful cause beats a synthetic no-reviewer failure.
284
+ const reason = reviewerFloor
285
+ ? `no cross-vendor reviewer available at or above task-declared ${reviewerFloor} floor (diversity rule)`
286
+ : "no cross-vendor reviewer available (diversity rule)";
264
287
  return cfg.review.required
265
- ? { gate: "review", pass: false, details: "no cross-vendor reviewer available (diversity rule); set review.required:false to waive", meta: { noEligibleReviewer: true } }
266
- : { gate: "review", pass: true, details: "WARNING: no cross-vendor reviewer available — review waived by config", meta: { noEligibleReviewer: true } };
288
+ ? { gate: "review", pass: false, details: `${reason}; set review.required:false to waive`, meta: { noEligibleReviewer: true, ...(reviewerFloor ? { reviewerFloor } : {}) } }
289
+ : { gate: "review", pass: true, details: `WARNING: ${reason} — review waived by config`, meta: { noEligibleReviewer: true, ...(reviewerFloor ? { reviewerFloor } : {}) } };
267
290
  }
268
291
  const measuredDiff = await fetchTaskDiff(worktree, baseRef);
269
292
  // Keep the reader payload identical to the text charged to the strict cap:
@@ -284,6 +307,8 @@ ${COMPLETION_FAKING_CHECKLIST}
284
307
  ## Acceptance criteria
285
308
  ${task.acceptance.map((a) => `- ${renderAcceptanceItem(a)}`).join("\n")}
286
309
 
310
+ ${renderDeclaredWriteScope(task.files)}
311
+
287
312
  ## Diff
288
313
  \`\`\`diff
289
314
  ${diff}
@@ -301,7 +326,15 @@ Respond with ONLY this JSON:
301
326
  Approve iff no material finding remains; an empty findings list is a clean approval.
302
327
  The top-level comments array is optional. Use it only for actionable line-anchored feedback.
303
328
  `;
304
- const raw = await runLlm(getAdapter(reviewer.adapter, adapters), reviewer.model, prompt, worktree, via ? { driver: via.driver, keep: via.keep, onSlot: via.onSlot, name: via.nameFor("review", reviewer.adapter), label: via.labelFor("review") } : undefined,
329
+ let concludedOnInactivity = false;
330
+ const raw = await runLlm(getAdapter(reviewer.adapter, adapters), reviewer.model, prompt, worktree, via ? {
331
+ driver: via.driver,
332
+ keep: via.keep,
333
+ onSlot: via.onSlot,
334
+ name: via.nameFor("review", reviewer.adapter),
335
+ label: via.labelFor("review"),
336
+ onInactivity: () => { concludedOnInactivity = true; },
337
+ } : undefined,
305
338
  // frontier reviewers routinely need >5min on a configured-cap-sized diff, and `claude -p` buffers all
306
339
  // output until completion — runLlm's 300s default killed reviews mid-flight, returning empty
307
340
  // stdout that read as "unparseable" and escalated to re-implementation of green code
@@ -325,14 +358,22 @@ The top-level comments array is optional. Use it only for actionable line-anchor
325
358
  saved = undefined; // persistence is evidence, not a gate input — never fail the gate on it
326
359
  }
327
360
  }
328
- const failure = cause === "malformed-verdict"
329
- ? "review output unparseable"
330
- : "review dispatch failed — no structurally valid nonce-bound response; output unparseable";
361
+ const failure = concludedOnInactivity
362
+ ? "review dispatch concluded on the inactivity policy without a structurally valid nonce-bound response; output unparseable"
363
+ : cause === "malformed-verdict"
364
+ ? "review output unparseable"
365
+ : "review dispatch failed — no structurally valid nonce-bound response; output unparseable";
331
366
  return {
332
367
  gate: "review",
333
368
  pass: false,
334
369
  details: `${failure} (reviewer ${reviewer.adapter}:${reviewer.model}; cause: ${cause}${saved ? `; raw saved: ${saved}` : ""}) — failing closed`,
335
- meta: { ...policyMeta, reviewer: channelKey(reviewer), unparseable: true, cause },
370
+ meta: {
371
+ ...policyMeta,
372
+ reviewer: channelKey(reviewer),
373
+ unparseable: true,
374
+ cause,
375
+ ...(concludedOnInactivity ? { classification: "infra", infra: true } : {}),
376
+ },
336
377
  };
337
378
  }
338
379
  const decided = findings !== null
@@ -589,7 +589,18 @@ export async function runGates(task, ctx) {
589
589
  const second = await dispatch((adapters) => reviewGate(task, ctx.worktree, ctx.baseRef, ctx.author, ctx.channels, adapters, ctx.cfg, retryVia, [...(ctx.excludeReviewers ?? []), flaked], ctx.artifactDir));
590
590
  if (second.meta?.noEligibleReviewer !== true) {
591
591
  const retried = typeof second.meta?.reviewer === "string" ? second.meta.reviewer : "none";
592
- rv = { ...second, meta: { ...second.meta, reviewRetry: { flaked, retried } } };
592
+ rv = {
593
+ ...second,
594
+ // `details` is lifted onto the journal's gate-result row; meta.reviewRetry is not. Keep the
595
+ // re-route visible in the result text a reader actually opens, including on a red retry.
596
+ details: `review re-route: ${flaked} produced no parseable verdict; replaced by ${retried}\n${second.details}`,
597
+ meta: { ...second.meta, reviewRetry: { flaked, retried } },
598
+ };
599
+ }
600
+ else if (task.routingHints?.floor) {
601
+ // Preserve the original no-answer cause when no replacement exists, but name the declared
602
+ // floor that correctly refused a lower-tier fallback.
603
+ rv = { ...rv, details: `${rv.details}\nreview re-route refused: ${second.details}` };
593
604
  }
594
605
  }
595
606
  return invocations.length ? { ...rv, meta: { ...rv.meta, invocations } } : rv;
@@ -18,7 +18,6 @@ export interface RunOptions {
18
18
  mode?: RoutingMode;
19
19
  narrate?: (event: JournalEvent) => void;
20
20
  exit?: (code: number) => void;
21
- supervise?: boolean;
22
21
  }
23
22
  export type ModeSource = "run flag" | "spec" | "repo config" | "global config" | "default";
24
23
  export interface ResolvedRunMode {
@@ -19,16 +19,15 @@ import { addEvidence, attributeBlocked, blockedTasks, getTask, graphDefinitionHa
19
19
  import { GATE_NAMES } from "../graph/schema.js";
20
20
  import { augmentRetryBrief, consult, renderRetryGuidance } from "./consult.js";
21
21
  import { runEnvironment } from "./environment.js";
22
- import { cleanupRunWorktrees, gitHead, linkNodeModules, npmDependencyInstallCommand, npmDependencyManifestChanged, runWithForkBudget, sh, shGit, WORKTREE_LAYOUT_CONTRACT, worktreePath } from "./git.js";
22
+ import { cleanupRunWorktrees, gitHead, linkNodeModules, npmDependencyInstallCommand, npmDependencyManifestChanged, preserveWorktree, resolvedCapacity, runWithForkBudget, sameCapacity, sh, shGit, WORKTREE_LAYOUT_CONTRACT, worktreePath } from "./git.js";
23
23
  import { runInteractiveSeed } from "./interactive-seed.js";
24
- import { activeRetryBan, classifyTaskFailure, classifyWorkerResultCause, engagementComparable, formatPriorFindingEvidence, GATE_FINGERPRINT_CAP, GATE_SATISFIED_RELEASE, identicalGateFailures, journaledFailureBrief, Journal, loadRoutingProfile, newRunId, normalizeGateFailure, pendingRepairFindings, phaseForGate, readPriorRunEvidence, recordedTaskFailureKind, repairsSinceApproval, reviewRoundsSinceApproval, runHasEnded, structuredFindings, upheldFeedbackByTask } from "./journal.js";
24
+ import { activeRetryBan, classifyTaskFailure, classifyWorkerResultCause, deferredReviewFindings, engagementComparable, formatPriorFindingEvidence, GATE_FINGERPRINT_CAP, GATE_SATISFIED_RELEASE, identicalGateFailures, isDeferredFinding, journaledFailureBrief, Journal, loadRoutingProfile, newRunId, normalizeGateFailure, outstandingReviewFindings, pendingRepairFindings, phaseForGate, readPriorRunEvidence, recordedTaskFailureKind, renderStructuredReviewFinding, repairsSinceApproval, reviewRoundsSinceApproval, runHasEnded, structuredFindings, upheldFeedbackByTask } from "./journal.js";
25
25
  import { isDiffCapPark } from "../gates/review.js";
26
26
  import { acquireApprovalSerialization, acquireRunLock, releaseRunLock } from "./lock.js";
27
27
  import { ensureIntegration, integrationBranch, integrationHead, mergeTask, verifyIntegrationTip } from "./merge.js";
28
28
  import { nextChannel, route } from "../route/router.js";
29
29
  import { desiredPanes } from "./reconcile.js";
30
30
  import { harvestCpuFlatWindowMs, NUDGEABLE_ADAPTERS, PANE_READ_ROWS, StallProgressTracker, stallSnapshotBannerRows, WorkerTreeCpuAccountant, } from "./stall.js";
31
- import { armSupervision } from "./supervision.js";
32
31
  // Compatibility exports for the daemon liveness tests and existing consumers. The implementation
33
32
  // lives in stall.ts so gate dispatch can depend on it without importing the daemon.
34
33
  export { harvestCpuFlatWindowMs, resetHarvestCpuFlatMsForTests, setHarvestCpuFlatMsForTests, workerTreeCpuMs, } from "./stall.js";
@@ -243,6 +242,10 @@ function repairBrief(findings, diff, baseRef) {
243
242
  // v1.70 T5: default request-changes rounds a task may draw before it parks. OBS-419 keeps this as the
244
243
  // no-ceiling behavior; an operator may narrow only the next engagement on the approval that releases it.
245
244
  const REVIEW_ROUND_CAP = 2;
245
+ // T2: one heading per fact. The first is owed a passing review; the second already drew one and was
246
+ // accepted with the reviewer's own rationale, which travels beside (but never changes) its identity.
247
+ const OUTSTANDING_FINDINGS_HEADING = "## Outstanding review findings — a review has NOT passed on these yet";
248
+ const DEFERRED_FINDINGS_HEADING = "## Deferred review findings — a reviewer ACCEPTED these with a rationale and did NOT block on them; do not re-litigate, fix only if your change touches them";
246
249
  // OBS-419: the newest approval starts the current engagement, so it is also the sole authority for
247
250
  // that engagement's optional ceiling. Stop at the newest approval even when the field is absent: a
248
251
  // later ordinary release restores the module default instead of inheriting an older operator limit.
@@ -393,7 +396,7 @@ function lastVerifyCycle(events) {
393
396
  if (e.event === "tip-verify-start") {
394
397
  const { tip, cmdHash } = e.data;
395
398
  cur = typeof tip === "string" && typeof cmdHash === "string"
396
- ? { tip, cmdHash, gates: new Set(), failed: false, forgiven: false }
399
+ ? { tip, cmdHash, gates: new Set(), failed: false, forgiven: false, capacities: [e.data.capacity] }
397
400
  : undefined;
398
401
  afterRunEnd = false;
399
402
  continue;
@@ -411,9 +414,14 @@ function lastVerifyCycle(events) {
411
414
  continue;
412
415
  }
413
416
  if (!cur || afterRunEnd || cur.tip !== tip || cur.cmdHash !== cmdHash) {
414
- cur = { tip, cmdHash, gates: new Set(), failed: false, forgiven: false };
417
+ cur = { tip, cmdHash, gates: new Set(), failed: false, forgiven: false, capacities: [] };
415
418
  }
416
419
  afterRunEnd = false;
420
+ // T7: EVERY verdict row's own capacity, not just the start row's. The start row is a statement of
421
+ // intent written before a single command ran; the green this cache would carry forward lives on
422
+ // these rows, so a row whose capacity differs from the session's — or which is malformed — has to
423
+ // be able to sink the cycle by itself.
424
+ cur.capacities.push(e.data.capacity);
417
425
  if (e.event === "tip-verify-failed")
418
426
  cur.failed = true;
419
427
  else {
@@ -441,23 +449,31 @@ function lastVerifyCycle(events) {
441
449
  export async function verifyIntegrationTipCached(intWt, commands, journal, opts = {}) {
442
450
  const cmdHash = commandsHash(commands);
443
451
  const tip = await gitHead(intWt);
452
+ // T7: the capacity this session's verify children WOULD run under — the third thing a carried
453
+ // green must match, beside the tip and the command set. A cached verdict is the one place a green
454
+ // crosses a session boundary with nothing re-run, and a session resumed at a different concurrency
455
+ // divides the machine by a different number: that green was established in another world, so it is
456
+ // not carried forward and the commands run again. A pre-T7 cycle records no capacity and keeps
457
+ // exactly the behaviour it has today.
458
+ const capacity = resolvedCapacity();
444
459
  const porcelain = await shGit("git status --porcelain", intWt);
445
460
  const clean = porcelain.code === 0 && porcelain.stdout.trim() === "";
446
461
  const last = lastVerifyCycle(journal.read());
447
462
  const cached = last !== undefined && !last.failed && !last.forgiven && last.tip === tip && last.cmdHash === cmdHash
448
- && Object.keys(commands).every((g) => last.gates.has(g));
463
+ && Object.keys(commands).every((g) => last.gates.has(g))
464
+ && last.capacities.every((recorded) => sameCapacity(recorded, capacity));
449
465
  // A pair can be verified red and then green without either SHA or command hash changing (for
450
466
  // example, an external service or ignored fixture recovers). Delimit attempts explicitly so that
451
467
  // the earlier red cannot remain latched into the later complete green cycle.
452
- journal.append("tip-verify-start", undefined, { tip, cmdHash, gates: Object.keys(commands), cached: clean && cached });
468
+ journal.append("tip-verify-start", undefined, { tip, cmdHash, capacity, gates: Object.keys(commands), cached: clean && cached });
453
469
  if (clean && cached) {
454
- journal.append("tip-verify-cached", undefined, { tip, cmdHash, gates: Object.keys(commands) });
470
+ journal.append("tip-verify-cached", undefined, { tip, cmdHash, capacity, gates: Object.keys(commands) });
455
471
  // The skip must not read as a red. Every surface derives the tip's verdict from this cycle's
456
472
  // `tip-verify` events (cockpit derive.ts tipVerificationPassed: a run-end claiming "passed" with
457
473
  // ZERO events is fail-closed to FALSE), so a carried-forward green still journals its per-gate
458
474
  // pass — `cached: true` keeps it honest about not having re-run the command.
459
475
  for (const gate of Object.keys(commands)) {
460
- journal.append("tip-verify", undefined, { gate, cmd: commands[gate], pass: true, exitCode: 0, cached: true, tip, cmdHash });
476
+ journal.append("tip-verify", undefined, { gate, cmd: commands[gate], pass: true, exitCode: 0, cached: true, tip, cmdHash, capacity });
461
477
  }
462
478
  return false;
463
479
  }
@@ -465,7 +481,7 @@ export async function verifyIntegrationTipCached(intWt, commands, journal, opts
465
481
  for (const r of await verifyIntegrationTip(intWt, commands, journal.dir, opts.baseline)) {
466
482
  if (r.pass) {
467
483
  // Q121s: a forgiven pass journals its fingerprints — honest about what was carried, never a silent green.
468
- journal.append("tip-verify", undefined, { gate: r.gate, cmd: r.cmd, pass: true, exitCode: r.exitCode, details: r.details, ...(r.forgiven ? { forgiven: true, fingerprints: r.fingerprints } : {}), tip, cmdHash });
484
+ journal.append("tip-verify", undefined, { gate: r.gate, cmd: r.cmd, pass: true, exitCode: r.exitCode, details: r.details, ...(r.forgiven ? { forgiven: true, fingerprints: r.fingerprints } : {}), tip, cmdHash, capacity });
469
485
  }
470
486
  else {
471
487
  journal.append("tip-verify-failed", undefined, {
@@ -477,6 +493,7 @@ export async function verifyIntegrationTipCached(intWt, commands, journal, opts
477
493
  lastMergedTask: opts.lastMergedTask,
478
494
  tip,
479
495
  cmdHash,
496
+ capacity,
480
497
  });
481
498
  tipFailed = true;
482
499
  }
@@ -721,13 +738,9 @@ export async function runDaemon(repoRoot, opts = {}) {
721
738
  // subprocess, so the optional-chain open below is a no-op there). Cosmetic-only: any failure is
722
739
  // swallowed (never affects the run); the operator closes a surviving watch pane.
723
740
  const lock = acquireRunLock(repoRoot, runId);
724
- // T16: the orchestrator seat IS this daemon, so this is where the tier gets armed — the writer half
725
- // T3 shipped had no caller outside its own tests, which made the reader honest and useless: it read
726
- // ABSENT for the entire life of every run. Armed immediately after the lock (the first instant this
727
- // process owns the run) and held to the last, so the beat's span is the run's span. armSupervision
728
- // never throws, so an unwritable beat can never take a run down; it is deregistered in BOTH exits
729
- // below, because the signal reaper exits the process before the finally can run.
730
- const supervision = opts.supervise === false ? undefined : armSupervision(repoRoot, "orchestrator");
741
+ // D10: the lock is this daemon's liveness record and already carries its pid; status prints that
742
+ // identity beside the supervision row. The `orchestrator` tier belongs exclusively to the seated
743
+ // supervisor, so a run never beats or stands down that seat's record on the daemon's behalf.
731
744
  // v1.54 T2: declared before the try so the finally can always deregister (a throw before
732
745
  // registration leaves it undefined — the guard below covers that path).
733
746
  let onTermination;
@@ -855,7 +868,6 @@ export async function runDaemon(repoRoot, opts = {}) {
855
868
  }
856
869
  catch { /* cosmetic — visibility is never a gate */ }
857
870
  }
858
- supervision?.disarm(); // T16: same reason as the lock — this seat stood down, it did not die
859
871
  releaseRunLock(repoRoot); // the process dies at exit() below — the finally never runs on this path
860
872
  }
861
873
  abortRun(new Error(`terminated by ${sig}`));
@@ -904,6 +916,10 @@ export async function runDaemon(repoRoot, opts = {}) {
904
916
  const replayedGateResults = resumeLifecycleOpen
905
917
  ? journal.replayCurrentAttemptGateResults()
906
918
  : new Map();
919
+ // T7: the capacity THIS session resolved — read inside the run's fork budget, so it is the number
920
+ // every shell this run spawns will divide the machine by. Recorded evidence from another session
921
+ // is only reusable against this.
922
+ const sessionCapacity = resolvedCapacity();
907
923
  const replayedExclusions = opts.resume ? journal.replayExcludedChannels() : new Set();
908
924
  if (opts.resume) {
909
925
  // v1.53 T5: a superseded run is dead — resuming it beside its successor is the exact
@@ -1145,6 +1161,17 @@ export async function runDaemon(repoRoot, opts = {}) {
1145
1161
  await park(t, `humanGate: "${t.title}" requires approval before dispatch`, "human-gate", null, 0, startMs);
1146
1162
  return;
1147
1163
  }
1164
+ // The driver owns how a checkout is created, but runDaemon owns the destructive transition:
1165
+ // every task-checkout recreation passes through this wrapper before any driver can remove the
1166
+ // old path. A preservation failure throws and therefore leaves the old checkout in place. The
1167
+ // row is deliberately written before the later worktree-recreation row so the journal cannot
1168
+ // describe only the commits it carried while omitting uncommitted work the removal destroyed.
1169
+ const recreateTaskWorktree = async (taskBranch, taskBase, priorWt) => {
1170
+ const ref = await preserveWorktree(priorWt);
1171
+ if (ref)
1172
+ journal.append("worktree-preserved", t.id, { ref });
1173
+ return driver.worktree(repoRoot, taskBranch, taskBase);
1174
+ };
1148
1175
  const r = route(t, cfg, channels, profile, undefined, demotedChannels);
1149
1176
  for (const lint of r.lints)
1150
1177
  journal.append("routing-lint", t.id, { lint });
@@ -1216,6 +1243,15 @@ export async function runDaemon(repoRoot, opts = {}) {
1216
1243
  let gateSubject;
1217
1244
  const journalGateResult = (g) => {
1218
1245
  const blocking = gateFailed(g) && (g.gate === "review" || g.gate === "acceptance");
1246
+ // T2: a review that PASSED while DEFERRING a concern still recorded a defect — the prompt
1247
+ // promises the deferral is recorded and never dropped, and a details string is not a record a
1248
+ // later round can match. The blocking projection above is the only writer of structured
1249
+ // findings today, so on this row it writes nothing and every structured reader goes blind.
1250
+ // Same shape, same identity, on the passing row: the verdict is untouched (`pass` stays true),
1251
+ // only the projection widens to the rows the reviewer itself classified as deferred.
1252
+ const deferred = !blocking && g.gate === "review" && g.pass === true
1253
+ ? deferredReviewFindings(g.details)
1254
+ : [];
1219
1255
  // R3 (OBS-186): a gate that DECLINED has no verdict to state, and this row is the ONE seam every
1220
1256
  // fold outside this file shares. Writing `pass: false` for a decline is what turned a skip into
1221
1257
  // a failure at all of them at once — the engagement round budget (reviewRoundsSinceApproval,
@@ -1260,6 +1296,9 @@ export async function runDaemon(repoRoot, opts = {}) {
1260
1296
  ...(blocking ? {
1261
1297
  taskContentDigest: contentDigest,
1262
1298
  findings: structuredFindings(g.gate, g.details),
1299
+ } : deferred.length > 0 ? {
1300
+ taskContentDigest: contentDigest,
1301
+ findings: deferred,
1263
1302
  } : {}),
1264
1303
  // v2.0 T2 (OBS-554): the gate's OWN measurement, lifted verbatim from the meta run-gates
1265
1304
  // stamped WHERE THE GATE RAN. Nothing here re-derives a duration by subtracting journal
@@ -1270,6 +1309,27 @@ export async function runDaemon(repoRoot, opts = {}) {
1270
1309
  // every recalibration this telemetry funds, a gap is honest and a zero is a lie. The
1271
1310
  // seven-gate closed set is asserted end-to-end in tests/run/gate-telemetry.test.ts.
1272
1311
  ...gateMeasurement(g.meta),
1312
+ // T7: the capacity the gate's own command child ran under, lifted verbatim from the result
1313
+ // the battery produced — read where the shell built that child's environment, never
1314
+ // re-derived from the run's own budget, which would answer a different number than the
1315
+ // operator's export did. It is this row's only COMPARABLE identity: the two load samples
1316
+ // beside it are endpoint reads of a gate whose interior neither of them saw, so no reader
1317
+ // compares them, and matching capacity never claims the machine was calm. A gate that ran
1318
+ // no command carries nothing — it divided nothing.
1319
+ //
1320
+ // `dirtiedBy` is the one hole in that lift: run-gates REPLACES a green battery verdict with
1321
+ // a refusal when the command left the worktree dirty (run-gates.ts, three sites: the legacy
1322
+ // batch, the per-command loop and the merge-candidate full suite), and the refusal is a
1323
+ // fresh verdict object carrying nothing off the result it replaced. That command's child DID
1324
+ // run, so its row still owes the world it ran in. `sessionCapacity` is that world and not a
1325
+ // re-derivation of it: it applies the same precedence `shell` does — an operator export
1326
+ // first — and was read inside the same fork budget every gate child of this run is spawned
1327
+ // under, so it is by construction the number that child received. The flag is set only where
1328
+ // a command of THIS gate ran and dirtied the tree, so the round-entry refusal (no command
1329
+ // ran) and the round-end withdrawal (lands on a gate that runs no command) still carry
1330
+ // nothing.
1331
+ ...(g.capacity ? { capacity: g.capacity }
1332
+ : g.meta?.dirtiedBy === g.gate ? { capacity: sessionCapacity } : {}),
1273
1333
  });
1274
1334
  };
1275
1335
  // R3 (OBS-186): judge ‖ review are launched together and publish in COMPLETION order
@@ -1457,7 +1517,7 @@ export async function runDaemon(repoRoot, opts = {}) {
1457
1517
  const priorTaskTip = await gitHead(priorWt);
1458
1518
  const priorTaskSubject = await gateCommitSubject(taskBase, priorTaskTip, priorWt);
1459
1519
  const commitsToCarry = await commitsAheadOf(taskBase, priorWt);
1460
- const wt = await driver.worktree(repoRoot, taskBranch, taskBase);
1520
+ const wt = await recreateTaskWorktree(taskBranch, taskBase, priorWt);
1461
1521
  const carriedCommits = await cherryPickCommits(wt, commitsToCarry);
1462
1522
  // Reuse is about the tree the gates will actually inspect. The integration tip may have moved
1463
1523
  // while the daemon was down, so compare after recreating the task on today's taskBase rather
@@ -1550,7 +1610,21 @@ export async function runDaemon(repoRoot, opts = {}) {
1550
1610
  const canonicalCurrentCommit = currentTaskSubject === replayedGates.commit;
1551
1611
  const recreatedLegacyCommit = priorTaskTip === replayedGates.commit
1552
1612
  && priorTaskSubject === currentTaskSubject;
1553
- let reusable = exactCurrentCommit || canonicalCurrentCommit || recreatedLegacyCommit;
1613
+ // T7: the commit says the gates would inspect the same TREE; it says nothing about the
1614
+ // machine they measured it on. A resume is a new session and may have resolved a different
1615
+ // concurrency, so the rows behind a replayed green must also have been measured under the
1616
+ // capacity this session resolved — otherwise a contiguous green prefix spans two worlds.
1617
+ // Rows from before this stamp carry no capacity and replay exactly as they do today.
1618
+ const replayedCapacities = priorEvents
1619
+ .filter((e) => e.event === "gate-result" && e.taskId === t.id && e.data.commit === replayedGates.commit)
1620
+ .map((e) => e.data.capacity);
1621
+ const sameWorld = replayedCapacities.every((recorded) => sameCapacity(recorded, sessionCapacity));
1622
+ if (!sameWorld) {
1623
+ journal.append("gate-replay-capacity-changed", t.id, {
1624
+ commit: replayedGates.commit, recorded: replayedCapacities, resolved: sessionCapacity,
1625
+ });
1626
+ }
1627
+ let reusable = sameWorld && (exactCurrentCommit || canonicalCurrentCommit || recreatedLegacyCommit);
1554
1628
  const reused = [];
1555
1629
  const declaredGates = GATE_NAMES.filter((gate) => t.gates.includes(gate));
1556
1630
  for (const gate of declaredGates) {
@@ -1753,6 +1827,49 @@ export async function runDaemon(repoRoot, opts = {}) {
1753
1827
  const brief = journaledRows.join("\n\n");
1754
1828
  feedback = feedback ? `${brief}\n\n${feedback}` : brief;
1755
1829
  }
1830
+ // T6: both carries above are ATTEMPT-scoped — the funded repair is spent at the next
1831
+ // worker-launch (and budgeted at two), and the journaled brief is reset there too, so it hands
1832
+ // this dispatch only the LAST attempt's bytes. An unresolved review finding is a property of the
1833
+ // TASK: the moment one attempt fails for an unrelated reason — a red build, a refused tree, or a
1834
+ // death that journals no gate row at all — the finding is in neither carry and the next worker
1835
+ // re-derives the task from the spec and lands on the same gap the reviewer already anchored.
1836
+ // Re-derived from the journal on EVERY dispatch and retired only by a review that passes on this
1837
+ // task (journal.ts `outstandingReviewFindings`). Appended row-wise, because this round's own
1838
+ // feedback or a repair brief may already quote a finding and repeating it helps no worker.
1839
+ const outstandingFindings = outstandingReviewFindings(journaledSoFar, t.id);
1840
+ // T2: the two are different facts about the work and no one heading is true of both. A finding
1841
+ // the reviewer DEFERRED was accepted with a rationale by a review that did not block on it; a
1842
+ // blocking one is still waiting for a review to pass. Filing the deferral under the blocking
1843
+ // heading tells the next worker a passing review is owed on a concern that already drew one —
1844
+ // the exact falsehood this carry exists to remove, restated in the brief that carries it.
1845
+ //
1846
+ // The de-dup below is BLOCKING-ONLY on purpose. A review round's raw bytes quote every finding
1847
+ // it recorded, deferrals included, and those bytes ride into the very next dispatch under the
1848
+ // repair brief's "fix ONLY what these findings name" — so on the ordinary immediate retry the
1849
+ // deferral is already stated, and stated AS BLOCKING. Suppressing its heading there because it
1850
+ // is "already quoted" leaves exactly the falsehood. A quoted BLOCKING finding is quoted
1851
+ // truthfully, so that one still de-dups; a deferral is instead CUT from the raw bytes and
1852
+ // restated once, under the only heading true of it.
1853
+ const deferredRows = outstandingFindings.filter(isDeferredFinding);
1854
+ const withoutDeferrals = (text) => deferredRows.reduce((brief, finding) => brief.replaceAll(renderStructuredReviewFinding(finding), ""), text).replace(/\n{3,}/g, "\n\n").trim();
1855
+ feedback = withoutDeferrals(feedback);
1856
+ if (repairFindings !== undefined)
1857
+ repairFindings = withoutDeferrals(repairFindings);
1858
+ const briefs = [
1859
+ [OUTSTANDING_FINDINGS_HEADING, outstandingFindings.filter((f) => !isDeferredFinding(f) && !feedback.includes(f.note))],
1860
+ [DEFERRED_FINDINGS_HEADING, deferredRows],
1861
+ ];
1862
+ for (const [heading, rows] of briefs) {
1863
+ if (rows.length === 0)
1864
+ continue;
1865
+ const brief = [heading, ...rows.map((f) => {
1866
+ const rationale = f.rationale === undefined
1867
+ ? ""
1868
+ : `\n Rationale: ${f.rationale}`;
1869
+ return `- ${f.path}: ${f.note}${rationale}`;
1870
+ })].join("\n");
1871
+ feedback = feedback ? `${feedback}\n\n${brief}` : brief;
1872
+ }
1756
1873
  retryMode = repairFindings
1757
1874
  ? "repair"
1758
1875
  : priorSession
@@ -1775,21 +1892,29 @@ export async function runDaemon(repoRoot, opts = {}) {
1775
1892
  lastContextTokens = undefined;
1776
1893
  graph = setStatus(graph, t.id, "running");
1777
1894
  saveGraph(repoRoot, graph);
1778
- journal.append("task-dispatch", t.id, { assignment, attempt, provenance: dispatchProvenance(r.provenance), retryMode });
1895
+ // T6: a dispatch that carries an outstanding finding says so, and names it. Without this the
1896
+ // ledger cannot tell a carried dispatch from an amnesiac one — the exact question a run that
1897
+ // spends two frontier attempts re-deriving a known defect has to be able to answer afterwards.
1898
+ journal.append("task-dispatch", t.id, {
1899
+ assignment, attempt, provenance: dispatchProvenance(r.provenance), retryMode,
1900
+ ...(outstandingFindings.length > 0 ? { carriedFindings: outstandingFindings } : {}),
1901
+ });
1779
1902
  journal.phaseStart(t.id, "worker", { attempt, assignment });
1780
1903
  const taskBase = await integrationHead(intWt); // deps are merged → visible to this task
1781
1904
  const taskBranch = `${branch}--${t.id}`; // "--": a ref can't nest under the existing integration branch (locked decision 10)
1782
1905
  const priorWt = worktreePath(repoRoot, taskBranch);
1783
- const commitsToCarry = existsSync(priorWt) ? await commitsAheadOf(taskBase, priorWt) : [];
1784
- const wt = await driver.worktree(repoRoot, taskBranch, taskBase);
1906
+ const recreating = existsSync(priorWt);
1907
+ const commitsToCarry = recreating ? await commitsAheadOf(taskBase, priorWt) : [];
1908
+ const wt = await recreateTaskWorktree(taskBranch, taskBase, priorWt);
1785
1909
  // OBS-58: quota-failover and every retry recreate the task worktree from the integration tip —
1786
1910
  // cherry-pick prior attempts' landed commits forward so a failover dispatch cannot silently
1787
1911
  // orphan work a consult already verified as landed.
1788
1912
  let carriedCommits = [];
1789
1913
  if (commitsToCarry.length > 0) {
1790
1914
  carriedCommits = await cherryPickCommits(wt, commitsToCarry);
1791
- journal.append("worktree-recreation", t.id, { attempted: commitsToCarry, carried: carriedCommits });
1792
1915
  }
1916
+ if (recreating)
1917
+ journal.append("worktree-recreation", t.id, { attempted: commitsToCarry, carried: carriedCommits });
1793
1918
  // T2 review (material): harvest eligibility is "does this WORKTREE carry unverified work",
1794
1919
  // measured against taskBase — the same base the fast-kill's delta probe and the gates
1795
1920
  // themselves use. It was measured against this attempt's post-carry HEAD, which excluded
@@ -2892,6 +3017,15 @@ export async function runDaemon(repoRoot, opts = {}) {
2892
3017
  if (results.some((g) => g.gate === "test" && !g.pass))
2893
3018
  testGateFailed = true;
2894
3019
  if (results.every(gateSatisfied)) {
3020
+ // T6: every gate — the review included — is satisfied on this commit, so the failure brief
3021
+ // this loop is still holding describes nothing outstanding. It is dropped HERE, before the
3022
+ // merge, because a conflict below sends the task around the attempt loop again: a brief kept
3023
+ // across that retry hands the next worker findings a later review has since passed on, while
3024
+ // `carriedFindings` — re-derived from the journal, which retired them — is correctly empty,
3025
+ // leaving that dispatch's row indistinguishable from an amnesiac one. Rebuilt exactly as the
3026
+ // gate-fail brief is, so prior-RUN evidence (retired by its own rule, not by this reviewer)
3027
+ // survives and only this run's settled findings go.
3028
+ feedback = withCarriedEvidence("");
2895
3029
  const m = await mergeSerial(taskBranch, t, gated);
2896
3030
  if (m.tipMoved) {
2897
3031
  journal.append("tip-moved", t.id, m.tipMoved);
@@ -3188,9 +3322,6 @@ export async function runDaemon(repoRoot, opts = {}) {
3188
3322
  process.removeListener("SIGINT", onTermination);
3189
3323
  process.removeListener("SIGTERM", onTermination);
3190
3324
  }
3191
- // T16: every other exit — normal end, throw, termination unwind. disarm() is idempotent, so the
3192
- // signal path having already stood the tier down changes nothing here.
3193
- supervision?.disarm();
3194
3325
  try {
3195
3326
  releaseRunLock(repoRoot);
3196
3327
  }
package/dist/run/git.d.ts CHANGED
@@ -34,6 +34,54 @@ export declare const deriveForkCap: (concurrency: number, cores?: number) => num
34
34
  export declare const runWithForkBudget: <T>(concurrency: number, fn: () => Promise<T>) => Promise<T>;
35
35
  /** The cap owned by the run on this async context; the standalone default outside one. */
36
36
  export declare const resolvedForkCap: () => string;
37
+ /**
38
+ * T7: the CAPACITY a suite verdict was measured under — the fork cap the command's child actually
39
+ * received, and the core count that cap was divided from. Two verdicts are comparable only when both
40
+ * numbers match: a run resumed at a different concurrency divides the same machine by a different
41
+ * number, so a green measured in that other world is not evidence about this one.
42
+ *
43
+ * This pair is the WHOLE comparable identity, and the load averages a gate row already carries beside
44
+ * it are deliberately NOT part of it — no reader below ever feeds a load sample into the comparison.
45
+ * The capacity is deterministic and resolved here, where the child's environment is built. The load
46
+ * endpoints are neither: they are two samples taken at a gate's boundaries, and a gate's INTERIOR is
47
+ * invisible to them — on this milestone's own run a gate's interior reached well over twice what
48
+ * either of its own endpoints saw. So matching capacity establishes only that two measurements
49
+ * divided the same machine by the same number. It says nothing about whether the machine was calm.
50
+ */
51
+ export interface RunCapacity {
52
+ forkCap: number;
53
+ cores: number;
54
+ }
55
+ export type CapacityRead = {
56
+ state: "present";
57
+ capacity: RunCapacity;
58
+ } | {
59
+ state: "absent";
60
+ } | {
61
+ state: "malformed";
62
+ };
63
+ /**
64
+ * Three states, never two. A record carrying NO capacity is an older record from before this stamp
65
+ * existed: it keeps exactly the verdict it has today. A record carrying a capacity it cannot state —
66
+ * half the pair, an empty container, a zero, a negative, an unparseable value — is a NEWER record
67
+ * that is malformed, and reading it as an older one is how a fail-closed guard stops firing silently.
68
+ */
69
+ export declare function readCapacity(value: unknown): CapacityRead;
70
+ /**
71
+ * May a verdict recorded under `recorded` be reused — forgiven, cached, replayed — by a session
72
+ * running under `current`? Absent → yes, unchanged. Present and identical → yes. Malformed, a
73
+ * different capacity, or a current capacity the caller could not state → no.
74
+ */
75
+ export declare function sameCapacity(recorded: unknown, current: RunCapacity | undefined): boolean;
76
+ export declare const describeCapacity: (value: unknown) => string;
77
+ /**
78
+ * The capacity a child spawned on THIS async context would receive: the same precedence `shell`
79
+ * applies below — an operator export of the cap wins over the run's own derived value — beside the
80
+ * cores it was divided from. A caller holding a command's own result reads the capacity off THAT
81
+ * result (`ShResult.capacity`, stamped where the child's environment was built); this is for the
82
+ * decisions taken BEFORE any child exists — a cache hit, a reuse predicate.
83
+ */
84
+ export declare const resolvedCapacity: () => RunCapacity;
37
85
  /** The shipped shell ceiling: the fallback every caller gets when nothing measured a better one. */
38
86
  export declare const DEFAULT_SHELL_TIMEOUT_MS = 600000;
39
87
  export interface ShResult {
@@ -42,6 +90,8 @@ export interface ShResult {
42
90
  stderr: string;
43
91
  timedOut?: boolean;
44
92
  durationMs?: number;
93
+ /** T7: the capacity THIS child ran under, stamped where its environment was built (see `shell`). */
94
+ capacity?: RunCapacity;
45
95
  }
46
96
  export declare const setSpawnForTests: (fn: typeof spawn) => void;
47
97
  export declare const resetSpawnForTests: () => void;
@@ -63,6 +113,21 @@ export declare function cleanupRunWorktrees(repo: string, branch: string, opts:
63
113
  removeTaskIds: string[];
64
114
  }): Promise<void>;
65
115
  export declare function resolveIntegrationBranch(_repo: string, branch: string): Promise<string>;
116
+ /**
117
+ * Preserve the bytes an existing checkout holds before recreation removes it.
118
+ *
119
+ * `git stash create` cannot do this job: its apparent `-u` argument is accepted as a message and
120
+ * untracked files never enter the stash object. Build the snapshot through a disposable index
121
+ * instead. The index starts at HEAD (so no unrelated residue from the checkout's real index enters
122
+ * the tree), stages the complete working tree including ordinary untracked paths, and lives outside
123
+ * the repository so it cannot stage itself. None of these plumbing commands writes the checkout or
124
+ * its real index.
125
+ *
126
+ * The returned ref is the durable recovery handle. A clean checkout returns undefined and creates
127
+ * neither a commit nor a ref, keeping meaningful deaths visible rather than minting one ref per
128
+ * ordinary dispatch.
129
+ */
130
+ export declare function preserveWorktree(cwd: string): Promise<string | undefined>;
66
131
  export declare function createWorktree(repo: string, branch: string, baseRef: string): Promise<string>;
67
132
  export declare function linkNodeModules(repo: string, dir: string, { force }?: {
68
133
  force?: boolean | undefined;