tickmarkr 2.1.4 → 2.1.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -170,7 +170,8 @@ function reviewPreferIndex(c, prefer) {
170
170
  return i === -1 ? prefer.length : i;
171
171
  }
172
172
  export function pickReviewer(author, channels, exclude = [], // v1.1 failover: reviewer channels that already produced garbage for this task
173
- prefer = []) {
173
+ prefer = [], // v1.53 T2: review.prefer — reorders eligible channels, never changes eligibility
174
+ floor) {
174
175
  // FLEET-05 success criterion 2: an author not resolvable in the channel list yields NO reviewer.
175
176
  // The old `?? author.adapter` fallback compared an adapter id to vendor names, matched nothing, and
176
177
  // admitted every reviewer — including the author's own channel (fail-OPEN). null lands on reviewGate's
@@ -182,9 +183,25 @@ prefer = []) {
182
183
  // two independent axes: different vendor AND different base-model identity (ADDED TO the vendor
183
184
  // rule, never replacing it — a future edit can't silently drop either). The diversity filter runs
184
185
  // BEFORE preference ranking: prefer sorts survivors only, so no entry can resurrect an excluded channel.
185
- .filter((c) => c.vendor !== authorChannel.vendor && modelId(c.model) !== modelId(author.model) && !exclude.includes(channelKey(c)))
186
+ .filter((c) => c.vendor !== authorChannel.vendor
187
+ && modelId(c.model) !== modelId(author.model)
188
+ && !exclude.includes(channelKey(c))
189
+ && (floor === undefined || TIER_RANK[c.tier] >= TIER_RANK[floor]))
186
190
  .sort((a, b) => reviewPreferIndex(a, prefer) - reviewPreferIndex(b, prefer) || TIER_RANK[b.tier] - TIER_RANK[a.tier] || marginalCostRank(a) - marginalCostRank(b))[0] ?? null);
187
191
  }
192
+ /**
193
+ * This shows the reviewer what the task DECLARED, never what the diff may actually reach. The diff
194
+ * remains a separate stated input, so whether its touched paths fit the declaration stays a reviewer
195
+ * judgement rather than a guarantee made by this renderer.
196
+ */
197
+ export function renderDeclaredWriteScope(files) {
198
+ if (files.length === 0) {
199
+ return "## Declared write scope\nUnrestricted: this task declared no write-scope patterns.";
200
+ }
201
+ return `## Declared write scope
202
+ The task DECLARED these write-scope patterns:
203
+ ${files.map((path) => `- ${path}`).join("\n")}`;
204
+ }
188
205
  export async function reviewGate(task, worktree, baseRef, author, channels, adapters, cfg, via, excludeReviewers,
189
206
  // OBS-196: run dir for raw-output persistence on an unparseable verdict; absent (older callers,
190
207
  // direct tests) skips persistence and changes nothing else.
@@ -257,13 +274,19 @@ artifactDir) {
257
274
  policy: "full",
258
275
  ...(promotedBy ? { promotedFrom: declaredPolicy, promotedBy } : {}),
259
276
  };
260
- const reviewer = pickReviewer(author, channels, excludeReviewers ?? [], cfg.review.prefer ?? []);
277
+ // A reviewer floor is opt-in at the task. Applying cfg.routing.floors here would change the
278
+ // historical seat for every task that never asked for review-tier coupling.
279
+ const reviewerFloor = task.routingHints?.floor;
280
+ const reviewer = pickReviewer(author, channels, excludeReviewers ?? [], cfg.review.prefer ?? [], reviewerFloor);
261
281
  if (!reviewer) {
262
282
  // meta.noEligibleReviewer lets run-gates' review-retry keep the ORIGINAL unparseable result when
263
283
  // the retry finds no second seat — a truthful cause beats a synthetic no-reviewer failure.
284
+ const reason = reviewerFloor
285
+ ? `no cross-vendor reviewer available at or above task-declared ${reviewerFloor} floor (diversity rule)`
286
+ : "no cross-vendor reviewer available (diversity rule)";
264
287
  return cfg.review.required
265
- ? { gate: "review", pass: false, details: "no cross-vendor reviewer available (diversity rule); set review.required:false to waive", meta: { noEligibleReviewer: true } }
266
- : { gate: "review", pass: true, details: "WARNING: no cross-vendor reviewer available — review waived by config", meta: { noEligibleReviewer: true } };
288
+ ? { gate: "review", pass: false, details: `${reason}; set review.required:false to waive`, meta: { noEligibleReviewer: true, ...(reviewerFloor ? { reviewerFloor } : {}) } }
289
+ : { gate: "review", pass: true, details: `WARNING: ${reason} — review waived by config`, meta: { noEligibleReviewer: true, ...(reviewerFloor ? { reviewerFloor } : {}) } };
267
290
  }
268
291
  const measuredDiff = await fetchTaskDiff(worktree, baseRef);
269
292
  // Keep the reader payload identical to the text charged to the strict cap:
@@ -284,6 +307,8 @@ ${COMPLETION_FAKING_CHECKLIST}
284
307
  ## Acceptance criteria
285
308
  ${task.acceptance.map((a) => `- ${renderAcceptanceItem(a)}`).join("\n")}
286
309
 
310
+ ${renderDeclaredWriteScope(task.files)}
311
+
287
312
  ## Diff
288
313
  \`\`\`diff
289
314
  ${diff}
@@ -301,7 +326,15 @@ Respond with ONLY this JSON:
301
326
  Approve iff no material finding remains; an empty findings list is a clean approval.
302
327
  The top-level comments array is optional. Use it only for actionable line-anchored feedback.
303
328
  `;
304
- const raw = await runLlm(getAdapter(reviewer.adapter, adapters), reviewer.model, prompt, worktree, via ? { driver: via.driver, keep: via.keep, onSlot: via.onSlot, name: via.nameFor("review", reviewer.adapter), label: via.labelFor("review") } : undefined,
329
+ let concludedOnInactivity = false;
330
+ const raw = await runLlm(getAdapter(reviewer.adapter, adapters), reviewer.model, prompt, worktree, via ? {
331
+ driver: via.driver,
332
+ keep: via.keep,
333
+ onSlot: via.onSlot,
334
+ name: via.nameFor("review", reviewer.adapter),
335
+ label: via.labelFor("review"),
336
+ onInactivity: () => { concludedOnInactivity = true; },
337
+ } : undefined,
305
338
  // frontier reviewers routinely need >5min on a configured-cap-sized diff, and `claude -p` buffers all
306
339
  // output until completion — runLlm's 300s default killed reviews mid-flight, returning empty
307
340
  // stdout that read as "unparseable" and escalated to re-implementation of green code
@@ -325,14 +358,22 @@ The top-level comments array is optional. Use it only for actionable line-anchor
325
358
  saved = undefined; // persistence is evidence, not a gate input — never fail the gate on it
326
359
  }
327
360
  }
328
- const failure = cause === "malformed-verdict"
329
- ? "review output unparseable"
330
- : "review dispatch failed — no structurally valid nonce-bound response; output unparseable";
361
+ const failure = concludedOnInactivity
362
+ ? "review dispatch concluded on the inactivity policy without a structurally valid nonce-bound response; output unparseable"
363
+ : cause === "malformed-verdict"
364
+ ? "review output unparseable"
365
+ : "review dispatch failed — no structurally valid nonce-bound response; output unparseable";
331
366
  return {
332
367
  gate: "review",
333
368
  pass: false,
334
369
  details: `${failure} (reviewer ${reviewer.adapter}:${reviewer.model}; cause: ${cause}${saved ? `; raw saved: ${saved}` : ""}) — failing closed`,
335
- meta: { ...policyMeta, reviewer: channelKey(reviewer), unparseable: true, cause },
370
+ meta: {
371
+ ...policyMeta,
372
+ reviewer: channelKey(reviewer),
373
+ unparseable: true,
374
+ cause,
375
+ ...(concludedOnInactivity ? { classification: "infra", infra: true } : {}),
376
+ },
336
377
  };
337
378
  }
338
379
  const decided = findings !== null
@@ -589,7 +589,18 @@ export async function runGates(task, ctx) {
589
589
  const second = await dispatch((adapters) => reviewGate(task, ctx.worktree, ctx.baseRef, ctx.author, ctx.channels, adapters, ctx.cfg, retryVia, [...(ctx.excludeReviewers ?? []), flaked], ctx.artifactDir));
590
590
  if (second.meta?.noEligibleReviewer !== true) {
591
591
  const retried = typeof second.meta?.reviewer === "string" ? second.meta.reviewer : "none";
592
- rv = { ...second, meta: { ...second.meta, reviewRetry: { flaked, retried } } };
592
+ rv = {
593
+ ...second,
594
+ // `details` is lifted onto the journal's gate-result row; meta.reviewRetry is not. Keep the
595
+ // re-route visible in the result text a reader actually opens, including on a red retry.
596
+ details: `review re-route: ${flaked} produced no parseable verdict; replaced by ${retried}\n${second.details}`,
597
+ meta: { ...second.meta, reviewRetry: { flaked, retried } },
598
+ };
599
+ }
600
+ else if (task.routingHints?.floor) {
601
+ // Preserve the original no-answer cause when no replacement exists, but name the declared
602
+ // floor that correctly refused a lower-tier fallback.
603
+ rv = { ...rv, details: `${rv.details}\nreview re-route refused: ${second.details}` };
593
604
  }
594
605
  }
595
606
  return invocations.length ? { ...rv, meta: { ...rv.meta, invocations } } : rv;
@@ -19,9 +19,9 @@ import { addEvidence, attributeBlocked, blockedTasks, getTask, graphDefinitionHa
19
19
  import { GATE_NAMES } from "../graph/schema.js";
20
20
  import { augmentRetryBrief, consult, renderRetryGuidance } from "./consult.js";
21
21
  import { runEnvironment } from "./environment.js";
22
- import { cleanupRunWorktrees, gitHead, linkNodeModules, npmDependencyInstallCommand, npmDependencyManifestChanged, preserveWorktree, runWithForkBudget, sh, shGit, WORKTREE_LAYOUT_CONTRACT, worktreePath } from "./git.js";
22
+ import { cleanupRunWorktrees, gitHead, linkNodeModules, npmDependencyInstallCommand, npmDependencyManifestChanged, preserveWorktree, resolvedCapacity, runWithForkBudget, sameCapacity, sh, shGit, WORKTREE_LAYOUT_CONTRACT, worktreePath } from "./git.js";
23
23
  import { runInteractiveSeed } from "./interactive-seed.js";
24
- import { activeRetryBan, classifyTaskFailure, classifyWorkerResultCause, engagementComparable, formatPriorFindingEvidence, GATE_FINGERPRINT_CAP, GATE_SATISFIED_RELEASE, identicalGateFailures, journaledFailureBrief, Journal, loadRoutingProfile, newRunId, normalizeGateFailure, outstandingReviewFindings, pendingRepairFindings, phaseForGate, readPriorRunEvidence, recordedTaskFailureKind, repairsSinceApproval, reviewRoundsSinceApproval, runHasEnded, structuredFindings, upheldFeedbackByTask } from "./journal.js";
24
+ import { activeRetryBan, classifyTaskFailure, classifyWorkerResultCause, deferredReviewFindings, engagementComparable, formatPriorFindingEvidence, GATE_FINGERPRINT_CAP, GATE_SATISFIED_RELEASE, identicalGateFailures, isDeferredFinding, journaledFailureBrief, Journal, loadRoutingProfile, newRunId, normalizeGateFailure, outstandingReviewFindings, pendingRepairFindings, phaseForGate, readPriorRunEvidence, recordedTaskFailureKind, renderStructuredReviewFinding, repairsSinceApproval, reviewRoundsSinceApproval, runHasEnded, structuredFindings, upheldFeedbackByTask } from "./journal.js";
25
25
  import { isDiffCapPark } from "../gates/review.js";
26
26
  import { acquireApprovalSerialization, acquireRunLock, releaseRunLock } from "./lock.js";
27
27
  import { ensureIntegration, integrationBranch, integrationHead, mergeTask, verifyIntegrationTip } from "./merge.js";
@@ -242,6 +242,10 @@ function repairBrief(findings, diff, baseRef) {
242
242
  // v1.70 T5: default request-changes rounds a task may draw before it parks. OBS-419 keeps this as the
243
243
  // no-ceiling behavior; an operator may narrow only the next engagement on the approval that releases it.
244
244
  const REVIEW_ROUND_CAP = 2;
245
+ // T2: one heading per fact. The first is owed a passing review; the second already drew one and was
246
+ // accepted with the reviewer's own rationale, which travels beside (but never changes) its identity.
247
+ const OUTSTANDING_FINDINGS_HEADING = "## Outstanding review findings — a review has NOT passed on these yet";
248
+ const DEFERRED_FINDINGS_HEADING = "## Deferred review findings — a reviewer ACCEPTED these with a rationale and did NOT block on them; do not re-litigate, fix only if your change touches them";
245
249
  // OBS-419: the newest approval starts the current engagement, so it is also the sole authority for
246
250
  // that engagement's optional ceiling. Stop at the newest approval even when the field is absent: a
247
251
  // later ordinary release restores the module default instead of inheriting an older operator limit.
@@ -392,7 +396,7 @@ function lastVerifyCycle(events) {
392
396
  if (e.event === "tip-verify-start") {
393
397
  const { tip, cmdHash } = e.data;
394
398
  cur = typeof tip === "string" && typeof cmdHash === "string"
395
- ? { tip, cmdHash, gates: new Set(), failed: false, forgiven: false }
399
+ ? { tip, cmdHash, gates: new Set(), failed: false, forgiven: false, capacities: [e.data.capacity] }
396
400
  : undefined;
397
401
  afterRunEnd = false;
398
402
  continue;
@@ -410,9 +414,14 @@ function lastVerifyCycle(events) {
410
414
  continue;
411
415
  }
412
416
  if (!cur || afterRunEnd || cur.tip !== tip || cur.cmdHash !== cmdHash) {
413
- cur = { tip, cmdHash, gates: new Set(), failed: false, forgiven: false };
417
+ cur = { tip, cmdHash, gates: new Set(), failed: false, forgiven: false, capacities: [] };
414
418
  }
415
419
  afterRunEnd = false;
420
+ // T7: EVERY verdict row's own capacity, not just the start row's. The start row is a statement of
421
+ // intent written before a single command ran; the green this cache would carry forward lives on
422
+ // these rows, so a row whose capacity differs from the session's — or which is malformed — has to
423
+ // be able to sink the cycle by itself.
424
+ cur.capacities.push(e.data.capacity);
416
425
  if (e.event === "tip-verify-failed")
417
426
  cur.failed = true;
418
427
  else {
@@ -440,23 +449,31 @@ function lastVerifyCycle(events) {
440
449
  export async function verifyIntegrationTipCached(intWt, commands, journal, opts = {}) {
441
450
  const cmdHash = commandsHash(commands);
442
451
  const tip = await gitHead(intWt);
452
+ // T7: the capacity this session's verify children WOULD run under — the third thing a carried
453
+ // green must match, beside the tip and the command set. A cached verdict is the one place a green
454
+ // crosses a session boundary with nothing re-run, and a session resumed at a different concurrency
455
+ // divides the machine by a different number: that green was established in another world, so it is
456
+ // not carried forward and the commands run again. A pre-T7 cycle records no capacity and keeps
457
+ // exactly the behaviour it has today.
458
+ const capacity = resolvedCapacity();
443
459
  const porcelain = await shGit("git status --porcelain", intWt);
444
460
  const clean = porcelain.code === 0 && porcelain.stdout.trim() === "";
445
461
  const last = lastVerifyCycle(journal.read());
446
462
  const cached = last !== undefined && !last.failed && !last.forgiven && last.tip === tip && last.cmdHash === cmdHash
447
- && Object.keys(commands).every((g) => last.gates.has(g));
463
+ && Object.keys(commands).every((g) => last.gates.has(g))
464
+ && last.capacities.every((recorded) => sameCapacity(recorded, capacity));
448
465
  // A pair can be verified red and then green without either SHA or command hash changing (for
449
466
  // example, an external service or ignored fixture recovers). Delimit attempts explicitly so that
450
467
  // the earlier red cannot remain latched into the later complete green cycle.
451
- journal.append("tip-verify-start", undefined, { tip, cmdHash, gates: Object.keys(commands), cached: clean && cached });
468
+ journal.append("tip-verify-start", undefined, { tip, cmdHash, capacity, gates: Object.keys(commands), cached: clean && cached });
452
469
  if (clean && cached) {
453
- journal.append("tip-verify-cached", undefined, { tip, cmdHash, gates: Object.keys(commands) });
470
+ journal.append("tip-verify-cached", undefined, { tip, cmdHash, capacity, gates: Object.keys(commands) });
454
471
  // The skip must not read as a red. Every surface derives the tip's verdict from this cycle's
455
472
  // `tip-verify` events (cockpit derive.ts tipVerificationPassed: a run-end claiming "passed" with
456
473
  // ZERO events is fail-closed to FALSE), so a carried-forward green still journals its per-gate
457
474
  // pass — `cached: true` keeps it honest about not having re-run the command.
458
475
  for (const gate of Object.keys(commands)) {
459
- journal.append("tip-verify", undefined, { gate, cmd: commands[gate], pass: true, exitCode: 0, cached: true, tip, cmdHash });
476
+ journal.append("tip-verify", undefined, { gate, cmd: commands[gate], pass: true, exitCode: 0, cached: true, tip, cmdHash, capacity });
460
477
  }
461
478
  return false;
462
479
  }
@@ -464,7 +481,7 @@ export async function verifyIntegrationTipCached(intWt, commands, journal, opts
464
481
  for (const r of await verifyIntegrationTip(intWt, commands, journal.dir, opts.baseline)) {
465
482
  if (r.pass) {
466
483
  // Q121s: a forgiven pass journals its fingerprints — honest about what was carried, never a silent green.
467
- journal.append("tip-verify", undefined, { gate: r.gate, cmd: r.cmd, pass: true, exitCode: r.exitCode, details: r.details, ...(r.forgiven ? { forgiven: true, fingerprints: r.fingerprints } : {}), tip, cmdHash });
484
+ journal.append("tip-verify", undefined, { gate: r.gate, cmd: r.cmd, pass: true, exitCode: r.exitCode, details: r.details, ...(r.forgiven ? { forgiven: true, fingerprints: r.fingerprints } : {}), tip, cmdHash, capacity });
468
485
  }
469
486
  else {
470
487
  journal.append("tip-verify-failed", undefined, {
@@ -476,6 +493,7 @@ export async function verifyIntegrationTipCached(intWt, commands, journal, opts
476
493
  lastMergedTask: opts.lastMergedTask,
477
494
  tip,
478
495
  cmdHash,
496
+ capacity,
479
497
  });
480
498
  tipFailed = true;
481
499
  }
@@ -898,6 +916,10 @@ export async function runDaemon(repoRoot, opts = {}) {
898
916
  const replayedGateResults = resumeLifecycleOpen
899
917
  ? journal.replayCurrentAttemptGateResults()
900
918
  : new Map();
919
+ // T7: the capacity THIS session resolved — read inside the run's fork budget, so it is the number
920
+ // every shell this run spawns will divide the machine by. Recorded evidence from another session
921
+ // is only reusable against this.
922
+ const sessionCapacity = resolvedCapacity();
901
923
  const replayedExclusions = opts.resume ? journal.replayExcludedChannels() : new Set();
902
924
  if (opts.resume) {
903
925
  // v1.53 T5: a superseded run is dead — resuming it beside its successor is the exact
@@ -1221,6 +1243,15 @@ export async function runDaemon(repoRoot, opts = {}) {
1221
1243
  let gateSubject;
1222
1244
  const journalGateResult = (g) => {
1223
1245
  const blocking = gateFailed(g) && (g.gate === "review" || g.gate === "acceptance");
1246
+ // T2: a review that PASSED while DEFERRING a concern still recorded a defect — the prompt
1247
+ // promises the deferral is recorded and never dropped, and a details string is not a record a
1248
+ // later round can match. The blocking projection above is the only writer of structured
1249
+ // findings today, so on this row it writes nothing and every structured reader goes blind.
1250
+ // Same shape, same identity, on the passing row: the verdict is untouched (`pass` stays true),
1251
+ // only the projection widens to the rows the reviewer itself classified as deferred.
1252
+ const deferred = !blocking && g.gate === "review" && g.pass === true
1253
+ ? deferredReviewFindings(g.details)
1254
+ : [];
1224
1255
  // R3 (OBS-186): a gate that DECLINED has no verdict to state, and this row is the ONE seam every
1225
1256
  // fold outside this file shares. Writing `pass: false` for a decline is what turned a skip into
1226
1257
  // a failure at all of them at once — the engagement round budget (reviewRoundsSinceApproval,
@@ -1265,6 +1296,9 @@ export async function runDaemon(repoRoot, opts = {}) {
1265
1296
  ...(blocking ? {
1266
1297
  taskContentDigest: contentDigest,
1267
1298
  findings: structuredFindings(g.gate, g.details),
1299
+ } : deferred.length > 0 ? {
1300
+ taskContentDigest: contentDigest,
1301
+ findings: deferred,
1268
1302
  } : {}),
1269
1303
  // v2.0 T2 (OBS-554): the gate's OWN measurement, lifted verbatim from the meta run-gates
1270
1304
  // stamped WHERE THE GATE RAN. Nothing here re-derives a duration by subtracting journal
@@ -1275,6 +1309,27 @@ export async function runDaemon(repoRoot, opts = {}) {
1275
1309
  // every recalibration this telemetry funds, a gap is honest and a zero is a lie. The
1276
1310
  // seven-gate closed set is asserted end-to-end in tests/run/gate-telemetry.test.ts.
1277
1311
  ...gateMeasurement(g.meta),
1312
+ // T7: the capacity the gate's own command child ran under, lifted verbatim from the result
1313
+ // the battery produced — read where the shell built that child's environment, never
1314
+ // re-derived from the run's own budget, which would answer a different number than the
1315
+ // operator's export did. It is this row's only COMPARABLE identity: the two load samples
1316
+ // beside it are endpoint reads of a gate whose interior neither of them saw, so no reader
1317
+ // compares them, and matching capacity never claims the machine was calm. A gate that ran
1318
+ // no command carries nothing — it divided nothing.
1319
+ //
1320
+ // `dirtiedBy` is the one hole in that lift: run-gates REPLACES a green battery verdict with
1321
+ // a refusal when the command left the worktree dirty (run-gates.ts, three sites: the legacy
1322
+ // batch, the per-command loop and the merge-candidate full suite), and the refusal is a
1323
+ // fresh verdict object carrying nothing off the result it replaced. That command's child DID
1324
+ // run, so its row still owes the world it ran in. `sessionCapacity` is that world and not a
1325
+ // re-derivation of it: it applies the same precedence `shell` does — an operator export
1326
+ // first — and was read inside the same fork budget every gate child of this run is spawned
1327
+ // under, so it is by construction the number that child received. The flag is set only where
1328
+ // a command of THIS gate ran and dirtied the tree, so the round-entry refusal (no command
1329
+ // ran) and the round-end withdrawal (lands on a gate that runs no command) still carry
1330
+ // nothing.
1331
+ ...(g.capacity ? { capacity: g.capacity }
1332
+ : g.meta?.dirtiedBy === g.gate ? { capacity: sessionCapacity } : {}),
1278
1333
  });
1279
1334
  };
1280
1335
  // R3 (OBS-186): judge ‖ review are launched together and publish in COMPLETION order
@@ -1555,7 +1610,21 @@ export async function runDaemon(repoRoot, opts = {}) {
1555
1610
  const canonicalCurrentCommit = currentTaskSubject === replayedGates.commit;
1556
1611
  const recreatedLegacyCommit = priorTaskTip === replayedGates.commit
1557
1612
  && priorTaskSubject === currentTaskSubject;
1558
- let reusable = exactCurrentCommit || canonicalCurrentCommit || recreatedLegacyCommit;
1613
+ // T7: the commit says the gates would inspect the same TREE; it says nothing about the
1614
+ // machine they measured it on. A resume is a new session and may have resolved a different
1615
+ // concurrency, so the rows behind a replayed green must also have been measured under the
1616
+ // capacity this session resolved — otherwise a contiguous green prefix spans two worlds.
1617
+ // Rows from before this stamp carry no capacity and replay exactly as they do today.
1618
+ const replayedCapacities = priorEvents
1619
+ .filter((e) => e.event === "gate-result" && e.taskId === t.id && e.data.commit === replayedGates.commit)
1620
+ .map((e) => e.data.capacity);
1621
+ const sameWorld = replayedCapacities.every((recorded) => sameCapacity(recorded, sessionCapacity));
1622
+ if (!sameWorld) {
1623
+ journal.append("gate-replay-capacity-changed", t.id, {
1624
+ commit: replayedGates.commit, recorded: replayedCapacities, resolved: sessionCapacity,
1625
+ });
1626
+ }
1627
+ let reusable = sameWorld && (exactCurrentCommit || canonicalCurrentCommit || recreatedLegacyCommit);
1559
1628
  const reused = [];
1560
1629
  const declaredGates = GATE_NAMES.filter((gate) => t.gates.includes(gate));
1561
1630
  for (const gate of declaredGates) {
@@ -1768,10 +1837,37 @@ export async function runDaemon(repoRoot, opts = {}) {
1768
1837
  // task (journal.ts `outstandingReviewFindings`). Appended row-wise, because this round's own
1769
1838
  // feedback or a repair brief may already quote a finding and repeating it helps no worker.
1770
1839
  const outstandingFindings = outstandingReviewFindings(journaledSoFar, t.id);
1771
- const unquoted = outstandingFindings.filter((f) => !feedback.includes(f.note));
1772
- if (unquoted.length > 0) {
1773
- const brief = ["## Outstanding review findings — a review has NOT passed on these yet",
1774
- ...unquoted.map((f) => `- ${f.path}: ${f.note}`)].join("\n");
1840
+ // T2: the two are different facts about the work and no one heading is true of both. A finding
1841
+ // the reviewer DEFERRED was accepted with a rationale by a review that did not block on it; a
1842
+ // blocking one is still waiting for a review to pass. Filing the deferral under the blocking
1843
+ // heading tells the next worker a passing review is owed on a concern that already drew one —
1844
+ // the exact falsehood this carry exists to remove, restated in the brief that carries it.
1845
+ //
1846
+ // The de-dup below is BLOCKING-ONLY on purpose. A review round's raw bytes quote every finding
1847
+ // it recorded, deferrals included, and those bytes ride into the very next dispatch under the
1848
+ // repair brief's "fix ONLY what these findings name" — so on the ordinary immediate retry the
1849
+ // deferral is already stated, and stated AS BLOCKING. Suppressing its heading there because it
1850
+ // is "already quoted" leaves exactly the falsehood. A quoted BLOCKING finding is quoted
1851
+ // truthfully, so that one still de-dups; a deferral is instead CUT from the raw bytes and
1852
+ // restated once, under the only heading true of it.
1853
+ const deferredRows = outstandingFindings.filter(isDeferredFinding);
1854
+ const withoutDeferrals = (text) => deferredRows.reduce((brief, finding) => brief.replaceAll(renderStructuredReviewFinding(finding), ""), text).replace(/\n{3,}/g, "\n\n").trim();
1855
+ feedback = withoutDeferrals(feedback);
1856
+ if (repairFindings !== undefined)
1857
+ repairFindings = withoutDeferrals(repairFindings);
1858
+ const briefs = [
1859
+ [OUTSTANDING_FINDINGS_HEADING, outstandingFindings.filter((f) => !isDeferredFinding(f) && !feedback.includes(f.note))],
1860
+ [DEFERRED_FINDINGS_HEADING, deferredRows],
1861
+ ];
1862
+ for (const [heading, rows] of briefs) {
1863
+ if (rows.length === 0)
1864
+ continue;
1865
+ const brief = [heading, ...rows.map((f) => {
1866
+ const rationale = f.rationale === undefined
1867
+ ? ""
1868
+ : `\n Rationale: ${f.rationale}`;
1869
+ return `- ${f.path}: ${f.note}${rationale}`;
1870
+ })].join("\n");
1775
1871
  feedback = feedback ? `${feedback}\n\n${brief}` : brief;
1776
1872
  }
1777
1873
  retryMode = repairFindings
package/dist/run/git.d.ts CHANGED
@@ -34,6 +34,54 @@ export declare const deriveForkCap: (concurrency: number, cores?: number) => num
34
34
  export declare const runWithForkBudget: <T>(concurrency: number, fn: () => Promise<T>) => Promise<T>;
35
35
  /** The cap owned by the run on this async context; the standalone default outside one. */
36
36
  export declare const resolvedForkCap: () => string;
37
+ /**
38
+ * T7: the CAPACITY a suite verdict was measured under — the fork cap the command's child actually
39
+ * received, and the core count that cap was divided from. Two verdicts are comparable only when both
40
+ * numbers match: a run resumed at a different concurrency divides the same machine by a different
41
+ * number, so a green measured in that other world is not evidence about this one.
42
+ *
43
+ * This pair is the WHOLE comparable identity, and the load averages a gate row already carries beside
44
+ * it are deliberately NOT part of it — no reader below ever feeds a load sample into the comparison.
45
+ * The capacity is deterministic and resolved here, where the child's environment is built. The load
46
+ * endpoints are neither: they are two samples taken at a gate's boundaries, and a gate's INTERIOR is
47
+ * invisible to them — on this milestone's own run a gate's interior reached well over twice what
48
+ * either of its own endpoints saw. So matching capacity establishes only that two measurements
49
+ * divided the same machine by the same number. It says nothing about whether the machine was calm.
50
+ */
51
+ export interface RunCapacity {
52
+ forkCap: number;
53
+ cores: number;
54
+ }
55
+ export type CapacityRead = {
56
+ state: "present";
57
+ capacity: RunCapacity;
58
+ } | {
59
+ state: "absent";
60
+ } | {
61
+ state: "malformed";
62
+ };
63
+ /**
64
+ * Three states, never two. A record carrying NO capacity is an older record from before this stamp
65
+ * existed: it keeps exactly the verdict it has today. A record carrying a capacity it cannot state —
66
+ * half the pair, an empty container, a zero, a negative, an unparseable value — is a NEWER record
67
+ * that is malformed, and reading it as an older one is how a fail-closed guard stops firing silently.
68
+ */
69
+ export declare function readCapacity(value: unknown): CapacityRead;
70
+ /**
71
+ * May a verdict recorded under `recorded` be reused — forgiven, cached, replayed — by a session
72
+ * running under `current`? Absent → yes, unchanged. Present and identical → yes. Malformed, a
73
+ * different capacity, or a current capacity the caller could not state → no.
74
+ */
75
+ export declare function sameCapacity(recorded: unknown, current: RunCapacity | undefined): boolean;
76
+ export declare const describeCapacity: (value: unknown) => string;
77
+ /**
78
+ * The capacity a child spawned on THIS async context would receive: the same precedence `shell`
79
+ * applies below — an operator export of the cap wins over the run's own derived value — beside the
80
+ * cores it was divided from. A caller holding a command's own result reads the capacity off THAT
81
+ * result (`ShResult.capacity`, stamped where the child's environment was built); this is for the
82
+ * decisions taken BEFORE any child exists — a cache hit, a reuse predicate.
83
+ */
84
+ export declare const resolvedCapacity: () => RunCapacity;
37
85
  /** The shipped shell ceiling: the fallback every caller gets when nothing measured a better one. */
38
86
  export declare const DEFAULT_SHELL_TIMEOUT_MS = 600000;
39
87
  export interface ShResult {
@@ -42,6 +90,8 @@ export interface ShResult {
42
90
  stderr: string;
43
91
  timedOut?: boolean;
44
92
  durationMs?: number;
93
+ /** T7: the capacity THIS child ran under, stamped where its environment was built (see `shell`). */
94
+ capacity?: RunCapacity;
45
95
  }
46
96
  export declare const setSpawnForTests: (fn: typeof spawn) => void;
47
97
  export declare const resetSpawnForTests: () => void;
package/dist/run/git.js CHANGED
@@ -63,6 +63,53 @@ export const deriveForkCap = (concurrency, cores = availableParallelism()) => Ma
63
63
  export const runWithForkBudget = (concurrency, fn) => forkBudget.run(String(deriveForkCap(concurrency)), fn);
64
64
  /** The cap owned by the run on this async context; the standalone default outside one. */
65
65
  export const resolvedForkCap = () => forkBudget.getStore() ?? DEFAULT_FORK_CAP;
66
+ const positiveInt = (v) => typeof v === "number" && Number.isInteger(v) && v > 0;
67
+ /**
68
+ * Three states, never two. A record carrying NO capacity is an older record from before this stamp
69
+ * existed: it keeps exactly the verdict it has today. A record carrying a capacity it cannot state —
70
+ * half the pair, an empty container, a zero, a negative, an unparseable value — is a NEWER record
71
+ * that is malformed, and reading it as an older one is how a fail-closed guard stops firing silently.
72
+ */
73
+ export function readCapacity(value) {
74
+ if (value === undefined)
75
+ return { state: "absent" };
76
+ if (value === null || typeof value !== "object")
77
+ return { state: "malformed" };
78
+ const { forkCap, cores } = value;
79
+ return positiveInt(forkCap) && positiveInt(cores)
80
+ ? { state: "present", capacity: { forkCap, cores } }
81
+ : { state: "malformed" };
82
+ }
83
+ /**
84
+ * May a verdict recorded under `recorded` be reused — forgiven, cached, replayed — by a session
85
+ * running under `current`? Absent → yes, unchanged. Present and identical → yes. Malformed, a
86
+ * different capacity, or a current capacity the caller could not state → no.
87
+ */
88
+ export function sameCapacity(recorded, current) {
89
+ const read = readCapacity(recorded);
90
+ if (read.state === "absent")
91
+ return true;
92
+ if (read.state === "malformed" || current === undefined)
93
+ return false;
94
+ return read.capacity.forkCap === current.forkCap && read.capacity.cores === current.cores;
95
+ }
96
+ export const describeCapacity = (value) => {
97
+ const read = readCapacity(value);
98
+ return read.state === "present"
99
+ ? `fork cap ${read.capacity.forkCap} of ${read.capacity.cores} cores`
100
+ : read.state === "absent" ? "an unrecorded capacity" : "a malformed capacity";
101
+ };
102
+ /**
103
+ * The capacity a child spawned on THIS async context would receive: the same precedence `shell`
104
+ * applies below — an operator export of the cap wins over the run's own derived value — beside the
105
+ * cores it was divided from. A caller holding a command's own result reads the capacity off THAT
106
+ * result (`ShResult.capacity`, stamped where the child's environment was built); this is for the
107
+ * decisions taken BEFORE any child exists — a cache hit, a reuse predicate.
108
+ */
109
+ export const resolvedCapacity = () => ({
110
+ forkCap: Number(FORK_CAP_ENV in process.env ? process.env[FORK_CAP_ENV] : resolvedForkCap()),
111
+ cores: availableParallelism(),
112
+ });
66
113
  /** The shipped shell ceiling: the fallback every caller gets when nothing measured a better one. */
67
114
  export const DEFAULT_SHELL_TIMEOUT_MS = 600000;
68
115
  /**
@@ -103,6 +150,13 @@ function shell(cmd, cwd, timeoutMs, login) {
103
150
  // OBS-110: apply the run's own fork cap only when the operator has not already set one.
104
151
  if (!(FORK_CAP_ENV in env))
105
152
  env[FORK_CAP_ENV] = resolvedForkCap();
153
+ // T7: the capacity every result of this shell carries, read HERE — off the environment the child
154
+ // is about to receive, after the precedence above has settled. An operator export is already in
155
+ // `env`, so what gets recorded is the operator's number, which is the case a release was re-taken
156
+ // for; re-deriving the run's own budget after the command returned would stamp a cap no child ran
157
+ // under. `Number` of an unparseable export is NaN, which every reader treats as malformed and
158
+ // therefore fails closed — the honest direction when the cap in play cannot be stated.
159
+ const capacity = { forkCap: Number(env[FORK_CAP_ENV]), cores: availableParallelism() };
106
160
  const attempt = () => new Promise((resolve) => {
107
161
  const startedAt = Date.now();
108
162
  // detached: bash gets its own process group so a timeout can kill the whole tree —
@@ -120,7 +174,7 @@ function shell(cmd, cwd, timeoutMs, login) {
120
174
  clearTimeout(timer);
121
175
  stdout += stdoutDecoder.end();
122
176
  stderr += stderrDecoder.end();
123
- resolve({ code, stdout, stderr: err ?? stderr, timedOut, durationMs: Date.now() - startedAt });
177
+ resolve({ code, stdout, stderr: err ?? stderr, timedOut, durationMs: Date.now() - startedAt, capacity });
124
178
  };
125
179
  const timer = setTimeout(() => {
126
180
  timedOut = true;
@@ -170,7 +224,7 @@ function shell(cmd, cwd, timeoutMs, login) {
170
224
  // Bounded, and the bound is what makes a persisting shortage a REPORTED failure rather than a
171
225
  // wedged daemon: past it the caller gets the refusal's own text under exit 127, as before.
172
226
  if (n >= SPAWN_ATTEMPT_LIMIT) {
173
- return { code: 127, stdout: "", stderr: String(r.refused), durationMs: Date.now() - startedAt };
227
+ return { code: 127, stdout: "", stderr: String(r.refused), durationMs: Date.now() - startedAt, capacity };
174
228
  }
175
229
  await new Promise((wake) => setTimeout(wake, SPAWN_RETRY_BACKOFF_MS * n));
176
230
  }
@@ -36,6 +36,7 @@ export interface StructuredFinding {
36
36
  path: string;
37
37
  symbol: string;
38
38
  note: string;
39
+ rationale?: string;
39
40
  fingerprint: string;
40
41
  }
41
42
  export declare const UNIDENTIFIED = "<unidentified>";
@@ -52,6 +53,15 @@ export declare const UNIDENTIFIED = "<unidentified>";
52
53
  * normalized words (see toFinding).
53
54
  */
54
55
  export declare function structuredFindings(gate: string, details: string, _scopeFiles?: string[]): StructuredFinding[];
56
+ export declare function isDeferredFinding(finding: StructuredFinding): boolean;
57
+ /**
58
+ * The findings a PASSING review DEFERRED — the rows a blocking-only projection drops on the floor.
59
+ * A passing review's details are prose; without this the deferral has no identity a later round can
60
+ * match, and every structured reader of the journal is blind to a defect the reviewer itself named.
61
+ */
62
+ export declare function deferredReviewFindings(details: string): StructuredFinding[];
63
+ /** The exact review.ts details fragment represented by a structured review finding. */
64
+ export declare function renderStructuredReviewFinding(finding: StructuredFinding): string;
55
65
  export interface PriorRunJournal {
56
66
  runId: string;
57
67
  events: JournalEvent[];
@@ -126,6 +136,19 @@ export declare function journaledFailureBrief(events: JournalEvent[], taskId: st
126
136
  * finding was dropped at the exact moment the operator paid for another attempt to fix it. A review
127
137
  * that DECLINED (`skipped`) is not a verdict and neither adds nor retires — fail closed. Findings are
128
138
  * keyed by fingerprint, so a reviewer restating one across rounds carries it once, not once per round.
139
+ *
140
+ * v2.1.5 T2: a passing review settles the findings it BLOCKED on. It does not settle the ones it
141
+ * DEFERRED — those it saw, declined to block on, and recorded a rationale for, and nothing has fixed
142
+ * them. So a pass retires the blocking set and re-seats its own deferrals, and the two retirements
143
+ * stay distinguishable: the blocking finding is gone, the deferral travels on as accepted work.
144
+ *
145
+ * A deferral's bound is the SAME single release as a blocking finding's — the operator accepting the
146
+ * review gate itself (`GATE_SATISFIED_RELEASE` stamped `gate: "review"`), the one approval in which a
147
+ * human actually looked at what the reviewer waved through. It is deliberately NOT bounded by a round
148
+ * count or by a time window: both retire a finding by arithmetic nobody read, which is the silent drop
149
+ * this fold exists to refuse. Nor can it accumulate — a reviewer restating the same path/note round
150
+ * after round re-seats ONE fingerprint, and a revised rationale replaces the prior rationale on that
151
+ * row. N rounds of the same concern therefore carry the newest accepted explanation once, not N rows.
129
152
  */
130
153
  export declare function outstandingReviewFindings(events: JournalEvent[], taskId: string): StructuredFinding[];
131
154
  /** The findings a funded repair must carry into the next dispatch, or undefined if none is pending. */