tickmarkr 2.5.9 → 2.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (39) hide show
  1. package/dist/adapters/claude-code.js +19 -6
  2. package/dist/adapters/prompt.js +1 -1
  3. package/dist/adapters/types.d.ts +3 -0
  4. package/dist/cli/commands/init.js +1 -1
  5. package/dist/cli/commands/report.js +11 -1
  6. package/dist/cli/commands/verify.js +2 -0
  7. package/dist/compile/native.js +39 -4
  8. package/dist/drivers/orca.d.ts +1 -1
  9. package/dist/drivers/orca.js +21 -4
  10. package/dist/gates/baseline.d.ts +2 -0
  11. package/dist/gates/baseline.js +9 -1
  12. package/dist/gates/cache.d.ts +8 -0
  13. package/dist/gates/cache.js +26 -13
  14. package/dist/gates/review.d.ts +2 -3
  15. package/dist/gates/review.js +21 -24
  16. package/dist/gates/run-gates.js +11 -1
  17. package/dist/gates/test-manifest.d.ts +5 -0
  18. package/dist/gates/test-manifest.js +87 -7
  19. package/dist/gates/test-reporter.js +4 -0
  20. package/dist/graph/graph.d.ts +1 -1
  21. package/dist/graph/graph.js +6 -1
  22. package/dist/graph/schema.d.ts +2 -0
  23. package/dist/graph/schema.js +2 -0
  24. package/dist/route/preference.d.ts +1 -1
  25. package/dist/route/preference.js +10 -39
  26. package/dist/run/daemon.d.ts +7 -1
  27. package/dist/run/daemon.js +160 -19
  28. package/dist/run/git.d.ts +10 -0
  29. package/dist/run/git.js +49 -1
  30. package/dist/run/journal.d.ts +1 -1
  31. package/dist/run/journal.js +42 -6
  32. package/dist/run/merge.d.ts +1 -1
  33. package/dist/run/merge.js +30 -6
  34. package/dist/tui/cockpit/live-runtime.d.ts +4 -0
  35. package/dist/tui/cockpit/live-runtime.js +37 -3
  36. package/dist/tui/cockpit/live-store.d.ts +2 -0
  37. package/package.json +1 -1
  38. package/schema/rungraph.schema.json +7 -0
  39. package/skills/tickmarkr-overseer/SKILL.md +33 -8
@@ -437,16 +437,54 @@ export async function manifestFileCount(cmd, cwd) {
437
437
  return null;
438
438
  }
439
439
  }
440
+ /** Vitest's forks pool awaits the parallel phase, then throws before the single-fork phase
441
+ * on any rejected worker. Recover only that exact, fully accounted-for boundary. */
442
+ function strandedSingleForkFiles(files, nonce, run) {
443
+ const r = run.report;
444
+ if (run.killedFile || run.exitCode !== 1 || !r || r.nonce !== nonce || r.certificate?.exitCode !== 1)
445
+ return;
446
+ if (r.requested.length !== files.length || new Set(r.requested).size !== files.length
447
+ || r.requested.some(f => !files.includes(f)) || r.duplicateCompletions?.length)
448
+ return;
449
+ if (Object.keys(r.started).some(f => !files.includes(f))
450
+ || Object.keys(r.completed).some(f => !files.includes(f) || !Object.hasOwn(r.started, f)))
451
+ return;
452
+ const scheduling = r.scheduling;
453
+ if (!scheduling || Object.keys(scheduling).length !== files.length
454
+ || files.some(f => !Object.hasOwn(scheduling, f) || scheduling[f]?.pool !== "forks"
455
+ || typeof scheduling[f]?.singleFork !== "boolean"))
456
+ return;
457
+ const diagnostics = r.certificate.diagnostics;
458
+ if (!Array.isArray(diagnostics) || !diagnostics.length || r.certificate.errors !== diagnostics.length
459
+ || diagnostics.some(d => {
460
+ if (typeof d !== "string")
461
+ return true;
462
+ const timeout = /^(?:(.+): )?Error: \[vitest-worker\]: Timeout calling "[A-Za-z_$][\w$]*"$/.exec(d);
463
+ // The optional prefix is the reporter's testPath, not another error or arbitrary prose.
464
+ return !timeout || (timeout[1] !== undefined && !files.some(f => timeout[1] === f || timeout[1].endsWith(`/${f}`)));
465
+ }))
466
+ return;
467
+ const single = files.filter(f => scheduling[f].singleFork);
468
+ const parallel = files.filter(f => !scheduling[f].singleFork);
469
+ if (!single.length || !parallel.length || single.some(f => Object.hasOwn(r.started, f) || Object.hasOwn(r.completed, f)))
470
+ return;
471
+ if (parallel.some(f => !Object.hasOwn(r.started, f)
472
+ || !["passed", "skipped"].includes(r.completed[f]?.status ?? "")
473
+ || (r.completed[f]?.tests?.failed ?? 0) !== 0))
474
+ return;
475
+ return single;
476
+ }
440
477
  /** One configured runner execution, and its own collection under the same arguments and environment.
441
478
  * The installed runner is trusted (R28 add.1 option B); the nonce catches stale artifacts, not forgery. */
442
479
  export async function evaluateManifestedTest(cmd, cwd, opts) {
443
480
  const dir = opts.artifactDir ?? mkdtempSync(join(tmpdir(), "tickmarkr-test-report-"));
444
- const nonce = randomBytes(16).toString("hex");
445
- const reportPath = join(dir, `test-manifest-report-${nonce}.json`);
481
+ let nonce = randomBytes(16).toString("hex");
482
+ let reportPath = join(dir, `test-manifest-report-${nonce}.json`);
446
483
  const reporterPath = join(dir, `test-reporter-${nonce}.mjs`);
447
484
  const evidenceReceipts = [];
448
485
  let spawnedCommand = cmd;
449
486
  let manifestPath;
487
+ let recovery;
450
488
  const { env, verification } = manifestEnvironment(cwd);
451
489
  try {
452
490
  const invocation = await discoverTestManifest(cmd, cwd, { dir, nonce, env, overallCeilingMs: opts.overallCeilingMs, evidence: { ...opts.evidence, artifactDir: opts.evidence?.artifactDir ?? dir } });
@@ -463,18 +501,58 @@ export async function evaluateManifestedTest(cmd, cwd, opts) {
463
501
  }, null, 2) + "\n");
464
502
  writeFileSync(reporterPath, TEST_REPORTER_SOURCE);
465
503
  spawnedCommand = `${cmd}${invocation.separator} --reporter=${shq(reporterPath)} --outputFile=${shq(reportPath)}`;
466
- const invoked = await runManifestedTest(spawnedCommand, cwd, {
504
+ let invoked = await runManifestedTest(spawnedCommand, cwd, {
467
505
  evidence: { ...opts.evidence, artifactDir: opts.evidence?.artifactDir ?? dir }, manifest: files, nonce, reportPath, env,
468
506
  baselineDurations: opts.baselineDurations?.map((d) => ({ ...d, file: toManifestPath(d.file, cwd) })),
469
507
  longestFile: opts.longestFile,
470
508
  overallCeilingMs: opts.overallCeilingMs ?? DEFAULT_FILE_HANG_BUDGET_MS,
471
509
  pollMs: 20,
472
510
  });
473
- const verdict = verifyManifestReport({ manifest: files, nonce, exitCode: invoked.exitCode,
511
+ let verdict = verifyManifestReport({ manifest: files, nonce, exitCode: invoked.exitCode,
474
512
  report: invoked.report, killedFile: invoked.killedFile, hangBudgetMs: invoked.hangBudgetMs });
513
+ evidenceReceipts.push(...invoked.evidenceReceipts);
514
+ const stranded = strandedSingleForkFiles(files, nonce, invoked);
515
+ if (stranded) {
516
+ const first = invoked.report;
517
+ const retryNonce = randomBytes(16).toString("hex");
518
+ recovery = { firstNonce: nonce, firstReportPath: reportPath, retryNonce, files: stranded };
519
+ nonce = retryNonce;
520
+ reportPath = join(dir, `test-manifest-report-${nonce}.json`);
521
+ // Positional filters are substring matches (and OR with existing filters). Exclude every
522
+ // completed file as well, then require discovery to prove the exact retry set before launch.
523
+ const excluded = files.filter(f => !stranded.includes(f)).map(f => `--exclude=${shq(f.replace(/[\\*?[\]{}()!+@]/g, "\\$&"))}`).join(" ");
524
+ const retryCommand = `${cmd}${invocation.separator} ${stranded.map(f => shq(join(cwd, f))).join(" ")} ${excluded}`;
525
+ const listed = await discoverTestManifest(retryCommand, cwd, { dir, nonce, env,
526
+ overallCeilingMs: opts.overallCeilingMs, evidence: { ...opts.evidence, artifactDir: opts.evidence?.artifactDir ?? dir } });
527
+ evidenceReceipts.push(...listed.evidenceReceipts);
528
+ if (listed.files.length !== stranded.length || listed.files.some(f => !stranded.includes(f)))
529
+ throw new Error("single fork retry discovery does not match the stranded set");
530
+ writeFileSync(join(dir, `test-manifest-expected-${nonce}.json`), JSON.stringify({ nonce, files: stranded,
531
+ firstNonce: recovery.firstNonce, listingCommand: listed.listing, verification }, null, 2) + "\n");
532
+ spawnedCommand = `${retryCommand}${listed.separator} --reporter=${shq(reporterPath)} --outputFile=${shq(reportPath)}`;
533
+ invoked = await runManifestedTest(spawnedCommand, cwd, {
534
+ evidence: { ...opts.evidence, artifactDir: opts.evidence?.artifactDir ?? dir }, manifest: stranded, nonce, reportPath, env,
535
+ baselineDurations: opts.baselineDurations?.map(d => ({ ...d, file: toManifestPath(d.file, cwd) })),
536
+ longestFile: opts.longestFile, overallCeilingMs: opts.overallCeilingMs ?? DEFAULT_FILE_HANG_BUDGET_MS, pollMs: 20,
537
+ });
538
+ evidenceReceipts.push(...invoked.evidenceReceipts);
539
+ verdict = verifyManifestReport({ manifest: stranded, nonce, exitCode: invoked.exitCode,
540
+ report: invoked.report, killedFile: invoked.killedFile, hangBudgetMs: invoked.hangBudgetMs });
541
+ // An all-skipped retry may contribute lifecycle accounting, but only the combined manifest
542
+ // can establish executed success. Never rewrite either invocation's persisted certificate.
543
+ if (verdict.pass || verdict.meta.noExecutedModules) {
544
+ const retry = invoked.report;
545
+ if (retry.certificate?.errors !== 0 || !Array.isArray(retry.certificate.diagnostics) || retry.certificate.diagnostics.length)
546
+ verdict = { kind: "fail-closed", pass: false, details: "single fork retry has unknown or nonempty runner diagnostics", meta: { classification: "infra", infra: true } };
547
+ else
548
+ verdict = verifyManifestReport({ manifest: files, nonce, exitCode: invoked.exitCode, report: {
549
+ ...retry, requested: files, started: { ...first.started, ...retry.started }, completed: { ...first.completed, ...retry.completed },
550
+ } });
551
+ }
552
+ verdict.details = `single fork retry ${nonce} after worker RPC timeout in ${recovery.firstNonce}: ${verdict.details}`;
553
+ }
475
554
  // Preserve the validator's verdict and classification; runner evidence only explains it.
476
555
  const evidenceReceipt = invoked.evidenceReceipt;
477
- evidenceReceipts.push(...invoked.evidenceReceipts);
478
556
  const evidenceRoot = opts.evidence?.artifactDir ?? dir;
479
557
  const stdoutPath = join(evidenceRoot, evidenceReceipt.stdout.path);
480
558
  const stderrPath = join(evidenceRoot, evidenceReceipt.stderr.path);
@@ -495,6 +573,7 @@ export async function evaluateManifestedTest(cmd, cwd, opts) {
495
573
  details: verdict.details + diagnostics,
496
574
  classification: verdict.meta.classification,
497
575
  meta: { ...verdict.meta, nonce, manifest: files, manifestPath, listingCommand: invocation.listing, verification,
576
+ ...(recovery ? { recovery, retryable: false } : {}),
498
577
  spawnedCommand, processExit: invoked.exitCode, pid: invoked.pid, stdoutPath, stderrPath },
499
578
  exitCode: invoked.exitCode ?? -1, reportPath };
500
579
  }
@@ -503,7 +582,8 @@ export async function evaluateManifestedTest(cmd, cwd, opts) {
503
582
  if (evidenceReceipt)
504
583
  evidenceReceipts.push(...(error.evidenceReceipts ?? [evidenceReceipt]));
505
584
  return { evidenceReceipt, evidenceReceipts, pass: false, kind: "infra", classification: "infra", exitCode: -1, reportPath,
506
- details: error instanceof Error ? error.message : String(error),
507
- meta: { classification: "infra", infra: true, manifestDiscoveryFailed: true, spawnedCommand, verification, ...(manifestPath ? { manifestPath } : {}) } };
585
+ details: (recovery ? `single fork retry ${recovery.retryNonce} after ${recovery.firstNonce}: ` : "") + (error instanceof Error ? error.message : String(error)),
586
+ meta: { classification: "infra", infra: true, manifestDiscoveryFailed: true, spawnedCommand, verification,
587
+ ...(recovery ? { recovery, retryable: false } : {}), ...(manifestPath ? { manifestPath } : {}) } };
508
588
  }
509
589
  }
@@ -35,6 +35,10 @@ export default class TickmarkrReporter {
35
35
  }
36
36
  onTestRunStart(specifications) {
37
37
  this.report.requested = specifications.map(s => this.file(s));
38
+ // The files-only CLI listing has no scheduling information. These are the resolved
39
+ // specifications the pool will consume, including per-file pool overrides.
40
+ this.report.scheduling = Object.fromEntries(specifications.filter(s => s.project?.config && typeof s.pool === 'string')
41
+ .map(s => [this.file(s), { pool: s.pool, singleFork: s.project.config.poolOptions?.forks?.singleFork === true }]));
38
42
  this.save();
39
43
  }
40
44
  onTestModuleStart(module) {
@@ -24,7 +24,7 @@ export declare function onDiskSpecHash(_repoRoot: string, graph: RunGraph): OnDi
24
24
  /** A task's definition minus its files[] (and run state): what a scope amendment must NOT move (OBS-1073). */
25
25
  export declare function taskDefinitionFingerprint(task: Task): string;
26
26
  export declare function graphDefinitionHash(g: RunGraph): string;
27
- export declare function taskContentDigest(task: Pick<Task, "goal" | "files" | "acceptance">): string;
27
+ export declare function taskContentDigest(task: Pick<Task, "goal" | "files" | "acceptance" | "outOfScope">): string;
28
28
  export declare function tickmarkrDir(repoRoot: string): string;
29
29
  export declare function loadGraph(repoRoot: string): RunGraph;
30
30
  export declare function saveGraph(repoRoot: string, g: RunGraph): void;
@@ -94,9 +94,14 @@ export function graphDefinitionHash(g) {
94
94
  // finding; changing the goal, write surface or acceptance contract does. Keep the full digest here:
95
95
  // unlike graphDefinitionHash this value is persisted beside evidence and is the fail-closed join a
96
96
  // later run uses, so there is no benefit in making collision diagnosis less explicit.
97
+ // OBS-1126: outOfScope joins the identity only when present — an absent list keeps the historical
98
+ // bytes so every digest persisted before the field existed still joins.
97
99
  export function taskContentDigest(task) {
98
100
  return createHash("sha256")
99
- .update(JSON.stringify({ goal: task.goal, files: task.files, acceptance: task.acceptance }))
101
+ .update(JSON.stringify({
102
+ goal: task.goal, files: task.files, acceptance: task.acceptance,
103
+ ...(task.outOfScope ? { outOfScope: task.outOfScope } : {}),
104
+ }))
100
105
  .digest("hex");
101
106
  }
102
107
  export function tickmarkrDir(repoRoot) {
@@ -80,6 +80,7 @@ export declare const TaskSchema: z.ZodObject<{
80
80
  kind: z.ZodLiteral<"fixture">;
81
81
  paths: z.ZodArray<z.ZodString>;
82
82
  }, z.core.$strip>], "kind">>>;
83
+ outOfScope: z.ZodOptional<z.ZodArray<z.ZodString>>;
83
84
  gates: z.ZodDefault<z.ZodArray<z.ZodEnum<{
84
85
  build: "build";
85
86
  test: "test";
@@ -181,6 +182,7 @@ export declare const RunGraphSchema: z.ZodObject<{
181
182
  kind: z.ZodLiteral<"fixture">;
182
183
  paths: z.ZodArray<z.ZodString>;
183
184
  }, z.core.$strip>], "kind">>>;
185
+ outOfScope: z.ZodOptional<z.ZodArray<z.ZodString>>;
184
186
  gates: z.ZodDefault<z.ZodArray<z.ZodEnum<{
185
187
  build: "build";
186
188
  test: "test";
@@ -71,6 +71,8 @@ export const TaskSchema = z.object({
71
71
  }))
72
72
  .optional(),
73
73
  pins: z.array(PinSchema).optional(),
74
+ // OBS-1126: declared bounds the task must not cross; absent ⇒ no key (sealed graph bytes hold).
75
+ outOfScope: z.array(z.string().min(1)).optional(),
74
76
  gates: z
75
77
  .array(z.enum(GATE_NAMES))
76
78
  .default(["build", "test", "lint", "evidence", "scope", "acceptance", "review"])
@@ -1,5 +1,5 @@
1
1
  import { type AuthHealth } from "../adapters/types.js";
2
- import type { TickmarkrConfig } from "../config/config.js";
2
+ import { type TickmarkrConfig } from "../config/config.js";
3
3
  export interface Disallowed {
4
4
  by: "deny" | "allow";
5
5
  entry: string;
@@ -1,4 +1,5 @@
1
1
  import { channelKey, channelsFromConfig } from "../adapters/types.js";
2
+ import { DENY_SCOPES, denyEntriesAt } from "../config/config.js";
2
3
  import { validateGraph } from "../graph/schema.js";
3
4
  import { route, RoutingError } from "./router.js";
4
5
  const PREFERENCE_ROLES = ["worker", "judge", "review", "consult"];
@@ -121,47 +122,17 @@ export function entryMatchesChannel(entry, c, allowFamilyMatching = false) {
121
122
  export function exclusionCollector(c, routingOrCfg, role = "worker") {
122
123
  const routing = "routing" in routingOrCfg ? routingOrCfg.routing : routingOrCfg;
123
124
  const out = [];
124
- const { allow, deny } = routing ?? {};
125
- for (const entry of deny?.adapters ?? []) {
126
- if (entryMatchesChannel(entry, c, true)) {
127
- out.push({
128
- scope: "routing.deny.adapters",
129
- path: "routing.deny.adapters",
130
- configPath: "routing.deny.adapters",
131
- entry,
132
- by: "deny",
133
- });
134
- }
135
- }
136
- for (const entry of deny?.models ?? []) {
137
- if (entryMatchesChannel(entry, c, true)) {
138
- out.push({
139
- scope: "routing.deny.models",
140
- path: "routing.deny.models",
141
- configPath: "routing.deny.models",
142
- entry,
143
- by: "deny",
144
- });
145
- }
146
- }
147
- if (role === "worker") {
148
- for (const entry of deny?.workers?.adapters ?? []) {
149
- if (entryMatchesChannel(entry, c, true)) {
150
- out.push({
151
- scope: "routing.deny.workers.adapters",
152
- path: "routing.deny.workers.adapters",
153
- configPath: "routing.deny.workers.adapters",
154
- entry,
155
- by: "deny",
156
- });
157
- }
158
- }
159
- for (const entry of deny?.workers?.models ?? []) {
125
+ const { allow } = routing ?? {};
126
+ for (const scope of DENY_SCOPES) {
127
+ // Lists under workers apply only to worker seats; flat deny lists cover every role.
128
+ if (scope.path[2] === "workers" && role !== "worker")
129
+ continue;
130
+ for (const entry of denyEntriesAt(routing, scope) ?? []) {
160
131
  if (entryMatchesChannel(entry, c, true)) {
161
132
  out.push({
162
- scope: "routing.deny.workers.models",
163
- path: "routing.deny.workers.models",
164
- configPath: "routing.deny.workers.models",
133
+ scope: scope.dotted,
134
+ path: scope.dotted,
135
+ configPath: scope.dotted,
165
136
  entry,
166
137
  by: "deny",
167
138
  });
@@ -170,6 +170,11 @@ export declare const HARVESTED_RESULT_SUMMARY = "harvested: the worktree carries
170
170
  export declare function commandsHash(commands: Record<string, string>): string;
171
171
  export interface TipProof {
172
172
  kind: "fresh" | "reused" | "failed" | "incomplete";
173
+ /** Each completed gate keeps its own execution provenance, even in a mixed cycle. */
174
+ gates?: Array<{
175
+ gate: string;
176
+ kind: "fresh" | "reused";
177
+ }>;
173
178
  /** the commit the cycle's start row spoke for */
174
179
  tip?: string;
175
180
  }
@@ -177,7 +182,8 @@ export interface TipProof {
177
182
  * OBS-1077 close rider: what the engagement's LATEST verification cycle proved — its
178
183
  * `tip-verify-start` row and what followed, never a commit comparison. Exactly one kind per close:
179
184
  * fresh (every tip command ran AND passed), reused (an eligible cached cycle carried forward),
180
- * failed, or incomplete (cancelled, cut short, mixed or undelimited). Nothing before the start row
185
+ * failed, or incomplete (cancelled, cut short or undelimited). Completed mixed cycles retain each
186
+ * gate's kind beside the whole-cycle reused kind. Nothing before the start row
181
187
  * is read, so an unfinished cycle inherits nothing from an earlier green one.
182
188
  */
183
189
  export declare function runEndTipProof(events: readonly JournalEvent[]): TipProof;
@@ -33,7 +33,7 @@ import { cleanupRunWorktrees, deriveForkCap, FORK_CAP_ENV, gitHead, linkNodeModu
33
33
  import { runInteractiveSeed } from "./interactive-seed.js";
34
34
  import { classifyRepairDisposition, resolveScopeHints } from "./repair-disposition.js";
35
35
  import { applyScopeAmendments, activeRetryBan, classifyTaskFailure, classifyWorkerResultCause, deferredReviewFindings, engagementComparable, formatPriorFindingEvidence, GATE_FINGERPRINT_CAP, GATE_SATISFIED_RELEASE, identicalGateFailures, isDeferredFinding, journaledFailureBrief, Journal, loadRoutingProfile, newRunId, normalizeGateFailure, outstandingConsultGuidance, outstandingReviewFindings, pendingApprovalActions, pendingRechecks, pendingRepairFindings, phaseForGate, readPriorRunEvidence, recordedTaskFailureKind, RECHECK_RELEASE, renderStructuredReviewFinding, repairReachSinceApproval, repairsSinceApproval, reviewRoundsSinceApproval, runHasEnded, structuredFindings, upheldFeedbackByTask } from "./journal.js";
36
- import { isDiffCapPark, pickReviewer } from "../gates/review.js";
36
+ import { gateReviewerFloor, isDiffCapPark, pickReviewer } from "../gates/review.js";
37
37
  import { acquireApprovalSerialization, acquireRunLock, isPidLive, releaseRunLock } from "./lock.js";
38
38
  import { ensureIntegration, integrationBranch, integrationHead, mergeTask, reusedTipEvidence, verifyIntegrationTip } from "./merge.js";
39
39
  import { climbChannel, marginalCostRank, nextChannel, route } from "../route/router.js";
@@ -734,7 +734,8 @@ function lastVerifyCycle(events) {
734
734
  * OBS-1077 close rider: what the engagement's LATEST verification cycle proved — its
735
735
  * `tip-verify-start` row and what followed, never a commit comparison. Exactly one kind per close:
736
736
  * fresh (every tip command ran AND passed), reused (an eligible cached cycle carried forward),
737
- * failed, or incomplete (cancelled, cut short, mixed or undelimited). Nothing before the start row
737
+ * failed, or incomplete (cancelled, cut short or undelimited). Completed mixed cycles retain each
738
+ * gate's kind beside the whole-cycle reused kind. Nothing before the start row
738
739
  * is read, so an unfinished cycle inherits nothing from an earlier green one.
739
740
  */
740
741
  export function runEndTipProof(events) {
@@ -759,14 +760,25 @@ export function runEndTipProof(events) {
759
760
  if ((events[start].data.cached === true) !== cachedRow || (cachedRow && carried !== rows.length))
760
761
  return proof("incomplete");
761
762
  // D-131: fresh only when EVERY command executed; one per-gate persisted verdict makes the cycle reused.
762
- return carried > 0 ? proof("reused") : proof("fresh");
763
+ return {
764
+ ...proof(carried > 0 ? "reused" : "fresh"),
765
+ gates: gates.map((gate) => ({
766
+ gate: String(gate),
767
+ kind: rows.some((r) => r.data.gate === gate && r.data.cached === true) ? "reused" : "fresh",
768
+ })),
769
+ };
763
770
  }
764
771
  /** The close notification's statement of the proof — one clause per kind. */
765
772
  export function formatTipProof(p) {
766
773
  const at = p.tip ? ` ${p.tip.slice(0, 12)}` : "";
774
+ const gateReading = p.gates?.map(({ gate, kind }) => `${gate}: ${kind === "fresh" ? "verified fresh" : "cached (reused) — carried, not re-run"}`).join("; ");
775
+ const suffix = gateReading ? `; ${gateReading}` : "";
776
+ if (p.kind === "reused" && p.gates?.some(({ kind }) => kind === "fresh")) {
777
+ return `tip proof: ${p.kind} — commit${at}; ${gateReading}`;
778
+ }
767
779
  switch (p.kind) {
768
- case "fresh": return `tip proof: fresh — every tip command ran and passed on${at || " the integration tip"}`;
769
- case "reused": return `tip proof: reused — carried from verified commit${at}, commands not re-run`;
780
+ case "fresh": return `tip proof: fresh — every tip command ran and passed on${at || " the integration tip"}${suffix}`;
781
+ case "reused": return `tip proof: reused — carried from verified commit${at}, commands not re-run${suffix}`;
770
782
  case "failed": return `tip proof: failed —${at ? ` commit${at}` : ""} latest verification cycle is red`;
771
783
  case "incomplete": return `tip proof: incomplete —${at ? ` commit${at}` : ""} latest verification cycle did not finish`;
772
784
  }
@@ -1476,6 +1488,77 @@ async function cherryPickCommits(wt, commits) {
1476
1488
  }
1477
1489
  return carried;
1478
1490
  }
1491
+ /** Fold lifetime dispatch/carry evidence, independent of attempt budgets and routing exclusions.
1492
+ * Recreation rows name SOURCE hashes, so ownership is joined by stable patch identity. Each
1493
+ * attempt owns only what the next carry (or current subject) adds beyond its own incoming set.
1494
+ * Gate-only recreations do not start an attempt or transfer authorship to the restored seat.
1495
+ */
1496
+ async function subjectAuthors(events, taskId, wt, base) {
1497
+ const cache = new Map();
1498
+ const patches = async (commits) => {
1499
+ const ids = new Set();
1500
+ for (const commit of commits) {
1501
+ if (!cache.has(commit)) {
1502
+ const diff = await shGit(`git show --format= --binary ${shq(commit)}`, wt);
1503
+ if (diff.code !== 0)
1504
+ throw new Error(`cannot read author patch ${commit}`);
1505
+ const id = execFileSync("git", ["patch-id", "--stable"], {
1506
+ cwd: wt, input: diff.stdout, encoding: "utf8", maxBuffer: 32 * 1024 * 1024,
1507
+ }).trim().split(/\s+/)[0];
1508
+ cache.set(commit, id || undefined); // an empty commit authored no patch
1509
+ }
1510
+ const id = cache.get(commit);
1511
+ if (id)
1512
+ ids.add(id);
1513
+ }
1514
+ return ids;
1515
+ };
1516
+ let current;
1517
+ let previous;
1518
+ let awaitingCarry = false;
1519
+ const owners = new Map();
1520
+ const attribute = (ids, attempt) => {
1521
+ for (const id of ids) {
1522
+ if (attempt?.incoming.has(id))
1523
+ continue;
1524
+ const authors = owners.get(id) ?? new Set();
1525
+ authors.add(attempt?.author ?? "unknown author (missing task-dispatch assignment)");
1526
+ owners.set(id, authors);
1527
+ }
1528
+ };
1529
+ try {
1530
+ for (const row of events) {
1531
+ if (row.taskId !== taskId)
1532
+ continue;
1533
+ if (row.event === "task-dispatch") {
1534
+ previous = current;
1535
+ const a = row.data.assignment;
1536
+ current = { author: typeof a?.adapter === "string" && typeof a?.model === "string"
1537
+ ? `${a.adapter}:${a.model}` : "unknown author (missing task-dispatch assignment)", incoming: new Set() };
1538
+ awaitingCarry = true;
1539
+ }
1540
+ else if (row.event === "worktree-recreation") {
1541
+ const carried = await patches(Array.isArray(row.data.carried) ? row.data.carried : []);
1542
+ attribute(carried, awaitingCarry ? previous : current);
1543
+ if (awaitingCarry && current)
1544
+ current.incoming = carried;
1545
+ awaitingCarry = false;
1546
+ }
1547
+ else if (["worker-launch", "worker-result", "gate-result", "task-human"].includes(row.event)) {
1548
+ // The initial checkout has no recreation row. Once work/gates start, a later
1549
+ // recreation is a restore of this attempt, not the input of a new dispatch.
1550
+ awaitingCarry = false;
1551
+ }
1552
+ }
1553
+ const subject = await patches(await commitsAheadOf(base, wt));
1554
+ attribute(subject, current);
1555
+ return [...new Set([...subject].flatMap((id) => [...(owners.get(id) ?? [])]))];
1556
+ }
1557
+ catch (error) {
1558
+ // An unreadable history cannot silently remove an author from the exclusion set.
1559
+ return [`unknown author (${String(error)})`];
1560
+ }
1561
+ }
1479
1562
  // T7 (v1.86): a first run-end append that fails AFTER partial bytes landed leaves a torn tail at
1480
1563
  // EOF with no newline; a blind retry would glue the run-end line onto those bytes and readJsonl's
1481
1564
  // torn-line tolerance would drop the retry too — no terminal record despite a successful write.
@@ -2635,7 +2718,10 @@ export async function runDaemon(repoRoot, opts = {}) {
2635
2718
  const round = await runGates(task, ctx);
2636
2719
  let review = round.results.find((g) => g.gate === "review");
2637
2720
  while (review?.meta?.noVerdict === true) {
2638
- const next = pickReviewer(ctx.author, ctx.channels, badReviewers, cfg.review.prefer ?? [], task.routingHints?.floor, reviewHistory, undefined, demotedReviewers);
2721
+ // Recovery must honor the gate's floor, including the seats that just failed to
2722
+ // return a verdict; a below-floor alternative cannot replace the infra result.
2723
+ const { floor } = gateReviewerFloor(task, ctx.cfg, ctx.author, ctx.channels, [...(ctx.priorReviewers ?? []), ...badReviewers]);
2724
+ const next = pickReviewer(ctx.author, ctx.channels, badReviewers, cfg.review.prefer ?? [], floor, reviewHistory, undefined, demotedReviewers);
2639
2725
  if (!next)
2640
2726
  break;
2641
2727
  journal.append("review-infra-retry", t.id, { reviewer: channelKey(next), cause: review.meta.cause });
@@ -2744,6 +2830,9 @@ export async function runDaemon(repoRoot, opts = {}) {
2744
2830
  ...(Array.isArray(g.meta?.selectedTests) ? { selectedTests: g.meta.selectedTests } : {}),
2745
2831
  ...(g.meta?.fullSuite === true ? { fullSuite: true } : {}),
2746
2832
  ...(g.meta?.reapedGroup === true ? { reapedGroup: true } : {}),
2833
+ ...(g.gate === "review" && g.meta?.noEligibleReviewer === true ? {
2834
+ noEligibleReviewer: true, authorVendors: g.meta.authorVendors, unresolvedAuthors: g.meta.unresolvedAuthors,
2835
+ } : {}),
2747
2836
  ...(g.gate === "review" && typeof g.meta?.reviewer === "string" ? {
2748
2837
  reviewer: g.meta.reviewer,
2749
2838
  ...Object.fromEntries(["cause", "seatAuthoredBytes", "bytes", "rawPath", "briefPath", "timeoutMs", "unparseable", "noVerdict", "resolved", "reraised", "reviewerFloor", "reviewerFloorCause", "reviewerTier"]
@@ -2771,13 +2860,14 @@ export async function runDaemon(repoRoot, opts = {}) {
2771
2860
  // every recalibration this telemetry funds, a gap is honest and a zero is a lie. The
2772
2861
  // seven-gate closed set is asserted end-to-end in tests/run/gate-telemetry.test.ts.
2773
2862
  ...gateMeasurement(g.meta),
2774
- // Fresh producer evidence is copied verbatim; absent/historical evidence is never minted here.
2775
- ...(g.meta?.reused ? {} : {
2776
- ...(g.evidenceReceipt ? { evidenceReceipt: g.evidenceReceipt } : {}),
2777
- ...(g.evidenceReceipts ? { evidenceReceipts: g.evidenceReceipts } : {}),
2778
- ...Object.fromEntries(["nonce", "stdoutPath", "stderrPath", "classification"]
2779
- .filter(key => g.meta?.[key] !== undefined).map(key => [key, g.meta[key]])),
2780
- }),
2863
+ // Carried receipts retain their original invocation and artifact root.
2864
+ ...(g.evidenceReceipt ? { evidenceReceipt: g.evidenceReceipt } : {}),
2865
+ ...(g.evidenceReceipts ? { evidenceReceipts: g.evidenceReceipts } : {}),
2866
+ ...(g.meta?.reused ? {
2867
+ reused: true,
2868
+ ...(g.originRunRoot ? { originRunRoot: g.originRunRoot } : {}),
2869
+ } : Object.fromEntries(["nonce", "stdoutPath", "stderrPath", "classification"]
2870
+ .filter(key => g.meta?.[key] !== undefined).map(key => [key, g.meta[key]]))),
2781
2871
  // T7: the capacity the gate's own command child ran under, lifted verbatim from the result
2782
2872
  // the battery produced — read where the shell built that child's environment, never
2783
2873
  // re-derived from the run's own budget, which would answer a different number than the
@@ -3005,7 +3095,7 @@ export async function runDaemon(repoRoot, opts = {}) {
3005
3095
  // OBS-130/T15: both gate-resume paths consume the persisted task branch with no worker dispatch.
3006
3096
  // An operator approval skips its exact failed gate by authority; observed results skip only the
3007
3097
  // contiguous green prefix whose recorded commit is still the task branch tip.
3008
- const satisfiedGate = satisfiedGates.get(t.id);
3098
+ let satisfiedGate = satisfiedGates.get(t.id);
3009
3099
  const replayedGates = fundedRerun ? undefined : replayedGateResults.get(t.id);
3010
3100
  const recheck = approvalAction?.authority === "battery";
3011
3101
  resumeGateReplay: if (satisfiedGate || replayedGates || recheck) {
@@ -3042,6 +3132,10 @@ export async function runDaemon(repoRoot, opts = {}) {
3042
3132
  // than against the stale worktree whose task-only history cannot see newly merged dependencies.
3043
3133
  const currentTaskTip = await gitHead(wt);
3044
3134
  const currentTaskSubject = await gateCommitSubject(taskBase, currentTaskTip, wt);
3135
+ // Recheck carries only review authority for the exact subject being gated, never tool greens.
3136
+ if (recheck) {
3137
+ satisfiedGate = journal.replaySatisfiedGates(new Map([[t.id, currentTaskSubject]])).get(t.id);
3138
+ }
3045
3139
  journal.append("worktree-recreation", t.id, { attempted: commitsToCarry, carried: carriedCommits });
3046
3140
  // OBS-212: same fail-closed rule as the dispatch path — but this path is worse, because it runs
3047
3141
  // ONLY the gates after the approved one and then MERGES. T3 took it on run-20260728-110135:
@@ -3115,7 +3209,7 @@ export async function runDaemon(repoRoot, opts = {}) {
3115
3209
  const reused = [];
3116
3210
  let remainingGates;
3117
3211
  if (recheck) {
3118
- remainingGates = GATE_NAMES.filter((gate) => t.gates.includes(gate));
3212
+ remainingGates = declaredGates.filter((gate) => gate !== satisfiedGate);
3119
3213
  }
3120
3214
  else if (satisfiedGate) {
3121
3215
  remainingGates = t.gates.filter((gate) => {
@@ -3212,10 +3306,18 @@ export async function runDaemon(repoRoot, opts = {}) {
3212
3306
  // deterministic-fingerprint occurrence/review round for budget accounting.
3213
3307
  ...(!satisfiedGate && !recheck ? { replayMeasurement: true } : {}),
3214
3308
  };
3309
+ if (recheck && satisfiedGate === "review" && gateSubject.commit !== currentTaskSubject) {
3310
+ satisfiedGate = undefined;
3311
+ resumedTask.gates = declaredGates;
3312
+ }
3313
+ if (recheck && satisfiedGate === "review" && !resumedTask.gates.includes("review")) {
3314
+ journal.append("gate-waiver-carried", t.id, { gate: "review", commit: gateSubject.commit, carried: true, release: RECHECK_RELEASE });
3315
+ }
3215
3316
  await trackedDriver.project?.(t.id, "in-review");
3216
3317
  await waitForBaseline(t.id);
3217
3318
  journal.phaseStart(t.id, "gates");
3218
- const { results } = await withCommandContext(t.id, () => runReviewRecovery(resumedTask, {
3319
+ const { results } = await withCommandContext(t.id, async () => runReviewRecovery(resumedTask, {
3320
+ carriedAuthors: await subjectAuthors(journal.read(), t.id, wt, taskBase),
3219
3321
  carriedFindings: outstandingReviewFindings(journal.read(), t.id),
3220
3322
  operatorContext,
3221
3323
  worktree: wt, baseRef: taskBase, result: priorResult, author: gateAuthor,
@@ -3234,7 +3336,7 @@ export async function runDaemon(repoRoot, opts = {}) {
3234
3336
  excludeReviewers: badReviewers,
3235
3337
  reviewHistory, demotedReviewers, priorReviewers: taskReviewers(t.id),
3236
3338
  // Leg-2 (OBS-1052): the run-scoped two-strike tally — without it retirement is inert and a
3237
- // flaking seat is re-asked on every task. (carriedAuthors is held back: see SURGEON-LEG2-REPORT.)
3339
+ // flaking seat is re-asked on every task.
3238
3340
  reviewNoVerdicts,
3239
3341
  recheck, // OBS-1055: a recheck discards cached reds — the battery re-measures what the operator questioned
3240
3342
  onGate: async (e) => {
@@ -3271,6 +3373,11 @@ export async function runDaemon(repoRoot, opts = {}) {
3271
3373
  graph = addEvidence(graph, t.id, { commits: approvedCommits, gateResults: results });
3272
3374
  saveGraph(repoRoot, graph);
3273
3375
  if (!results.every(gateSatisfied)) {
3376
+ const unavailableReview = results.find((g) => gateFailed(g) && g.meta?.noEligibleReviewer === true);
3377
+ if (unavailableReview) {
3378
+ await park(t, gateFailApprovalReason(t.id, unavailableReview.details, true), "gate-fail", gateAuthor, rs?.attempts ?? 0, startMs, gateFails, consults, tokens, metered, retryMode);
3379
+ return;
3380
+ }
3274
3381
  const infra = results.find((g) => g.meta?.infra === true);
3275
3382
  if (infra) {
3276
3383
  await park(t, `${infra.gate}: ${infra.details}`, "infra", gateAuthor, rs?.attempts ?? 0, startMs, gateFails, consults, tokens, metered, retryMode);
@@ -3299,7 +3406,7 @@ export async function runDaemon(repoRoot, opts = {}) {
3299
3406
  // Observed green gates are only measurements. If the resumed suffix is red, preserve that
3300
3407
  // result in the journal and return to the ordinary attempt/consult ladder, which rebuilds
3301
3408
  // feedback from those rows. Only an operator-authorized gate release parks on a new red.
3302
- if (!satisfiedGate) {
3409
+ if (!satisfiedGate || recheck) {
3303
3410
  // OBS-1055: a recheck red funds a repair of the pin's own work. A pinned task's repair sits
3304
3411
  // on the pin (OBS-1034's exemption keeps the tried list from excluding it); a pin the fleet
3305
3412
  // cannot seat parks naming the pin — never the ladder.
@@ -3706,6 +3813,20 @@ export async function runDaemon(repoRoot, opts = {}) {
3706
3813
  Object.defineProperty(workerSlotOpts, "agent", { value: assignment.adapter });
3707
3814
  const slot = await trackedDriver.slot(wt, `${t.id}-worker-${assignment.adapter}-a${attempt}-${runTag}`, workerSlotOpts);
3708
3815
  const sessionId = retryMode === "resume" ? priorSession.id : slot.name;
3816
+ const readResumeTranscript = () => {
3817
+ try {
3818
+ const sample = adapter.readSessionTranscript?.({ cwd: wt, id: sessionId });
3819
+ return sample && Number.isSafeInteger(sample.bytes) && sample.bytes >= 0 ? sample : null;
3820
+ }
3821
+ catch {
3822
+ return null;
3823
+ }
3824
+ };
3825
+ const resumeBaseline = retryMode === "resume" ? readResumeTranscript() : null;
3826
+ if (retryMode === "resume")
3827
+ journal.append("worker-resume-requested", t.id, {
3828
+ sessionId, attempt, workerDispatchOrdinal, baselineBytes: resumeBaseline?.bytes ?? null,
3829
+ });
3709
3830
  const icmd = retryMode === "resume"
3710
3831
  ? adapter.resumeCommand(sessionId, promptFile, assignment.model)
3711
3832
  : cfg.visibility.worker === "interactive" && driver.interactive
@@ -3733,6 +3854,8 @@ export async function runDaemon(repoRoot, opts = {}) {
3733
3854
  const groupFile = `${dispatchScript}.${nonce}.pgid`;
3734
3855
  workerOwners.set(slot, { taskId: t.id, attempt, groupFile });
3735
3856
  writeFileSync(dispatchScript, [
3857
+ // Shell startup can swallow the driver's leading cd; the payload owns its checkout too.
3858
+ `cd ${shq(wt)} || exit 1`,
3736
3859
  // A driver may launch inside the daemon's group. That group is never worker-owned.
3737
3860
  `worker_pgid=$(ps -o pgid= -p $$ 2>/dev/null); daemon_pgid=$(ps -o pgid= -p ${process.pid} 2>/dev/null)`,
3738
3861
  `if [ -n "$worker_pgid" ] && [ -n "$daemon_pgid" ] && [ "$worker_pgid" != "$daemon_pgid" ]; then printf '%s\\n' "$worker_pgid" > ${shq(groupFile)}; fi`,
@@ -3765,6 +3888,7 @@ export async function runDaemon(repoRoot, opts = {}) {
3765
3888
  journal.append("worker-launch", t.id, {
3766
3889
  attempt,
3767
3890
  retryMode,
3891
+ ...(retryMode === "resume" ? { sessionId, workerDispatchOrdinal } : {}),
3768
3892
  driver: trackedDriver.id,
3769
3893
  slot: { ...slot },
3770
3894
  workspace: driver.id === "herdr" ? process.env.HERDR_WORKSPACE_ID : undefined,
@@ -4792,6 +4916,17 @@ export async function runDaemon(repoRoot, opts = {}) {
4792
4916
  else {
4793
4917
  await closeSlot(slot);
4794
4918
  }
4919
+ if (retryMode === "resume") {
4920
+ const transcript = readResumeTranscript();
4921
+ const identity = resumeBaseline === null || transcript === null
4922
+ ? "unknown" : transcript.bytes > resumeBaseline.bytes ? "confirmed" : "unconfirmed";
4923
+ journal.append("worker-resume-identity", t.id, {
4924
+ sessionId, attempt, workerDispatchOrdinal, identity,
4925
+ baselineBytes: resumeBaseline?.bytes ?? null, observedBytes: transcript?.bytes ?? null,
4926
+ // This is evidence of file growth, not an independent runtime identity handshake.
4927
+ assumption: "external runtime appends to the requested session's own transcript",
4928
+ });
4929
+ }
4795
4930
  // SPEND-01: usage from the harness's own cwd-keyed structured store, read POST-HOC from disk —
4796
4931
  // `wt` is this task's private worktree, so the path is unique; the read is sliced to records
4797
4932
  // stamped at/after this attempt's dispatch instant. Never the harvested pane text, never the
@@ -5133,7 +5268,8 @@ export async function runDaemon(repoRoot, opts = {}) {
5133
5268
  }
5134
5269
  }
5135
5270
  else {
5136
- ({ results, commits } = await withCommandContext(t.id, () => runReviewRecovery(t, {
5271
+ ({ results, commits } = await withCommandContext(t.id, async () => runReviewRecovery(t, {
5272
+ carriedAuthors: await subjectAuthors(journal.read(), t.id, wt, taskBase),
5137
5273
  carriedFindings: outstandingFindings,
5138
5274
  operatorContext,
5139
5275
  worktree: wt, baseRef: taskBase, result, author: assignment,
@@ -5228,6 +5364,11 @@ export async function runDaemon(repoRoot, opts = {}) {
5228
5364
  // already journaled by onGate, before gateFails and before any ladder selection can fund an
5229
5365
  // identical retry in the same environment. Parsed judge refusals never carry infra and keep
5230
5366
  // flowing through the chargeable quality path below.
5367
+ const unavailableReview = results.find((g) => gateFailed(g) && g.meta?.noEligibleReviewer === true);
5368
+ if (unavailableReview) {
5369
+ await park(t, gateFailApprovalReason(t.id, unavailableReview.details, true), "gate-fail", assignment, attempt + 1, startMs, gateFails, consults, tokens, metered, retryMode);
5370
+ return;
5371
+ }
5231
5372
  const infraFailure = results.find((g) => gateFailed(g) && g.meta?.infra === true);
5232
5373
  if (infraFailure) {
5233
5374
  await park(t, `${infraFailure.gate}: ${infraFailure.details}${infraFailure.meta?.recoveryBlocked ? ` — ${infraFailure.meta.recoveryBlocked}` : ""}`, "infra", assignment, attempt + 1, startMs, gateFails, consults, tokens, metered, retryMode);