@haiyangbg/buildbeat 3.0.1 → 3.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (38) hide show
  1. package/CHANGELOG.md +40 -9
  2. package/SKILL.md +23 -300
  3. package/docs/CAPABILITY-MATRIX.md +1 -1
  4. package/docs/README.md +5 -5
  5. package/docs/RELEASING.md +5 -5
  6. package/docs/v2/RFC-0001-product-definition.md +5 -5
  7. package/docs/v2/RFC-0002-domain-model.md +1 -1
  8. package/docs/v2/RFC-0003-workflow-policy.md +2 -2
  9. package/docs/v2/SPEC-0001-events-v1.md +1 -1
  10. package/docs/v2/guide/01-quickstart.en.md +3 -1
  11. package/docs/v2/guide/01-quickstart.md +3 -1
  12. package/docs/v2/guide/02-workflow-guide.md +15 -0
  13. package/docs/v2/guide/07-approval-guide.en.md +4 -2
  14. package/docs/v2/guide/07-approval-guide.md +14 -2
  15. package/docs/v2/guide/09-security-boundaries.md +1 -1
  16. package/docs/v2/guide/10-recovery.en.md +21 -5
  17. package/docs/v2/guide/10-recovery.md +21 -5
  18. package/docs/v2/skill/01-principles.md +26 -0
  19. package/docs/v2/skill/02-project-layout.md +31 -0
  20. package/docs/v2/skill/03-collaboration-rules.md +45 -0
  21. package/docs/v2/skill/04-rhythm-and-rituals.md +102 -0
  22. package/docs/v2/skill/05-red-lines.md +13 -0
  23. package/docs/v2/skill/06-bootstrap-and-takeover.md +84 -0
  24. package/docs/v2/skill/07-templates-and-lessons.md +24 -0
  25. package/package.json +5 -14
  26. package/src/v2/cli/run-config-check.js +229 -0
  27. package/src/v2/cli/run.js +81 -15
  28. package/src/v2/engine/reducer.js +6 -0
  29. package/src/v2/engine/yaml-subset.js +52 -11
  30. package/src/v2/presets/policies/ui-render-gate.yaml +1 -1
  31. package/src/v2/runtime/decisions.js +33 -31
  32. package/src/v2/runtime/gc.js +59 -18
  33. package/src/v2/runtime/metrics.js +3 -2
  34. package/src/v2/runtime/orchestrator.js +308 -106
  35. package/src/v2/storage/event-ledger.js +22 -3
  36. package/src/v2/workspace/workspace-manager.js +244 -16
  37. package/templates/v2/CLAUDE.md +1 -1
  38. package/templates/v2/run-config.example.yaml +5 -1
@@ -21,9 +21,12 @@ import { EventLedger, canonicalJson } from "../storage/event-ledger.js";
21
21
  import {
22
22
  acquireLock,
23
23
  createWorkspace,
24
+ describeLockOwner,
24
25
  listChangedPaths,
26
+ liveParallelMarkers,
25
27
  readback,
26
28
  releaseLock,
29
+ withRepoGitLock,
27
30
  } from "../workspace/workspace-manager.js";
28
31
  import { writeRunRecord } from "./run-record.js";
29
32
  import { computeWorkCost } from "./work-cost.js";
@@ -52,31 +55,105 @@ function sha256(text) {
52
55
  return `sha256:${createHash("sha256").update(text, "utf8").digest("hex")}`;
53
56
  }
54
57
 
55
- // MVP is single project, single active run: driving a run takes a
56
- // repository-wide lock in addition to the per-run lock.
58
+ // By default one run drives a repository at a time: it holds the
59
+ // repository-wide active-run lock for its whole drive. A run whose config
60
+ // sets `parallel: true` instead holds a per-work lock and a marker, passing
61
+ // the active-run lock only briefly as a gate, so runs of different works can
62
+ // drive together while runs of the same work stay exclusive. Real incident:
63
+ // a session waited 3h23m behind another work's run although worktrees were
64
+ // already isolated; parallelism is opt-in because verifiers that bind fixed
65
+ // ports or share a database would collide.
57
66
  const ACTIVE_LOCK = "active-run";
58
67
 
59
68
  function lockActive(repoRoot) {
60
69
  try {
61
70
  acquireLock(repoRoot, ACTIVE_LOCK);
62
- } catch {
71
+ } catch (error) {
72
+ // A lock whose owner is gone was already reclaimed inside acquireLock;
73
+ // what reaches here is held (or unreadable), and the owner is the one
74
+ // fact a blocked caller needs.
75
+ const detail = error.lock?.detail;
63
76
  throw new OrchestratorError(
64
- "another run is active in this repository (MVP allows a single active run)",
77
+ `another run is active in this repository (MVP allows a single active run)${detail ? `; ${detail}` : ""}`,
65
78
  );
66
79
  }
67
80
  }
68
81
 
69
- function withRunLocks(repoRoot, runId, fn) {
82
+ function lockActiveWaiting(repoRoot, waitMs) {
83
+ const deadline = Date.now() + waitMs;
84
+ for (;;) {
85
+ try {
86
+ acquireLock(repoRoot, ACTIVE_LOCK);
87
+ return;
88
+ } catch (error) {
89
+ if (!error.lock || Date.now() >= deadline) {
90
+ break;
91
+ }
92
+ Atomics.wait(new Int32Array(new SharedArrayBuffer(4)), 0, 0, 25);
93
+ }
94
+ }
70
95
  lockActive(repoRoot);
96
+ }
97
+
98
+ function holdRunLock(repoRoot, runId, fn) {
99
+ acquireLock(repoRoot, runId);
100
+ try {
101
+ return fn();
102
+ } finally {
103
+ releaseLock(repoRoot, runId);
104
+ }
105
+ }
106
+
107
+ function withRunLocks(repoRoot, runId, fn, { workId = null, parallel = false } = {}) {
108
+ if (!parallel) {
109
+ lockActive(repoRoot);
110
+ try {
111
+ const running = liveParallelMarkers(repoRoot);
112
+ if (running.length > 0) {
113
+ throw new OrchestratorError(
114
+ `another run is active in this repository (parallel run(s) ${running
115
+ .map((marker) => (marker.owner ? `${marker.run}, ${describeLockOwner(marker.owner)}` : marker.run))
116
+ .join("; ")}); this run is exclusive (set parallel: true in its run config to drive alongside other works)`,
117
+ );
118
+ }
119
+ return holdRunLock(repoRoot, runId, fn);
120
+ } finally {
121
+ releaseLock(repoRoot, ACTIVE_LOCK);
122
+ }
123
+ }
124
+ if (!workId) {
125
+ throw new OrchestratorError("a parallel run needs its work id");
126
+ }
127
+ const workLock = `@work.${workId}`;
128
+ const marker = `@parallel.${runId}`;
129
+ try {
130
+ acquireLock(repoRoot, workLock);
131
+ } catch (error) {
132
+ const detail = error.lock?.detail;
133
+ throw new OrchestratorError(
134
+ `another run of ${workId} is active (runs of the same work never drive together)${detail ? `; ${detail}` : ""}`,
135
+ );
136
+ }
71
137
  try {
72
- acquireLock(repoRoot, runId);
138
+ // Gate: an exclusive run holds active-run for its whole drive, so a
139
+ // parallel run cannot slip in while it runs; the marker, created under
140
+ // the gate, is what an exclusive run checks before it starts. Another
141
+ // parallel run holds the gate for milliseconds, so a busy gate is waited
142
+ // for briefly before it is reported (two parallel starts at the same
143
+ // instant must both get through).
144
+ lockActiveWaiting(repoRoot, 2000);
73
145
  try {
74
- return fn();
146
+ acquireLock(repoRoot, marker);
75
147
  } finally {
76
- releaseLock(repoRoot, runId);
148
+ releaseLock(repoRoot, ACTIVE_LOCK);
149
+ }
150
+ try {
151
+ return holdRunLock(repoRoot, runId, fn);
152
+ } finally {
153
+ releaseLock(repoRoot, marker);
77
154
  }
78
155
  } finally {
79
- releaseLock(repoRoot, ACTIVE_LOCK);
156
+ releaseLock(repoRoot, workLock);
80
157
  }
81
158
  }
82
159
 
@@ -149,12 +226,17 @@ function makeContext(options, ledger, workspace) {
149
226
  // review rounds could not be raised from the run config, and approving
150
227
  // resume-review re-asked the same question forever.
151
228
  context.runBudgets = options.budgets ?? {};
152
- context.maxAttemptsFor = (step) =>
229
+ context.budgetLimitFor = (step) =>
153
230
  (context.runBudgets.maxAttempts?.[step] ??
154
231
  workflow.budgets?.maxAttempts?.[step] ??
155
232
  maxAttemptsPerStep) +
156
- (ledger.state.budgetExtensions?.[step] ?? 0) +
157
- (ledger.state.steps[step]?.infraAttempts ?? 0);
233
+ (ledger.state.budgetExtensions?.[step] ?? 0);
234
+ context.maxAttemptsFor = (step) => context.budgetLimitFor(step) +
235
+ (ledger.state.steps[step]?.infraAttempts ?? 0) +
236
+ (ledger.state.steps[step]?.freeAttempts ?? 0);
237
+ // Refunds must not move this ceiling: otherwise a success-only loop has
238
+ // an ever-growing limit. Human extensions deliberately raise it.
239
+ context.totalAttemptsFor = (step) => context.budgetLimitFor(step) * 3;
158
240
  context.policies = options.policies ?? [];
159
241
  context.allowedPaths = options.allowedPaths ?? null;
160
242
  context.reviewTriage = options.reviewTriage ?? null;
@@ -179,17 +261,64 @@ function makeContext(options, ledger, workspace) {
179
261
  evidenceDigest: lastEvidence?.digest ?? "UNVERIFIED",
180
262
  };
181
263
  };
182
- context.waitHuman = (transition, reasons, kind = "boundary") => {
264
+ context.waitHuman = (transition, reasons, kind = "boundary", grants = []) => {
183
265
  ledger.append({
184
266
  type: "HUMAN_REQUESTED",
185
267
  actor: KERNEL,
186
268
  ts: context.now(),
187
- data: { transition, subject: context.subjectNow(), reasons, kind },
269
+ data: { transition, subject: context.subjectNow(), reasons, kind, ...(grants.length ? { grants } : {}) },
188
270
  });
189
271
  };
190
272
  return context;
191
273
  }
192
274
 
275
+ // Keep the charged budget distinct from the worker invocation number.
276
+ function budgetUsage(context, step) {
277
+ const state = context.ledger.state.steps[step];
278
+ const attempts = state?.attempts ?? 0;
279
+ const used = attempts - (state?.infraAttempts ?? 0) - (state?.freeAttempts ?? 0);
280
+ const failures = context.ledger.events.filter((event) =>
281
+ event.type === "STEP_FINISHED" && event.data.step === step &&
282
+ event.data.status !== "succeeded" && event.data.infra !== true).length;
283
+ return { attempts, used, failures, limit: context.budgetLimitFor(step) };
284
+ }
285
+
286
+ function budgetReasons(context, step, safeguard = false) {
287
+ const { attempts, used, failures, limit } = budgetUsage(context, step);
288
+ const charging = context.workflow.steps.find((item) => item.id === step)?.readonly
289
+ ? "each round is charged" : "successful attempts are not charged";
290
+ return [
291
+ safeguard
292
+ ? `${step} budget exhausted (runaway safeguard): ${attempts}/${context.totalAttemptsFor(step)} total attempt(s), ${failures} real failure(s)`
293
+ : `${step} budget exhausted: ${used}/${limit} charged attempt(s) used, ${failures} real failure(s) (${charging})`,
294
+ `approve resume-${step} = ${safeguard ? "raise the safeguard and continue" : "one more attempt"}; reject = end this run and decide the merge on the evidence you have`,
295
+ ];
296
+ }
297
+
298
+ function workReviewBudget(context, step) {
299
+ const stepDef = context.workflow.steps.find((item) => item.id === step);
300
+ const cap = context.runBudgets.reviewRoundsPerWork;
301
+ if (cap === undefined || !(step === "review" || stepDef?.worker === "reviewer")) return null;
302
+ const prior = computeWorkCost(context.repoRoot, context.ledger.state.run.work, {
303
+ excludeRun: context.ledger.state.run.id,
304
+ });
305
+ const rounds = prior.reviewRounds + (context.ledger.state.steps[step]?.attempts ?? 0);
306
+ const allowed = cap + (context.ledger.state.workReviewGrants ?? 0);
307
+ return { rounds, allowed, exhausted: rounds >= allowed };
308
+ }
309
+
310
+ // Every cap the next attempt of `step` would hit. Recorded on the request
311
+ // at stop time, so one approval lifts both layers (run and work) at once.
312
+ function budgetGrants(context, step) {
313
+ const grants = [];
314
+ if ((context.ledger.state.steps[step]?.attempts ?? 0) >= context.maxAttemptsFor(step) ||
315
+ (context.ledger.state.steps[step]?.attempts ?? 0) >= context.totalAttemptsFor(step)) {
316
+ grants.push({ step, scope: "run" });
317
+ }
318
+ if (workReviewBudget(context, step)?.exhausted) grants.push({ step, scope: "work" });
319
+ return grants;
320
+ }
321
+
193
322
  // Evaluates configured policies of `type` for `appliesTo`, records every
194
323
  // verdict as a POLICY_EVALUATED event, and reports what the kernel must do.
195
324
  // ADVISORY failures are recorded but never gate (doctor reports the gap).
@@ -249,10 +378,7 @@ function settleOutcome(context, step, outcome, tree, exec) {
249
378
  // A step that failed its final attempt can never run again, so routing
250
379
  // to fix would spend a worker on a candidate nothing can verify.
251
380
  if ((ledger.state.steps[step]?.attempts ?? 0) >= context.maxAttemptsFor(step)) {
252
- context.waitHuman(`resume-${step}`, [
253
- `budget exhausted: ${step} failed its final attempt (maxAttempts=${context.maxAttemptsFor(step)}); not routing to fix`,
254
- `approving resume-${step} grants one more attempt; rejecting ends the run`,
255
- ]);
381
+ context.waitHuman(`resume-${step}`, budgetReasons(context, step), "budget", budgetGrants(context, step));
256
382
  return null;
257
383
  }
258
384
  }
@@ -336,24 +462,19 @@ function drive(context, startStep, { skipBoundaryOnce = false } = {}) {
336
462
  // every run of the work, superseded ones included, so "one run per
337
463
  // round" cannot slip past the per-run budget. Reaching the cap is a
338
464
  // human decision (review once more, or merge/close as-is), not a stop.
339
- const workCap = context.runBudgets.reviewRoundsPerWork;
340
- if (workCap !== undefined && (stepDef.worker === "reviewer" || step === "review")) {
341
- const prior = computeWorkCost(context.repoRoot, ledger.state.run.work, {
342
- excludeRun: ledger.state.run.id,
343
- });
344
- const rounds = prior.reviewRounds + (ledger.state.steps[step]?.attempts ?? 0);
345
- const allowed = workCap + (ledger.state.workReviewGrants ?? 0);
346
- if (rounds >= allowed) {
347
- context.waitHuman(
348
- `enter-${step}`,
349
- [
350
- `work review cap reached: ${rounds} review round(s) across ${prior.runs + 1} run(s) of ${ledger.state.run.work} (budgets.reviewRoundsPerWork=${workCap})`,
351
- `approve enter-${step} to review once more, or reject and merge/close the work on the evidence you have`,
352
- ],
353
- "work-review-cap",
354
- );
355
- return;
356
- }
465
+ const workBudget = workReviewBudget(context, step);
466
+ if (workBudget?.exhausted) {
467
+ const { failures } = budgetUsage(context, step);
468
+ context.waitHuman(
469
+ `enter-${step}`,
470
+ [
471
+ `${step} work review budget exhausted: ${workBudget.rounds}/${workBudget.allowed} review round(s) across the work, ${failures} real failure(s) in this run`,
472
+ `approve enter-${step} = one more review round (also lifts this run's review cap if it is spent); reject = end this run and decide the merge on the evidence you have`,
473
+ ],
474
+ "work-review-cap",
475
+ budgetGrants(context, step),
476
+ );
477
+ return;
357
478
  }
358
479
 
359
480
  const preGate = runPolicyGate(context, "pre", step);
@@ -375,10 +496,12 @@ function drive(context, startStep, { skipBoundaryOnce = false } = {}) {
375
496
  const attempt = (ledger.state.steps[step]?.attempts ?? 0) + 1;
376
497
  const maxAttempts = context.maxAttemptsFor(step);
377
498
  if (attempt > maxAttempts) {
378
- context.waitHuman(`resume-${step}`, [
379
- `budget exhausted: ${step} would exceed maxAttempts=${maxAttempts}`,
380
- `approving resume-${step} grants one more attempt; rejecting ends the run`,
381
- ]);
499
+ context.waitHuman(`resume-${step}`, budgetReasons(context, step), "budget", budgetGrants(context, step));
500
+ return;
501
+ }
502
+
503
+ if (attempt > context.totalAttemptsFor(step)) {
504
+ context.waitHuman(`resume-${step}`, budgetReasons(context, step, true), "budget", budgetGrants(context, step));
382
505
  return;
383
506
  }
384
507
 
@@ -581,11 +704,13 @@ function drive(context, startStep, { skipBoundaryOnce = false } = {}) {
581
704
  stepStatus === "crashed" ||
582
705
  stepStatus === "invalid-output" ||
583
706
  (stepStatus === "failed" && exec.exitCode === 75);
707
+ const free = stepStatus === "succeeded" && stepDef.readonly !== true;
584
708
  ledger.append({
585
709
  type: "STEP_FINISHED",
586
710
  actor: KERNEL,
587
711
  ts: now(),
588
- data: { step, attempt, status: stepStatus, exitCode: exec.exitCode, ...(infra ? { infra: true } : {}) },
712
+ data: { step, attempt, status: stepStatus, exitCode: exec.exitCode,
713
+ ...(infra ? { infra: true } : {}), ...(free ? { free: true } : {}) },
589
714
  });
590
715
  ledger.append({
591
716
  type: "BUDGET_CONSUMED",
@@ -593,7 +718,7 @@ function drive(context, startStep, { skipBoundaryOnce = false } = {}) {
593
718
  ts: now(),
594
719
  data: {
595
720
  kind: "attempts",
596
- amount: infra ? 0 : 1,
721
+ amount: infra || free ? 0 : 1,
597
722
  remaining: context.maxAttemptsFor(step) - attempt,
598
723
  },
599
724
  });
@@ -736,26 +861,31 @@ function drive(context, startStep, { skipBoundaryOnce = false } = {}) {
736
861
  outcome = "succeeded";
737
862
  }
738
863
  const routed = settleOutcome(context, step, outcome, tree, exec);
739
- // Finding triage gate (reviewTriage: required): blocking findings stop
740
- // for a human verdict before any fixer runs. Findings are prescriptions,
741
- // not facts — auto-routing them to a fixer burned four oscillation
742
- // rounds in the deploy campaign before a human stopped the loop.
743
- if (routed && outcome === "findings-blocking" && context.reviewTriage === "required") {
744
- context.waitHuman(
745
- `enter-${routed}`,
746
- [
747
- `review found ${blockingFindings.length} blocking finding(s); triage before ${routed} runs`,
748
- ...blockingFindings
749
- .slice(0, 5)
750
- .map(
751
- (finding) =>
752
- `[${finding.severity} ${fingerprintFinding(finding)}] ${finding.summary.slice(0, 200)}`,
753
- ),
754
- `adjudicate fingerprints (findings adjudicate), then approve enter-${routed} or reject the run`,
755
- ],
756
- "finding-triage",
757
- );
758
- return;
864
+ // Ask before spending fix/verify workers: one approval covers the next
865
+ // round and both review caps, with the grant bound to this request.
866
+ if (routed && outcome === "findings-blocking") {
867
+ const isReview = step === "review" || stepDef.worker === "reviewer";
868
+ const grants = isReview ? budgetGrants(context, step) : [];
869
+ const triage = context.reviewTriage === "required";
870
+ if (triage || grants.length) {
871
+ const { used, limit, failures } = budgetUsage(context, step);
872
+ const workBudget = workReviewBudget(context, step);
873
+ context.waitHuman(
874
+ `enter-${routed}`,
875
+ [
876
+ ...(grants.length ? [
877
+ `${step} budget exhausted: ${used}/${limit} review round(s) used in this run${workBudget ? `, ${workBudget.rounds}/${workBudget.allowed} across the work` : ""}, ${failures} real failure(s); approve enter-${routed} = fix + re-verify + one more review round; reject = end this run and decide the merge on the evidence you have`,
878
+ ] : []),
879
+ `review found ${blockingFindings.length} blocking finding(s); ${triage ? "triage" : "approve another round"} before ${routed} runs`,
880
+ ...blockingFindings.slice(0, 5).map((finding) =>
881
+ `[${finding.severity} ${fingerprintFinding(finding)}] ${finding.summary.slice(0, 200)}`),
882
+ `adjudicate fingerprints (findings adjudicate), then approve enter-${routed} or reject the run`,
883
+ ],
884
+ triage ? "finding-triage" : "budget",
885
+ grants,
886
+ );
887
+ return;
888
+ }
759
889
  }
760
890
  step = routed;
761
891
  }
@@ -797,13 +927,26 @@ function supersedeWaitingRuns(repoRoot, workId, newRunId, now) {
797
927
  continue;
798
928
  }
799
929
  try {
800
- ledger.append({
930
+ // Re-read under the lock: the run may have been approved, resumed or
931
+ // stopped since the scan, and the scan's read must not be written to.
932
+ const locked = EventLedger.open(ledgerPath);
933
+ const fresh = locked.state;
934
+ if (
935
+ locked.corruption ||
936
+ !fresh.run ||
937
+ fresh.run.work !== workId ||
938
+ fresh.terminal ||
939
+ fresh.run.status !== "WAITING_HUMAN"
940
+ ) {
941
+ continue;
942
+ }
943
+ locked.append({
801
944
  type: "RUN_TERMINAL",
802
945
  actor: KERNEL,
803
946
  ts: now(),
804
947
  data: { status: "SUPERSEDED", reason: `superseded by ${newRunId} (same work ${workId})` },
805
948
  });
806
- writeRunRecord({ repoRoot, ledger, ts: now() });
949
+ writeRunRecord({ repoRoot, ledger: locked, ts: now() });
807
950
  superseded.push(entry);
808
951
  } finally {
809
952
  releaseLock(repoRoot, entry);
@@ -856,7 +999,7 @@ export function startRun(options) {
856
999
  }
857
1000
 
858
1001
  return withRunLocks(repoRoot, runId, () => {
859
- const workspace = createWorkspace({ repoRoot, runId, base });
1002
+ const workspace = withRepoGitLock(repoRoot, () => createWorkspace({ repoRoot, runId, base }));
860
1003
  const context = makeContext(options, ledger, workspace);
861
1004
  const now = context.now;
862
1005
  const supersession =
@@ -904,7 +1047,7 @@ export function startRun(options) {
904
1047
  superseded: supersession.superseded,
905
1048
  supersedeSkipped: supersession.skipped,
906
1049
  };
907
- });
1050
+ }, { workId, parallel: options.parallel === true });
908
1051
  }
909
1052
 
910
1053
  function resumeStepFromTransition(transition) {
@@ -917,15 +1060,8 @@ function resumeStepFromTransition(transition) {
917
1060
  return null;
918
1061
  }
919
1062
 
920
- export function resumeRun(options) {
921
- const { repoRoot, workflow, workflowDigest, runId, planDigest } = options;
922
- if (!repoRoot || !runId) {
923
- throw new OrchestratorError("repoRoot and runId are required");
924
- }
925
- if (options.requires?.length) {
926
- assertRequires(options.requires);
927
- }
928
- const { ledger, ledgerPath } = openLedgerFor(repoRoot, runId);
1063
+ function resumeTarget(options, { ledger, ledgerPath }) {
1064
+ const { repoRoot, workflowDigest, runId } = options;
929
1065
  const state = ledger.state;
930
1066
  if (!state.run) {
931
1067
  throw new OrchestratorError(`no ledger for run ${runId}; use startRun`);
@@ -936,12 +1072,11 @@ export function resumeRun(options) {
936
1072
  );
937
1073
  }
938
1074
  if (state.terminal) {
939
- return { runId, ledgerPath, state, resumed: false, reason: "run is terminal" };
1075
+ return { early: { runId, ledgerPath, state, resumed: false, reason: "run is terminal" } };
940
1076
  }
941
1077
  if (state.run.status === "WAITING_HUMAN" && state.pendingHuman) {
942
- return { runId, ledgerPath, state, resumed: false, reason: "waiting on a human decision" };
1078
+ return { early: { runId, ledgerPath, state, resumed: false, reason: "waiting on a human decision" } };
943
1079
  }
944
-
945
1080
  const bound = state.workspaces[runId];
946
1081
  if (!bound) {
947
1082
  throw new OrchestratorError(`run ${runId} has no bound workspace; cannot resume`);
@@ -952,15 +1087,45 @@ export function resumeRun(options) {
952
1087
  "worktree missing; recovery requires a human decision",
953
1088
  );
954
1089
  }
955
- const workspace = {
956
- workspaceId: runId,
957
- repoRoot,
958
- worktreePath,
959
- branch: bound.branch,
960
- base: bound.base,
1090
+ return {
1091
+ workspace: {
1092
+ workspaceId: runId,
1093
+ repoRoot,
1094
+ worktreePath,
1095
+ branch: bound.branch,
1096
+ base: bound.base,
1097
+ },
961
1098
  };
1099
+ }
1100
+
1101
+ export function resumeRun(options) {
1102
+ const { repoRoot, runId, planDigest } = options;
1103
+ if (!repoRoot || !runId) {
1104
+ throw new OrchestratorError("repoRoot and runId are required");
1105
+ }
1106
+ if (options.requires?.length) {
1107
+ assertRequires(options.requires);
1108
+ }
1109
+ // The read before the locks only answers early (terminal, waiting on a
1110
+ // human) without taking the repository lock. Everything resume decides is
1111
+ // decided again on a ledger read under the locks: another session may
1112
+ // have approved, resumed or stopped the run in between, and writing
1113
+ // through the earlier read would fork the hash chain.
1114
+ const outer = openLedgerFor(repoRoot, runId);
1115
+ const outerLedger = outer.ledger;
1116
+ const outside = resumeTarget(options, outer);
1117
+ if (outside.early) {
1118
+ return outside.early;
1119
+ }
962
1120
 
963
1121
  return withRunLocks(repoRoot, runId, () => {
1122
+ const { ledger, ledgerPath } = openLedgerFor(repoRoot, runId);
1123
+ const target = resumeTarget(options, { ledger, ledgerPath });
1124
+ if (target.early) {
1125
+ return target.early;
1126
+ }
1127
+ const { workspace } = target;
1128
+ const state = ledger.state;
964
1129
  const context = makeContext(options, ledger, workspace);
965
1130
  const now = context.now;
966
1131
 
@@ -1004,36 +1169,73 @@ export function resumeRun(options) {
1004
1169
  // "one more"; record the grant before driving or the same request
1005
1170
  // comes straight back (the pilot's app-login runs ended CANCELLED
1006
1171
  // with their candidates in production because of exactly that).
1007
- const requestKind = [...ledger.events]
1008
- .reverse()
1009
- .find((event) => event.type === "HUMAN_REQUESTED" && event.data.transition === approval.transition)
1010
- ?.data.kind;
1011
- if (requestKind === "work-review-cap") {
1012
- ledger.append({
1013
- type: "BUDGET_EXTENDED",
1014
- actor: KERNEL,
1015
- ts: now(),
1016
- data: {
1017
- step,
1018
- amount: 1,
1019
- scope: "work",
1020
- maxAttempts: (context.runBudgets.reviewRoundsPerWork ?? 0) + (state.workReviewGrants ?? 0) + 1,
1021
- approvalRef: approval.decisionRef,
1022
- },
1023
- });
1172
+ const decisionIndex = ledger.events.findIndex((event) =>
1173
+ event.type === "DECISION_RECORDED" && event.data.decisionRef === approval.decisionRef);
1174
+ // APPROVAL_STALE is itself a new request, with no inherited grants.
1175
+ // Restrict lookup to this decision, rather than an older request for
1176
+ // the same transition (or one whose subject was refreshed).
1177
+ const request = ledger.events.slice(0, decisionIndex).reverse().find((event) =>
1178
+ event.type === "HUMAN_REQUESTED" || event.type === "APPROVAL_STALE");
1179
+ const requestData = request?.type === "HUMAN_REQUESTED" &&
1180
+ request.data.transition === approval.transition ? request.data : null;
1181
+ // The grant plan is fixed once, then pinned on the first
1182
+ // BUDGET_EXTENDED it produces; a resume after a crash between two
1183
+ // grants replays that plan instead of re-deriving it from a state the
1184
+ // first grant already raised.
1185
+ const applied = ledger.events.filter((event) =>
1186
+ event.type === "BUDGET_EXTENDED" && event.data.approvalRef === approval.decisionRef);
1187
+ let grants;
1188
+ if (Array.isArray(applied[0]?.data.grants)) {
1189
+ grants = applied[0].data.grants;
1024
1190
  } else if (
1025
- approval.transition.startsWith("resume-") &&
1026
- (state.steps[step]?.attempts ?? 0) >= context.maxAttemptsFor(step)
1191
+ Array.isArray(requestData?.grants) &&
1192
+ (canonicalJson(requestData.subject) === canonicalJson(approval.subject) ||
1193
+ // resume --adopt answers this very request with a new candidate by
1194
+ // design; the grant belongs to the round, not to a candidate. Real
1195
+ // incident: the session's hand fix was adopted and the run stopped
1196
+ // again at resume-review for the round the human had just granted.
1197
+ (ledger.events[decisionIndex]?.data.adopted &&
1198
+ approval.subject.planDigest === requestData.subject.planDigest))
1027
1199
  ) {
1200
+ grants = requestData.grants;
1201
+ } else {
1202
+ // Requests without grants: ledgers written before grants existed,
1203
+ // or a request refreshed by APPROVAL_STALE.
1204
+ grants = [];
1205
+ const attempts = state.steps[step]?.attempts ?? 0;
1206
+ const runCapApproval = approval.transition.startsWith("resume-") &&
1207
+ (attempts >= context.maxAttemptsFor(step) || attempts >= context.totalAttemptsFor(step));
1208
+ if (requestData?.kind === "work-review-cap") grants.push({ step, scope: "work" });
1209
+ if (runCapApproval) grants.push({ step, scope: "run" });
1210
+ if (grants.length > 0) grants.push(...budgetGrants(context, step));
1211
+ }
1212
+ const plan = [];
1213
+ const planned = new Set();
1214
+ for (const grant of grants) {
1215
+ const key = `${grant.step}:${grant.scope}`;
1216
+ if (!planned.has(key)) {
1217
+ planned.add(key);
1218
+ plan.push({ step: grant.step, scope: grant.scope });
1219
+ }
1220
+ }
1221
+ const extended = new Set(applied.map((event) => `${event.data.step}:${event.data.scope ?? "run"}`));
1222
+ for (const grant of plan) {
1223
+ const key = `${grant.step}:${grant.scope}`;
1224
+ if (extended.has(key)) continue;
1225
+ extended.add(key);
1028
1226
  ledger.append({
1029
1227
  type: "BUDGET_EXTENDED",
1030
1228
  actor: KERNEL,
1031
1229
  ts: now(),
1032
1230
  data: {
1033
- step,
1231
+ step: grant.step,
1034
1232
  amount: 1,
1035
- maxAttempts: context.maxAttemptsFor(step) + 1,
1233
+ ...(grant.scope === "work" ? { scope: "work" } : {}),
1234
+ maxAttempts: grant.scope === "work"
1235
+ ? (context.runBudgets.reviewRoundsPerWork ?? 0) + (ledger.state.workReviewGrants ?? 0) + 1
1236
+ : context.maxAttemptsFor(grant.step) + 1,
1036
1237
  approvalRef: approval.decisionRef,
1238
+ grants: plan,
1037
1239
  },
1038
1240
  });
1039
1241
  }
@@ -1089,5 +1291,5 @@ export function resumeRun(options) {
1089
1291
  drive(context, startStep);
1090
1292
  }
1091
1293
  return { runId, ledgerPath, state: ledger.state, resumed: true, reason: null };
1092
- });
1294
+ }, { workId: outerLedger.state.run.work, parallel: options.parallel === true });
1093
1295
  }
@@ -4,7 +4,7 @@
4
4
  // further appends — recovery is a human decision, never a silent repair.
5
5
 
6
6
  import { createHash } from "node:crypto";
7
- import { appendFileSync, existsSync, mkdirSync, readFileSync } from "node:fs";
7
+ import { appendFileSync, existsSync, mkdirSync, readFileSync, statSync } from "node:fs";
8
8
  import { dirname } from "node:path";
9
9
 
10
10
  import { ENVELOPE_VERSION, GENESIS_DIGEST, validateEventInput } from "../domain/event-registry.js";
@@ -49,6 +49,9 @@ export class EventLedger {
49
49
  #run = null;
50
50
  #work = null;
51
51
  #reducer;
52
+ // Bytes of the file this instance has read or written. A writer whose
53
+ // count no longer matches the file is stale: someone else appended since.
54
+ #size = 0;
52
55
 
53
56
  constructor(filePath, reducer = RUN_REDUCER) {
54
57
  this.path = filePath;
@@ -72,7 +75,10 @@ export class EventLedger {
72
75
  }
73
76
 
74
77
  #load() {
75
- const lines = readFileSync(this.path, "utf8")
78
+ const raw = readFileSync(this.path);
79
+ this.#size = raw.length;
80
+ const lines = raw
81
+ .toString("utf8")
76
82
  .split("\n")
77
83
  .filter((line) => line.length > 0);
78
84
  for (const [index, line] of lines.entries()) {
@@ -142,8 +148,21 @@ export class EventLedger {
142
148
  };
143
149
  event.digest = eventDigest(event);
144
150
  const nextState = this.#reducer.applyEvent(this.state, event);
151
+ // Guard against a stale writer: every event carries seq and prev from
152
+ // this instance's memory, so appending after another writer would fork
153
+ // the hash chain and leave the ledger corrupted for good. Refuse
154
+ // instead; re-reading and retrying is always safe. Writers re-read under
155
+ // the run lock, so this only fires on a writer that forgot to.
156
+ const onDisk = existsSync(this.path) ? statSync(this.path).size : 0;
157
+ if (onDisk !== this.#size) {
158
+ throw new LedgerError(
159
+ `ledger for ${runId} changed on disk since it was read (another writer); re-read it and retry`,
160
+ );
161
+ }
162
+ const line = `${JSON.stringify(event)}\n`;
145
163
  mkdirSync(dirname(this.path), { recursive: true });
146
- appendFileSync(this.path, `${JSON.stringify(event)}\n`, "utf8");
164
+ appendFileSync(this.path, line, "utf8");
165
+ this.#size += Buffer.byteLength(line, "utf8");
147
166
  this.events.push(event);
148
167
  this.state = nextState;
149
168
  this.lastDigest = event.digest;