@haiyangbg/buildbeat 3.0.1 → 3.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +40 -9
- package/SKILL.md +23 -300
- package/docs/CAPABILITY-MATRIX.md +1 -1
- package/docs/README.md +5 -5
- package/docs/RELEASING.md +5 -5
- package/docs/v2/RFC-0001-product-definition.md +5 -5
- package/docs/v2/RFC-0002-domain-model.md +1 -1
- package/docs/v2/RFC-0003-workflow-policy.md +2 -2
- package/docs/v2/SPEC-0001-events-v1.md +1 -1
- package/docs/v2/guide/01-quickstart.en.md +3 -1
- package/docs/v2/guide/01-quickstart.md +3 -1
- package/docs/v2/guide/02-workflow-guide.md +15 -0
- package/docs/v2/guide/07-approval-guide.en.md +4 -2
- package/docs/v2/guide/07-approval-guide.md +14 -2
- package/docs/v2/guide/09-security-boundaries.md +1 -1
- package/docs/v2/guide/10-recovery.en.md +21 -5
- package/docs/v2/guide/10-recovery.md +21 -5
- package/docs/v2/skill/01-principles.md +26 -0
- package/docs/v2/skill/02-project-layout.md +31 -0
- package/docs/v2/skill/03-collaboration-rules.md +45 -0
- package/docs/v2/skill/04-rhythm-and-rituals.md +102 -0
- package/docs/v2/skill/05-red-lines.md +13 -0
- package/docs/v2/skill/06-bootstrap-and-takeover.md +84 -0
- package/docs/v2/skill/07-templates-and-lessons.md +24 -0
- package/package.json +5 -14
- package/src/v2/cli/run-config-check.js +229 -0
- package/src/v2/cli/run.js +81 -15
- package/src/v2/engine/reducer.js +6 -0
- package/src/v2/engine/yaml-subset.js +52 -11
- package/src/v2/presets/policies/ui-render-gate.yaml +1 -1
- package/src/v2/runtime/decisions.js +33 -31
- package/src/v2/runtime/gc.js +59 -18
- package/src/v2/runtime/metrics.js +3 -2
- package/src/v2/runtime/orchestrator.js +308 -106
- package/src/v2/storage/event-ledger.js +22 -3
- package/src/v2/workspace/workspace-manager.js +244 -16
- package/templates/v2/CLAUDE.md +1 -1
- package/templates/v2/run-config.example.yaml +5 -1
|
@@ -21,9 +21,12 @@ import { EventLedger, canonicalJson } from "../storage/event-ledger.js";
|
|
|
21
21
|
import {
|
|
22
22
|
acquireLock,
|
|
23
23
|
createWorkspace,
|
|
24
|
+
describeLockOwner,
|
|
24
25
|
listChangedPaths,
|
|
26
|
+
liveParallelMarkers,
|
|
25
27
|
readback,
|
|
26
28
|
releaseLock,
|
|
29
|
+
withRepoGitLock,
|
|
27
30
|
} from "../workspace/workspace-manager.js";
|
|
28
31
|
import { writeRunRecord } from "./run-record.js";
|
|
29
32
|
import { computeWorkCost } from "./work-cost.js";
|
|
@@ -52,31 +55,105 @@ function sha256(text) {
|
|
|
52
55
|
return `sha256:${createHash("sha256").update(text, "utf8").digest("hex")}`;
|
|
53
56
|
}
|
|
54
57
|
|
|
55
|
-
//
|
|
56
|
-
// repository-wide lock
|
|
58
|
+
// By default one run drives a repository at a time: it holds the
|
|
59
|
+
// repository-wide active-run lock for its whole drive. A run whose config
|
|
60
|
+
// sets `parallel: true` instead holds a per-work lock and a marker, passing
|
|
61
|
+
// the active-run lock only briefly as a gate, so runs of different works can
|
|
62
|
+
// drive together while runs of the same work stay exclusive. Real incident:
|
|
63
|
+
// a session waited 3h23m behind another work's run although worktrees were
|
|
64
|
+
// already isolated; parallelism is opt-in because verifiers that bind fixed
|
|
65
|
+
// ports or share a database would collide.
|
|
57
66
|
const ACTIVE_LOCK = "active-run";
|
|
58
67
|
|
|
59
68
|
function lockActive(repoRoot) {
|
|
60
69
|
try {
|
|
61
70
|
acquireLock(repoRoot, ACTIVE_LOCK);
|
|
62
|
-
} catch {
|
|
71
|
+
} catch (error) {
|
|
72
|
+
// A lock whose owner is gone was already reclaimed inside acquireLock;
|
|
73
|
+
// what reaches here is held (or unreadable), and the owner is the one
|
|
74
|
+
// fact a blocked caller needs.
|
|
75
|
+
const detail = error.lock?.detail;
|
|
63
76
|
throw new OrchestratorError(
|
|
64
|
-
|
|
77
|
+
`another run is active in this repository (MVP allows a single active run)${detail ? `; ${detail}` : ""}`,
|
|
65
78
|
);
|
|
66
79
|
}
|
|
67
80
|
}
|
|
68
81
|
|
|
69
|
-
function
|
|
82
|
+
function lockActiveWaiting(repoRoot, waitMs) {
|
|
83
|
+
const deadline = Date.now() + waitMs;
|
|
84
|
+
for (;;) {
|
|
85
|
+
try {
|
|
86
|
+
acquireLock(repoRoot, ACTIVE_LOCK);
|
|
87
|
+
return;
|
|
88
|
+
} catch (error) {
|
|
89
|
+
if (!error.lock || Date.now() >= deadline) {
|
|
90
|
+
break;
|
|
91
|
+
}
|
|
92
|
+
Atomics.wait(new Int32Array(new SharedArrayBuffer(4)), 0, 0, 25);
|
|
93
|
+
}
|
|
94
|
+
}
|
|
70
95
|
lockActive(repoRoot);
|
|
96
|
+
}
|
|
97
|
+
|
|
98
|
+
function holdRunLock(repoRoot, runId, fn) {
|
|
99
|
+
acquireLock(repoRoot, runId);
|
|
100
|
+
try {
|
|
101
|
+
return fn();
|
|
102
|
+
} finally {
|
|
103
|
+
releaseLock(repoRoot, runId);
|
|
104
|
+
}
|
|
105
|
+
}
|
|
106
|
+
|
|
107
|
+
function withRunLocks(repoRoot, runId, fn, { workId = null, parallel = false } = {}) {
|
|
108
|
+
if (!parallel) {
|
|
109
|
+
lockActive(repoRoot);
|
|
110
|
+
try {
|
|
111
|
+
const running = liveParallelMarkers(repoRoot);
|
|
112
|
+
if (running.length > 0) {
|
|
113
|
+
throw new OrchestratorError(
|
|
114
|
+
`another run is active in this repository (parallel run(s) ${running
|
|
115
|
+
.map((marker) => (marker.owner ? `${marker.run}, ${describeLockOwner(marker.owner)}` : marker.run))
|
|
116
|
+
.join("; ")}); this run is exclusive (set parallel: true in its run config to drive alongside other works)`,
|
|
117
|
+
);
|
|
118
|
+
}
|
|
119
|
+
return holdRunLock(repoRoot, runId, fn);
|
|
120
|
+
} finally {
|
|
121
|
+
releaseLock(repoRoot, ACTIVE_LOCK);
|
|
122
|
+
}
|
|
123
|
+
}
|
|
124
|
+
if (!workId) {
|
|
125
|
+
throw new OrchestratorError("a parallel run needs its work id");
|
|
126
|
+
}
|
|
127
|
+
const workLock = `@work.${workId}`;
|
|
128
|
+
const marker = `@parallel.${runId}`;
|
|
129
|
+
try {
|
|
130
|
+
acquireLock(repoRoot, workLock);
|
|
131
|
+
} catch (error) {
|
|
132
|
+
const detail = error.lock?.detail;
|
|
133
|
+
throw new OrchestratorError(
|
|
134
|
+
`another run of ${workId} is active (runs of the same work never drive together)${detail ? `; ${detail}` : ""}`,
|
|
135
|
+
);
|
|
136
|
+
}
|
|
71
137
|
try {
|
|
72
|
-
|
|
138
|
+
// Gate: an exclusive run holds active-run for its whole drive, so a
|
|
139
|
+
// parallel run cannot slip in while it runs; the marker, created under
|
|
140
|
+
// the gate, is what an exclusive run checks before it starts. Another
|
|
141
|
+
// parallel run holds the gate for milliseconds, so a busy gate is waited
|
|
142
|
+
// for briefly before it is reported (two parallel starts at the same
|
|
143
|
+
// instant must both get through).
|
|
144
|
+
lockActiveWaiting(repoRoot, 2000);
|
|
73
145
|
try {
|
|
74
|
-
|
|
146
|
+
acquireLock(repoRoot, marker);
|
|
75
147
|
} finally {
|
|
76
|
-
releaseLock(repoRoot,
|
|
148
|
+
releaseLock(repoRoot, ACTIVE_LOCK);
|
|
149
|
+
}
|
|
150
|
+
try {
|
|
151
|
+
return holdRunLock(repoRoot, runId, fn);
|
|
152
|
+
} finally {
|
|
153
|
+
releaseLock(repoRoot, marker);
|
|
77
154
|
}
|
|
78
155
|
} finally {
|
|
79
|
-
releaseLock(repoRoot,
|
|
156
|
+
releaseLock(repoRoot, workLock);
|
|
80
157
|
}
|
|
81
158
|
}
|
|
82
159
|
|
|
@@ -149,12 +226,17 @@ function makeContext(options, ledger, workspace) {
|
|
|
149
226
|
// review rounds could not be raised from the run config, and approving
|
|
150
227
|
// resume-review re-asked the same question forever.
|
|
151
228
|
context.runBudgets = options.budgets ?? {};
|
|
152
|
-
context.
|
|
229
|
+
context.budgetLimitFor = (step) =>
|
|
153
230
|
(context.runBudgets.maxAttempts?.[step] ??
|
|
154
231
|
workflow.budgets?.maxAttempts?.[step] ??
|
|
155
232
|
maxAttemptsPerStep) +
|
|
156
|
-
(ledger.state.budgetExtensions?.[step] ?? 0)
|
|
157
|
-
|
|
233
|
+
(ledger.state.budgetExtensions?.[step] ?? 0);
|
|
234
|
+
context.maxAttemptsFor = (step) => context.budgetLimitFor(step) +
|
|
235
|
+
(ledger.state.steps[step]?.infraAttempts ?? 0) +
|
|
236
|
+
(ledger.state.steps[step]?.freeAttempts ?? 0);
|
|
237
|
+
// Refunds must not move this ceiling: otherwise a success-only loop has
|
|
238
|
+
// an ever-growing limit. Human extensions deliberately raise it.
|
|
239
|
+
context.totalAttemptsFor = (step) => context.budgetLimitFor(step) * 3;
|
|
158
240
|
context.policies = options.policies ?? [];
|
|
159
241
|
context.allowedPaths = options.allowedPaths ?? null;
|
|
160
242
|
context.reviewTriage = options.reviewTriage ?? null;
|
|
@@ -179,17 +261,64 @@ function makeContext(options, ledger, workspace) {
|
|
|
179
261
|
evidenceDigest: lastEvidence?.digest ?? "UNVERIFIED",
|
|
180
262
|
};
|
|
181
263
|
};
|
|
182
|
-
context.waitHuman = (transition, reasons, kind = "boundary") => {
|
|
264
|
+
context.waitHuman = (transition, reasons, kind = "boundary", grants = []) => {
|
|
183
265
|
ledger.append({
|
|
184
266
|
type: "HUMAN_REQUESTED",
|
|
185
267
|
actor: KERNEL,
|
|
186
268
|
ts: context.now(),
|
|
187
|
-
data: { transition, subject: context.subjectNow(), reasons, kind },
|
|
269
|
+
data: { transition, subject: context.subjectNow(), reasons, kind, ...(grants.length ? { grants } : {}) },
|
|
188
270
|
});
|
|
189
271
|
};
|
|
190
272
|
return context;
|
|
191
273
|
}
|
|
192
274
|
|
|
275
|
+
// Keep the charged budget distinct from the worker invocation number.
|
|
276
|
+
function budgetUsage(context, step) {
|
|
277
|
+
const state = context.ledger.state.steps[step];
|
|
278
|
+
const attempts = state?.attempts ?? 0;
|
|
279
|
+
const used = attempts - (state?.infraAttempts ?? 0) - (state?.freeAttempts ?? 0);
|
|
280
|
+
const failures = context.ledger.events.filter((event) =>
|
|
281
|
+
event.type === "STEP_FINISHED" && event.data.step === step &&
|
|
282
|
+
event.data.status !== "succeeded" && event.data.infra !== true).length;
|
|
283
|
+
return { attempts, used, failures, limit: context.budgetLimitFor(step) };
|
|
284
|
+
}
|
|
285
|
+
|
|
286
|
+
function budgetReasons(context, step, safeguard = false) {
|
|
287
|
+
const { attempts, used, failures, limit } = budgetUsage(context, step);
|
|
288
|
+
const charging = context.workflow.steps.find((item) => item.id === step)?.readonly
|
|
289
|
+
? "each round is charged" : "successful attempts are not charged";
|
|
290
|
+
return [
|
|
291
|
+
safeguard
|
|
292
|
+
? `${step} budget exhausted (runaway safeguard): ${attempts}/${context.totalAttemptsFor(step)} total attempt(s), ${failures} real failure(s)`
|
|
293
|
+
: `${step} budget exhausted: ${used}/${limit} charged attempt(s) used, ${failures} real failure(s) (${charging})`,
|
|
294
|
+
`approve resume-${step} = ${safeguard ? "raise the safeguard and continue" : "one more attempt"}; reject = end this run and decide the merge on the evidence you have`,
|
|
295
|
+
];
|
|
296
|
+
}
|
|
297
|
+
|
|
298
|
+
function workReviewBudget(context, step) {
|
|
299
|
+
const stepDef = context.workflow.steps.find((item) => item.id === step);
|
|
300
|
+
const cap = context.runBudgets.reviewRoundsPerWork;
|
|
301
|
+
if (cap === undefined || !(step === "review" || stepDef?.worker === "reviewer")) return null;
|
|
302
|
+
const prior = computeWorkCost(context.repoRoot, context.ledger.state.run.work, {
|
|
303
|
+
excludeRun: context.ledger.state.run.id,
|
|
304
|
+
});
|
|
305
|
+
const rounds = prior.reviewRounds + (context.ledger.state.steps[step]?.attempts ?? 0);
|
|
306
|
+
const allowed = cap + (context.ledger.state.workReviewGrants ?? 0);
|
|
307
|
+
return { rounds, allowed, exhausted: rounds >= allowed };
|
|
308
|
+
}
|
|
309
|
+
|
|
310
|
+
// Every cap the next attempt of `step` would hit. Recorded on the request
|
|
311
|
+
// at stop time, so one approval lifts both layers (run and work) at once.
|
|
312
|
+
function budgetGrants(context, step) {
|
|
313
|
+
const grants = [];
|
|
314
|
+
if ((context.ledger.state.steps[step]?.attempts ?? 0) >= context.maxAttemptsFor(step) ||
|
|
315
|
+
(context.ledger.state.steps[step]?.attempts ?? 0) >= context.totalAttemptsFor(step)) {
|
|
316
|
+
grants.push({ step, scope: "run" });
|
|
317
|
+
}
|
|
318
|
+
if (workReviewBudget(context, step)?.exhausted) grants.push({ step, scope: "work" });
|
|
319
|
+
return grants;
|
|
320
|
+
}
|
|
321
|
+
|
|
193
322
|
// Evaluates configured policies of `type` for `appliesTo`, records every
|
|
194
323
|
// verdict as a POLICY_EVALUATED event, and reports what the kernel must do.
|
|
195
324
|
// ADVISORY failures are recorded but never gate (doctor reports the gap).
|
|
@@ -249,10 +378,7 @@ function settleOutcome(context, step, outcome, tree, exec) {
|
|
|
249
378
|
// A step that failed its final attempt can never run again, so routing
|
|
250
379
|
// to fix would spend a worker on a candidate nothing can verify.
|
|
251
380
|
if ((ledger.state.steps[step]?.attempts ?? 0) >= context.maxAttemptsFor(step)) {
|
|
252
|
-
context.waitHuman(`resume-${step}`,
|
|
253
|
-
`budget exhausted: ${step} failed its final attempt (maxAttempts=${context.maxAttemptsFor(step)}); not routing to fix`,
|
|
254
|
-
`approving resume-${step} grants one more attempt; rejecting ends the run`,
|
|
255
|
-
]);
|
|
381
|
+
context.waitHuman(`resume-${step}`, budgetReasons(context, step), "budget", budgetGrants(context, step));
|
|
256
382
|
return null;
|
|
257
383
|
}
|
|
258
384
|
}
|
|
@@ -336,24 +462,19 @@ function drive(context, startStep, { skipBoundaryOnce = false } = {}) {
|
|
|
336
462
|
// every run of the work, superseded ones included, so "one run per
|
|
337
463
|
// round" cannot slip past the per-run budget. Reaching the cap is a
|
|
338
464
|
// human decision (review once more, or merge/close as-is), not a stop.
|
|
339
|
-
const
|
|
340
|
-
if (
|
|
341
|
-
const
|
|
342
|
-
|
|
343
|
-
|
|
344
|
-
|
|
345
|
-
|
|
346
|
-
|
|
347
|
-
|
|
348
|
-
|
|
349
|
-
|
|
350
|
-
|
|
351
|
-
|
|
352
|
-
],
|
|
353
|
-
"work-review-cap",
|
|
354
|
-
);
|
|
355
|
-
return;
|
|
356
|
-
}
|
|
465
|
+
const workBudget = workReviewBudget(context, step);
|
|
466
|
+
if (workBudget?.exhausted) {
|
|
467
|
+
const { failures } = budgetUsage(context, step);
|
|
468
|
+
context.waitHuman(
|
|
469
|
+
`enter-${step}`,
|
|
470
|
+
[
|
|
471
|
+
`${step} work review budget exhausted: ${workBudget.rounds}/${workBudget.allowed} review round(s) across the work, ${failures} real failure(s) in this run`,
|
|
472
|
+
`approve enter-${step} = one more review round (also lifts this run's review cap if it is spent); reject = end this run and decide the merge on the evidence you have`,
|
|
473
|
+
],
|
|
474
|
+
"work-review-cap",
|
|
475
|
+
budgetGrants(context, step),
|
|
476
|
+
);
|
|
477
|
+
return;
|
|
357
478
|
}
|
|
358
479
|
|
|
359
480
|
const preGate = runPolicyGate(context, "pre", step);
|
|
@@ -375,10 +496,12 @@ function drive(context, startStep, { skipBoundaryOnce = false } = {}) {
|
|
|
375
496
|
const attempt = (ledger.state.steps[step]?.attempts ?? 0) + 1;
|
|
376
497
|
const maxAttempts = context.maxAttemptsFor(step);
|
|
377
498
|
if (attempt > maxAttempts) {
|
|
378
|
-
context.waitHuman(`resume-${step}`,
|
|
379
|
-
|
|
380
|
-
|
|
381
|
-
|
|
499
|
+
context.waitHuman(`resume-${step}`, budgetReasons(context, step), "budget", budgetGrants(context, step));
|
|
500
|
+
return;
|
|
501
|
+
}
|
|
502
|
+
|
|
503
|
+
if (attempt > context.totalAttemptsFor(step)) {
|
|
504
|
+
context.waitHuman(`resume-${step}`, budgetReasons(context, step, true), "budget", budgetGrants(context, step));
|
|
382
505
|
return;
|
|
383
506
|
}
|
|
384
507
|
|
|
@@ -581,11 +704,13 @@ function drive(context, startStep, { skipBoundaryOnce = false } = {}) {
|
|
|
581
704
|
stepStatus === "crashed" ||
|
|
582
705
|
stepStatus === "invalid-output" ||
|
|
583
706
|
(stepStatus === "failed" && exec.exitCode === 75);
|
|
707
|
+
const free = stepStatus === "succeeded" && stepDef.readonly !== true;
|
|
584
708
|
ledger.append({
|
|
585
709
|
type: "STEP_FINISHED",
|
|
586
710
|
actor: KERNEL,
|
|
587
711
|
ts: now(),
|
|
588
|
-
data: { step, attempt, status: stepStatus, exitCode: exec.exitCode,
|
|
712
|
+
data: { step, attempt, status: stepStatus, exitCode: exec.exitCode,
|
|
713
|
+
...(infra ? { infra: true } : {}), ...(free ? { free: true } : {}) },
|
|
589
714
|
});
|
|
590
715
|
ledger.append({
|
|
591
716
|
type: "BUDGET_CONSUMED",
|
|
@@ -593,7 +718,7 @@ function drive(context, startStep, { skipBoundaryOnce = false } = {}) {
|
|
|
593
718
|
ts: now(),
|
|
594
719
|
data: {
|
|
595
720
|
kind: "attempts",
|
|
596
|
-
amount: infra ? 0 : 1,
|
|
721
|
+
amount: infra || free ? 0 : 1,
|
|
597
722
|
remaining: context.maxAttemptsFor(step) - attempt,
|
|
598
723
|
},
|
|
599
724
|
});
|
|
@@ -736,26 +861,31 @@ function drive(context, startStep, { skipBoundaryOnce = false } = {}) {
|
|
|
736
861
|
outcome = "succeeded";
|
|
737
862
|
}
|
|
738
863
|
const routed = settleOutcome(context, step, outcome, tree, exec);
|
|
739
|
-
//
|
|
740
|
-
//
|
|
741
|
-
|
|
742
|
-
|
|
743
|
-
|
|
744
|
-
context.
|
|
745
|
-
|
|
746
|
-
|
|
747
|
-
|
|
748
|
-
|
|
749
|
-
|
|
750
|
-
|
|
751
|
-
|
|
752
|
-
|
|
753
|
-
),
|
|
754
|
-
|
|
755
|
-
|
|
756
|
-
|
|
757
|
-
|
|
758
|
-
|
|
864
|
+
// Ask before spending fix/verify workers: one approval covers the next
|
|
865
|
+
// round and both review caps, with the grant bound to this request.
|
|
866
|
+
if (routed && outcome === "findings-blocking") {
|
|
867
|
+
const isReview = step === "review" || stepDef.worker === "reviewer";
|
|
868
|
+
const grants = isReview ? budgetGrants(context, step) : [];
|
|
869
|
+
const triage = context.reviewTriage === "required";
|
|
870
|
+
if (triage || grants.length) {
|
|
871
|
+
const { used, limit, failures } = budgetUsage(context, step);
|
|
872
|
+
const workBudget = workReviewBudget(context, step);
|
|
873
|
+
context.waitHuman(
|
|
874
|
+
`enter-${routed}`,
|
|
875
|
+
[
|
|
876
|
+
...(grants.length ? [
|
|
877
|
+
`${step} budget exhausted: ${used}/${limit} review round(s) used in this run${workBudget ? `, ${workBudget.rounds}/${workBudget.allowed} across the work` : ""}, ${failures} real failure(s); approve enter-${routed} = fix + re-verify + one more review round; reject = end this run and decide the merge on the evidence you have`,
|
|
878
|
+
] : []),
|
|
879
|
+
`review found ${blockingFindings.length} blocking finding(s); ${triage ? "triage" : "approve another round"} before ${routed} runs`,
|
|
880
|
+
...blockingFindings.slice(0, 5).map((finding) =>
|
|
881
|
+
`[${finding.severity} ${fingerprintFinding(finding)}] ${finding.summary.slice(0, 200)}`),
|
|
882
|
+
`adjudicate fingerprints (findings adjudicate), then approve enter-${routed} or reject the run`,
|
|
883
|
+
],
|
|
884
|
+
triage ? "finding-triage" : "budget",
|
|
885
|
+
grants,
|
|
886
|
+
);
|
|
887
|
+
return;
|
|
888
|
+
}
|
|
759
889
|
}
|
|
760
890
|
step = routed;
|
|
761
891
|
}
|
|
@@ -797,13 +927,26 @@ function supersedeWaitingRuns(repoRoot, workId, newRunId, now) {
|
|
|
797
927
|
continue;
|
|
798
928
|
}
|
|
799
929
|
try {
|
|
800
|
-
|
|
930
|
+
// Re-read under the lock: the run may have been approved, resumed or
|
|
931
|
+
// stopped since the scan, and the scan's read must not be written to.
|
|
932
|
+
const locked = EventLedger.open(ledgerPath);
|
|
933
|
+
const fresh = locked.state;
|
|
934
|
+
if (
|
|
935
|
+
locked.corruption ||
|
|
936
|
+
!fresh.run ||
|
|
937
|
+
fresh.run.work !== workId ||
|
|
938
|
+
fresh.terminal ||
|
|
939
|
+
fresh.run.status !== "WAITING_HUMAN"
|
|
940
|
+
) {
|
|
941
|
+
continue;
|
|
942
|
+
}
|
|
943
|
+
locked.append({
|
|
801
944
|
type: "RUN_TERMINAL",
|
|
802
945
|
actor: KERNEL,
|
|
803
946
|
ts: now(),
|
|
804
947
|
data: { status: "SUPERSEDED", reason: `superseded by ${newRunId} (same work ${workId})` },
|
|
805
948
|
});
|
|
806
|
-
writeRunRecord({ repoRoot, ledger, ts: now() });
|
|
949
|
+
writeRunRecord({ repoRoot, ledger: locked, ts: now() });
|
|
807
950
|
superseded.push(entry);
|
|
808
951
|
} finally {
|
|
809
952
|
releaseLock(repoRoot, entry);
|
|
@@ -856,7 +999,7 @@ export function startRun(options) {
|
|
|
856
999
|
}
|
|
857
1000
|
|
|
858
1001
|
return withRunLocks(repoRoot, runId, () => {
|
|
859
|
-
const workspace = createWorkspace({ repoRoot, runId, base });
|
|
1002
|
+
const workspace = withRepoGitLock(repoRoot, () => createWorkspace({ repoRoot, runId, base }));
|
|
860
1003
|
const context = makeContext(options, ledger, workspace);
|
|
861
1004
|
const now = context.now;
|
|
862
1005
|
const supersession =
|
|
@@ -904,7 +1047,7 @@ export function startRun(options) {
|
|
|
904
1047
|
superseded: supersession.superseded,
|
|
905
1048
|
supersedeSkipped: supersession.skipped,
|
|
906
1049
|
};
|
|
907
|
-
});
|
|
1050
|
+
}, { workId, parallel: options.parallel === true });
|
|
908
1051
|
}
|
|
909
1052
|
|
|
910
1053
|
function resumeStepFromTransition(transition) {
|
|
@@ -917,15 +1060,8 @@ function resumeStepFromTransition(transition) {
|
|
|
917
1060
|
return null;
|
|
918
1061
|
}
|
|
919
1062
|
|
|
920
|
-
|
|
921
|
-
const { repoRoot,
|
|
922
|
-
if (!repoRoot || !runId) {
|
|
923
|
-
throw new OrchestratorError("repoRoot and runId are required");
|
|
924
|
-
}
|
|
925
|
-
if (options.requires?.length) {
|
|
926
|
-
assertRequires(options.requires);
|
|
927
|
-
}
|
|
928
|
-
const { ledger, ledgerPath } = openLedgerFor(repoRoot, runId);
|
|
1063
|
+
function resumeTarget(options, { ledger, ledgerPath }) {
|
|
1064
|
+
const { repoRoot, workflowDigest, runId } = options;
|
|
929
1065
|
const state = ledger.state;
|
|
930
1066
|
if (!state.run) {
|
|
931
1067
|
throw new OrchestratorError(`no ledger for run ${runId}; use startRun`);
|
|
@@ -936,12 +1072,11 @@ export function resumeRun(options) {
|
|
|
936
1072
|
);
|
|
937
1073
|
}
|
|
938
1074
|
if (state.terminal) {
|
|
939
|
-
return { runId, ledgerPath, state, resumed: false, reason: "run is terminal" };
|
|
1075
|
+
return { early: { runId, ledgerPath, state, resumed: false, reason: "run is terminal" } };
|
|
940
1076
|
}
|
|
941
1077
|
if (state.run.status === "WAITING_HUMAN" && state.pendingHuman) {
|
|
942
|
-
return { runId, ledgerPath, state, resumed: false, reason: "waiting on a human decision" };
|
|
1078
|
+
return { early: { runId, ledgerPath, state, resumed: false, reason: "waiting on a human decision" } };
|
|
943
1079
|
}
|
|
944
|
-
|
|
945
1080
|
const bound = state.workspaces[runId];
|
|
946
1081
|
if (!bound) {
|
|
947
1082
|
throw new OrchestratorError(`run ${runId} has no bound workspace; cannot resume`);
|
|
@@ -952,15 +1087,45 @@ export function resumeRun(options) {
|
|
|
952
1087
|
"worktree missing; recovery requires a human decision",
|
|
953
1088
|
);
|
|
954
1089
|
}
|
|
955
|
-
|
|
956
|
-
|
|
957
|
-
|
|
958
|
-
|
|
959
|
-
|
|
960
|
-
|
|
1090
|
+
return {
|
|
1091
|
+
workspace: {
|
|
1092
|
+
workspaceId: runId,
|
|
1093
|
+
repoRoot,
|
|
1094
|
+
worktreePath,
|
|
1095
|
+
branch: bound.branch,
|
|
1096
|
+
base: bound.base,
|
|
1097
|
+
},
|
|
961
1098
|
};
|
|
1099
|
+
}
|
|
1100
|
+
|
|
1101
|
+
export function resumeRun(options) {
|
|
1102
|
+
const { repoRoot, runId, planDigest } = options;
|
|
1103
|
+
if (!repoRoot || !runId) {
|
|
1104
|
+
throw new OrchestratorError("repoRoot and runId are required");
|
|
1105
|
+
}
|
|
1106
|
+
if (options.requires?.length) {
|
|
1107
|
+
assertRequires(options.requires);
|
|
1108
|
+
}
|
|
1109
|
+
// The read before the locks only answers early (terminal, waiting on a
|
|
1110
|
+
// human) without taking the repository lock. Everything resume decides is
|
|
1111
|
+
// decided again on a ledger read under the locks: another session may
|
|
1112
|
+
// have approved, resumed or stopped the run in between, and writing
|
|
1113
|
+
// through the earlier read would fork the hash chain.
|
|
1114
|
+
const outer = openLedgerFor(repoRoot, runId);
|
|
1115
|
+
const outerLedger = outer.ledger;
|
|
1116
|
+
const outside = resumeTarget(options, outer);
|
|
1117
|
+
if (outside.early) {
|
|
1118
|
+
return outside.early;
|
|
1119
|
+
}
|
|
962
1120
|
|
|
963
1121
|
return withRunLocks(repoRoot, runId, () => {
|
|
1122
|
+
const { ledger, ledgerPath } = openLedgerFor(repoRoot, runId);
|
|
1123
|
+
const target = resumeTarget(options, { ledger, ledgerPath });
|
|
1124
|
+
if (target.early) {
|
|
1125
|
+
return target.early;
|
|
1126
|
+
}
|
|
1127
|
+
const { workspace } = target;
|
|
1128
|
+
const state = ledger.state;
|
|
964
1129
|
const context = makeContext(options, ledger, workspace);
|
|
965
1130
|
const now = context.now;
|
|
966
1131
|
|
|
@@ -1004,36 +1169,73 @@ export function resumeRun(options) {
|
|
|
1004
1169
|
// "one more"; record the grant before driving or the same request
|
|
1005
1170
|
// comes straight back (the pilot's app-login runs ended CANCELLED
|
|
1006
1171
|
// with their candidates in production because of exactly that).
|
|
1007
|
-
const
|
|
1008
|
-
.
|
|
1009
|
-
|
|
1010
|
-
|
|
1011
|
-
|
|
1012
|
-
|
|
1013
|
-
|
|
1014
|
-
|
|
1015
|
-
|
|
1016
|
-
|
|
1017
|
-
|
|
1018
|
-
|
|
1019
|
-
|
|
1020
|
-
|
|
1021
|
-
|
|
1022
|
-
|
|
1023
|
-
|
|
1172
|
+
const decisionIndex = ledger.events.findIndex((event) =>
|
|
1173
|
+
event.type === "DECISION_RECORDED" && event.data.decisionRef === approval.decisionRef);
|
|
1174
|
+
// APPROVAL_STALE is itself a new request, with no inherited grants.
|
|
1175
|
+
// Restrict lookup to this decision, rather than an older request for
|
|
1176
|
+
// the same transition (or one whose subject was refreshed).
|
|
1177
|
+
const request = ledger.events.slice(0, decisionIndex).reverse().find((event) =>
|
|
1178
|
+
event.type === "HUMAN_REQUESTED" || event.type === "APPROVAL_STALE");
|
|
1179
|
+
const requestData = request?.type === "HUMAN_REQUESTED" &&
|
|
1180
|
+
request.data.transition === approval.transition ? request.data : null;
|
|
1181
|
+
// The grant plan is fixed once, then pinned on the first
|
|
1182
|
+
// BUDGET_EXTENDED it produces; a resume after a crash between two
|
|
1183
|
+
// grants replays that plan instead of re-deriving it from a state the
|
|
1184
|
+
// first grant already raised.
|
|
1185
|
+
const applied = ledger.events.filter((event) =>
|
|
1186
|
+
event.type === "BUDGET_EXTENDED" && event.data.approvalRef === approval.decisionRef);
|
|
1187
|
+
let grants;
|
|
1188
|
+
if (Array.isArray(applied[0]?.data.grants)) {
|
|
1189
|
+
grants = applied[0].data.grants;
|
|
1024
1190
|
} else if (
|
|
1025
|
-
|
|
1026
|
-
(
|
|
1191
|
+
Array.isArray(requestData?.grants) &&
|
|
1192
|
+
(canonicalJson(requestData.subject) === canonicalJson(approval.subject) ||
|
|
1193
|
+
// resume --adopt answers this very request with a new candidate by
|
|
1194
|
+
// design; the grant belongs to the round, not to a candidate. Real
|
|
1195
|
+
// incident: the session's hand fix was adopted and the run stopped
|
|
1196
|
+
// again at resume-review for the round the human had just granted.
|
|
1197
|
+
(ledger.events[decisionIndex]?.data.adopted &&
|
|
1198
|
+
approval.subject.planDigest === requestData.subject.planDigest))
|
|
1027
1199
|
) {
|
|
1200
|
+
grants = requestData.grants;
|
|
1201
|
+
} else {
|
|
1202
|
+
// Requests without grants: ledgers written before grants existed,
|
|
1203
|
+
// or a request refreshed by APPROVAL_STALE.
|
|
1204
|
+
grants = [];
|
|
1205
|
+
const attempts = state.steps[step]?.attempts ?? 0;
|
|
1206
|
+
const runCapApproval = approval.transition.startsWith("resume-") &&
|
|
1207
|
+
(attempts >= context.maxAttemptsFor(step) || attempts >= context.totalAttemptsFor(step));
|
|
1208
|
+
if (requestData?.kind === "work-review-cap") grants.push({ step, scope: "work" });
|
|
1209
|
+
if (runCapApproval) grants.push({ step, scope: "run" });
|
|
1210
|
+
if (grants.length > 0) grants.push(...budgetGrants(context, step));
|
|
1211
|
+
}
|
|
1212
|
+
const plan = [];
|
|
1213
|
+
const planned = new Set();
|
|
1214
|
+
for (const grant of grants) {
|
|
1215
|
+
const key = `${grant.step}:${grant.scope}`;
|
|
1216
|
+
if (!planned.has(key)) {
|
|
1217
|
+
planned.add(key);
|
|
1218
|
+
plan.push({ step: grant.step, scope: grant.scope });
|
|
1219
|
+
}
|
|
1220
|
+
}
|
|
1221
|
+
const extended = new Set(applied.map((event) => `${event.data.step}:${event.data.scope ?? "run"}`));
|
|
1222
|
+
for (const grant of plan) {
|
|
1223
|
+
const key = `${grant.step}:${grant.scope}`;
|
|
1224
|
+
if (extended.has(key)) continue;
|
|
1225
|
+
extended.add(key);
|
|
1028
1226
|
ledger.append({
|
|
1029
1227
|
type: "BUDGET_EXTENDED",
|
|
1030
1228
|
actor: KERNEL,
|
|
1031
1229
|
ts: now(),
|
|
1032
1230
|
data: {
|
|
1033
|
-
step,
|
|
1231
|
+
step: grant.step,
|
|
1034
1232
|
amount: 1,
|
|
1035
|
-
|
|
1233
|
+
...(grant.scope === "work" ? { scope: "work" } : {}),
|
|
1234
|
+
maxAttempts: grant.scope === "work"
|
|
1235
|
+
? (context.runBudgets.reviewRoundsPerWork ?? 0) + (ledger.state.workReviewGrants ?? 0) + 1
|
|
1236
|
+
: context.maxAttemptsFor(grant.step) + 1,
|
|
1036
1237
|
approvalRef: approval.decisionRef,
|
|
1238
|
+
grants: plan,
|
|
1037
1239
|
},
|
|
1038
1240
|
});
|
|
1039
1241
|
}
|
|
@@ -1089,5 +1291,5 @@ export function resumeRun(options) {
|
|
|
1089
1291
|
drive(context, startStep);
|
|
1090
1292
|
}
|
|
1091
1293
|
return { runId, ledgerPath, state: ledger.state, resumed: true, reason: null };
|
|
1092
|
-
});
|
|
1294
|
+
}, { workId: outerLedger.state.run.work, parallel: options.parallel === true });
|
|
1093
1295
|
}
|
|
@@ -4,7 +4,7 @@
|
|
|
4
4
|
// further appends — recovery is a human decision, never a silent repair.
|
|
5
5
|
|
|
6
6
|
import { createHash } from "node:crypto";
|
|
7
|
-
import { appendFileSync, existsSync, mkdirSync, readFileSync } from "node:fs";
|
|
7
|
+
import { appendFileSync, existsSync, mkdirSync, readFileSync, statSync } from "node:fs";
|
|
8
8
|
import { dirname } from "node:path";
|
|
9
9
|
|
|
10
10
|
import { ENVELOPE_VERSION, GENESIS_DIGEST, validateEventInput } from "../domain/event-registry.js";
|
|
@@ -49,6 +49,9 @@ export class EventLedger {
|
|
|
49
49
|
#run = null;
|
|
50
50
|
#work = null;
|
|
51
51
|
#reducer;
|
|
52
|
+
// Bytes of the file this instance has read or written. A writer whose
|
|
53
|
+
// count no longer matches the file is stale: someone else appended since.
|
|
54
|
+
#size = 0;
|
|
52
55
|
|
|
53
56
|
constructor(filePath, reducer = RUN_REDUCER) {
|
|
54
57
|
this.path = filePath;
|
|
@@ -72,7 +75,10 @@ export class EventLedger {
|
|
|
72
75
|
}
|
|
73
76
|
|
|
74
77
|
#load() {
|
|
75
|
-
const
|
|
78
|
+
const raw = readFileSync(this.path);
|
|
79
|
+
this.#size = raw.length;
|
|
80
|
+
const lines = raw
|
|
81
|
+
.toString("utf8")
|
|
76
82
|
.split("\n")
|
|
77
83
|
.filter((line) => line.length > 0);
|
|
78
84
|
for (const [index, line] of lines.entries()) {
|
|
@@ -142,8 +148,21 @@ export class EventLedger {
|
|
|
142
148
|
};
|
|
143
149
|
event.digest = eventDigest(event);
|
|
144
150
|
const nextState = this.#reducer.applyEvent(this.state, event);
|
|
151
|
+
// Guard against a stale writer: every event carries seq and prev from
|
|
152
|
+
// this instance's memory, so appending after another writer would fork
|
|
153
|
+
// the hash chain and leave the ledger corrupted for good. Refuse
|
|
154
|
+
// instead; re-reading and retrying is always safe. Writers re-read under
|
|
155
|
+
// the run lock, so this only fires on a writer that forgot to.
|
|
156
|
+
const onDisk = existsSync(this.path) ? statSync(this.path).size : 0;
|
|
157
|
+
if (onDisk !== this.#size) {
|
|
158
|
+
throw new LedgerError(
|
|
159
|
+
`ledger for ${runId} changed on disk since it was read (another writer); re-read it and retry`,
|
|
160
|
+
);
|
|
161
|
+
}
|
|
162
|
+
const line = `${JSON.stringify(event)}\n`;
|
|
145
163
|
mkdirSync(dirname(this.path), { recursive: true });
|
|
146
|
-
appendFileSync(this.path,
|
|
164
|
+
appendFileSync(this.path, line, "utf8");
|
|
165
|
+
this.#size += Buffer.byteLength(line, "utf8");
|
|
147
166
|
this.events.push(event);
|
|
148
167
|
this.state = nextState;
|
|
149
168
|
this.lastDigest = event.digest;
|