@haiyangbg/buildbeat 3.2.0 → 3.2.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -298,7 +298,7 @@ function budgetReasons(context, step, safeguard = false) {
298
298
  function workReviewBudget(context, step) {
299
299
  const stepDef = context.workflow.steps.find((item) => item.id === step);
300
300
  const cap = context.runBudgets.reviewRoundsPerWork;
301
- if (cap === undefined || !(step === "review" || stepDef?.worker === "reviewer")) return null;
301
+ if (cap === undefined || !isReviewStep(step, stepDef)) return null;
302
302
  const prior = computeWorkCost(context.repoRoot, context.ledger.state.run.work, {
303
303
  excludeRun: context.ledger.state.run.id,
304
304
  });
@@ -431,463 +431,518 @@ function settleOutcome(context, step, outcome, tree, exec) {
431
431
  return to;
432
432
  }
433
433
 
434
- function drive(context, startStep, { skipBoundaryOnce = false } = {}) {
435
- const { ledger, workflow, workspace, adapters, now } = context;
436
- let step = startStep;
437
- let firstStep = true;
438
- while (step) {
439
- if (workflow.terminal.has(step)) {
440
- context.waitHuman(
441
- `enter-${step}`,
442
- ["terminal step requires a human decision"],
443
- "final-decision",
444
- );
445
- return;
446
- }
447
- if (context.stopAt.includes(step) && !(skipBoundaryOnce && firstStep)) {
448
- context.waitHuman(`enter-${step}`, [`automation boundary: stopAt includes ${step}`]);
449
- return;
450
- }
451
- firstStep = false;
452
- const stepDef = workflow.steps.find((candidate) => candidate.id === step);
453
- const adapter = stepDef.worker ? adapters[stepDef.worker] : null;
454
- if (!adapter) {
455
- context.waitHuman(`enter-${step}`, [
456
- `no adapter configured for worker ${stepDef.worker ?? "(none)"}; attended handoff`,
457
- ]);
458
- return;
459
- }
434
+ // A step's review rounds are budgeted and capped per work: the step named
435
+ // "review" or any step run by the reviewer worker.
436
+ function isReviewStep(step, stepDef) {
437
+ return step === "review" || stepDef?.worker === "reviewer";
438
+ }
460
439
 
461
- // Work-level review cap (iteration 09): review rounds are counted across
462
- // every run of the work, superseded ones included, so "one run per
463
- // round" cannot slip past the per-run budget. Reaching the cap is a
464
- // human decision (review once more, or merge/close as-is), not a stop.
465
- const workBudget = workReviewBudget(context, step);
466
- if (workBudget?.exhausted) {
467
- const { failures } = budgetUsage(context, step);
468
- context.waitHuman(
469
- `enter-${step}`,
470
- [
471
- `${step} work review budget exhausted: ${workBudget.rounds}/${workBudget.allowed} review round(s) across the work, ${failures} real failure(s) in this run`,
472
- `approve enter-${step} = one more review round (also lifts this run's review cap if it is spent); reject = end this run and decide the merge on the evidence you have`,
473
- ],
474
- "work-review-cap",
475
- budgetGrants(context, step),
476
- );
477
- return;
478
- }
440
+ // Everything that may stop a run before a step starts: terminal step, stop
441
+ // boundary, missing adapter, the work-level review cap, pre policies and the
442
+ // step budgets. Returns what the step needs, or null when the run stopped.
443
+ function checkBeforeStep(context, step, { skipBoundary }) {
444
+ const { ledger, workflow, adapters, now } = context;
445
+ if (workflow.terminal.has(step)) {
446
+ context.waitHuman(
447
+ `enter-${step}`,
448
+ ["terminal step requires a human decision"],
449
+ "final-decision",
450
+ );
451
+ return null;
452
+ }
453
+ if (context.stopAt.includes(step) && !skipBoundary) {
454
+ context.waitHuman(`enter-${step}`, [`automation boundary: stopAt includes ${step}`]);
455
+ return null;
456
+ }
457
+ const stepDef = workflow.steps.find((candidate) => candidate.id === step);
458
+ const adapter = stepDef.worker ? adapters[stepDef.worker] : null;
459
+ if (!adapter) {
460
+ context.waitHuman(`enter-${step}`, [
461
+ `no adapter configured for worker ${stepDef.worker ?? "(none)"}; attended handoff`,
462
+ ]);
463
+ return null;
464
+ }
479
465
 
480
- const preGate = runPolicyGate(context, "pre", step);
481
- if (preGate.action === "block") {
482
- ledger.append({
483
- type: "RUN_TERMINAL",
484
- actor: KERNEL,
485
- ts: now(),
486
- data: { status: "FAILED", reason: `pre policy blocked ${step}` },
487
- });
488
- writeRunRecord({ repoRoot: context.repoRoot, ledger, ts: now() });
489
- return;
466
+ // Work-level review cap (iteration 09): review rounds are counted across
467
+ // every run of the work, superseded ones included, so "one run per
468
+ // round" cannot slip past the per-run budget. Reaching the cap is a
469
+ // human decision (review once more, or merge/close as-is), not a stop.
470
+ const workBudget = workReviewBudget(context, step);
471
+ if (workBudget?.exhausted) {
472
+ const { failures } = budgetUsage(context, step);
473
+ context.waitHuman(
474
+ `enter-${step}`,
475
+ [
476
+ `${step} work review budget exhausted: ${workBudget.rounds}/${workBudget.allowed} review round(s) across the work, ${failures} real failure(s) in this run`,
477
+ `approve enter-${step} = one more review round (also lifts this run's review cap if it is spent); reject = end this run and decide the merge on the evidence you have`,
478
+ ],
479
+ "work-review-cap",
480
+ budgetGrants(context, step),
481
+ );
482
+ return null;
483
+ }
484
+
485
+ const preGate = runPolicyGate(context, "pre", step);
486
+ if (preGate.action === "block") {
487
+ ledger.append({
488
+ type: "RUN_TERMINAL",
489
+ actor: KERNEL,
490
+ ts: now(),
491
+ data: { status: "FAILED", reason: `pre policy blocked ${step}` },
492
+ });
493
+ writeRunRecord({ repoRoot: context.repoRoot, ledger, ts: now() });
494
+ return null;
495
+ }
496
+ if (preGate.action === "wait") {
497
+ context.waitHuman(`resume-${step}`, policyReasons(preGate.rows));
498
+ return null;
499
+ }
500
+
501
+ const attempt = (ledger.state.steps[step]?.attempts ?? 0) + 1;
502
+ const maxAttempts = context.maxAttemptsFor(step);
503
+ if (attempt > maxAttempts) {
504
+ context.waitHuman(`resume-${step}`, budgetReasons(context, step), "budget", budgetGrants(context, step));
505
+ return null;
506
+ }
507
+
508
+ if (attempt > context.totalAttemptsFor(step)) {
509
+ context.waitHuman(`resume-${step}`, budgetReasons(context, step, true), "budget", budgetGrants(context, step));
510
+ return null;
511
+ }
512
+
513
+ return { stepDef, adapter, attempt };
514
+ }
515
+
516
+ // Records STEP_STARTED and prepares the worker's input: the read-only
517
+ // snapshot, the output path, anchored findings, the envelope prompt and the
518
+ // incremental-review base.
519
+ function beginStep(context, step, stepDef, attempt) {
520
+ const { ledger, workspace, now } = context;
521
+ const adapter = context.adapters[stepDef.worker];
522
+ ledger.append({
523
+ type: "STEP_STARTED",
524
+ actor: KERNEL,
525
+ ts: now(),
526
+ data: {
527
+ step,
528
+ attempt,
529
+ worker: stepDef.worker,
530
+ adapter: adapter.name,
531
+ workspaceId: workspace.workspaceId,
532
+ },
533
+ });
534
+ const before = stepDef.readonly ? readback(workspace.worktreePath) : null;
535
+ const outputsDir = join(context.runtimeDir, "runs", ledger.state.run.id, "outputs");
536
+ mkdirSync(outputsDir, { recursive: true });
537
+ const outputPath = join(outputsDir, `${step}-${attempt}.json`);
538
+ // Anchored review: readonly (reviewer) steps receive the adjudicated
539
+ // findings history so a fresh reviewer inherits settled verdicts instead
540
+ // of re-litigating them; writing steps get the latest review findings
541
+ // with their adjudication status (the fixer's worklist).
542
+ const input = { workId: ledger.state.run.work, runId: ledger.state.run.id, step, attempt };
543
+ const anchor = buildAnchor(context.repoRoot, ledger.state.run.work);
544
+ if (anchor && stepDef.readonly) {
545
+ input.anchor = anchor;
546
+ } else if (anchor) {
547
+ const lastReview = [...ledger.state.evidence]
548
+ .reverse()
549
+ .find((item) => item.kind === "review");
550
+ if (lastReview?.findings?.length) {
551
+ const adjudicated = latestAdjudications(
552
+ readFindingsAccount(context.repoRoot, ledger.state.run.work),
553
+ );
554
+ input.findings = lastReview.findings.map((finding) => ({
555
+ severity: finding.severity,
556
+ summary: finding.summary,
557
+ fingerprint: fingerprintFinding(finding),
558
+ adjudication: adjudicated.get(fingerprintFinding(finding))?.action ?? "open",
559
+ }));
490
560
  }
491
- if (preGate.action === "wait") {
492
- context.waitHuman(`resume-${step}`, policyReasons(preGate.rows));
493
- return;
561
+ }
562
+ // Envelope (C6): the worker's prompt, materialised into the run
563
+ // directory and handed over as BUILDBEAT_PROMPT / input.envelope.
564
+ const prompt = materialisePrompt({
565
+ envelope: context.envelope,
566
+ worker: stepDef.worker,
567
+ runtimeDir: context.runtimeDir,
568
+ runId: ledger.state.run.id,
569
+ step,
570
+ attempt,
571
+ repoRoot: context.repoRoot,
572
+ });
573
+ if (prompt) {
574
+ input.envelope = { promptRef: prompt.ref, file: prompt.file, digest: context.envelope.digest, vars: context.envelope.vars };
575
+ }
576
+ // Incremental review (C7): tell a reviewer which candidate the last
577
+ // review saw when it is an ancestor of this one.
578
+ if (stepDef.readonly) {
579
+ const head = before.head;
580
+ const lastReviewed = lastReviewedCandidate(context.repoRoot, ledger.state.run.work, workspace.worktreePath, head);
581
+ if (lastReviewed) {
582
+ input.lastReviewed = lastReviewed;
494
583
  }
584
+ }
585
+ return { before, outputPath, input, prompt };
586
+ }
495
587
 
496
- const attempt = (ledger.state.steps[step]?.attempts ?? 0) + 1;
497
- const maxAttempts = context.maxAttemptsFor(step);
498
- if (attempt > maxAttempts) {
499
- context.waitHuman(`resume-${step}`, budgetReasons(context, step), "budget", budgetGrants(context, step));
500
- return;
588
+ // Runs the worker, or references identical passed evidence (verification
589
+ // reuse), then records the command evidence for the tree git reads back.
590
+ function executeOrReuse(context, step, stepDef, adapter, attempt, { before, outputPath, input, prompt }) {
591
+ const { ledger, workspace, now } = context;
592
+ // Verification reuse (C7): same tree + same worker + same envelope that
593
+ // already passed is referenced, not re-run. Failures always re-run.
594
+ let stepCacheKey = null;
595
+ let reused = null;
596
+ if (context.cache[step] === "tree") {
597
+ const current = readback(workspace.worktreePath);
598
+ if (!current.dirty) {
599
+ stepCacheKey = cacheKey({
600
+ tree: treeHash(workspace.worktreePath),
601
+ worker: stepDef.worker,
602
+ adapterSpec: context.adapterConfigs[stepDef.worker] ?? null,
603
+ adapterName: adapter.name,
604
+ envelopeDigest: context.envelope?.digest ?? null,
605
+ });
606
+ reused = findReusableEvidence(context.repoRoot, stepCacheKey);
501
607
  }
608
+ }
609
+ let exec;
610
+ if (reused) {
611
+ const at = now();
612
+ exec = {
613
+ adapter: "cache",
614
+ command: `reuse ${reused.run} ${reused.evidenceRef}`,
615
+ exitCode: 0,
616
+ signal: null,
617
+ stdout: `REUSED: identical tree/worker/envelope already passed in ${reused.run} (${reused.evidenceRef}, ${reused.digest}); not re-run`,
618
+ stderr: "",
619
+ timedOut: false,
620
+ spawnError: null,
621
+ startedAt: at,
622
+ finishedAt: at,
623
+ };
624
+ } else {
625
+ exec = adapter.execute({
626
+ step,
627
+ worker: stepDef.worker,
628
+ workspacePath: workspace.worktreePath,
629
+ input,
630
+ timeoutMs: context.stepTimeoutMs,
631
+ outputPath,
632
+ // Live output streams + marker land in the run directory so `status`
633
+ // can answer "is it still doing something" while the step runs.
634
+ liveDir: join(context.runtimeDir, "runs", ledger.state.run.id),
635
+ promptPath: prompt?.path ?? null,
636
+ vars: context.envelope?.vars ?? null,
637
+ });
638
+ }
639
+ const tree = readback(workspace.worktreePath);
640
+ const evidence = collectCommandEvidence({
641
+ runtimeDir: context.runtimeDir,
642
+ runId: ledger.state.run.id,
643
+ step,
644
+ attempt,
645
+ execResult: exec,
646
+ subject: tree.head,
647
+ grade: reused ? reused.grade : stepDef.grade ?? "L2",
648
+ redact: context.redact,
649
+ });
650
+ ledger.append({
651
+ type: "EVIDENCE_RECORDED",
652
+ actor: KERNEL,
653
+ ts: now(),
654
+ data: {
655
+ evidenceRef: toRepoRef(context.repoRoot, evidence.location),
656
+ kind: evidence.kind,
657
+ subject: evidence.subject,
658
+ digest: evidence.digest,
659
+ status: evidence.status,
660
+ grade: evidence.grade,
661
+ ...(stepCacheKey ? { cacheKey: stepCacheKey } : {}),
662
+ ...(reused ? { reused: { run: reused.run, evidenceRef: reused.evidenceRef, digest: reused.digest } } : {}),
663
+ },
664
+ });
502
665
 
503
- if (attempt > context.totalAttemptsFor(step)) {
504
- context.waitHuman(`resume-${step}`, budgetReasons(context, step, true), "budget", budgetGrants(context, step));
505
- return;
506
- }
666
+ return { exec, tree };
667
+ }
507
668
 
669
+ // Classifies what the step did and records it: read-only enforcement, the
670
+ // worker envelope, status and infrastructure failures, review findings,
671
+ // scope, the pinned candidate and post policies. Returns the result to
672
+ // route, or null when the run stopped.
673
+ function recordStepResult(context, step, stepDef, attempt, { before, outputPath }, { exec, tree }) {
674
+ const { ledger, workspace, now } = context;
675
+ // Read-only enforcement: a reviewer that changed the workspace is a
676
+ // policy violation, not a candidate (invariants 9/17).
677
+ if (stepDef.readonly && (tree.head !== before.head || tree.dirty !== before.dirty)) {
508
678
  ledger.append({
509
- type: "STEP_STARTED",
679
+ type: "POLICY_EVALUATED",
510
680
  actor: KERNEL,
511
681
  ts: now(),
512
682
  data: {
513
- step,
514
- attempt,
515
- worker: stepDef.worker,
516
- adapter: adapter.name,
517
- workspaceId: workspace.workspaceId,
683
+ policy: "step.readonly",
684
+ phase: "action",
685
+ result: "BLOCK",
686
+ enforcement: "LOCAL_ENFORCED",
687
+ reason: `read-only step ${step} modified the workspace`,
518
688
  },
519
689
  });
520
- const before = stepDef.readonly ? readback(workspace.worktreePath) : null;
521
- const outputsDir = join(context.runtimeDir, "runs", ledger.state.run.id, "outputs");
522
- mkdirSync(outputsDir, { recursive: true });
523
- const outputPath = join(outputsDir, `${step}-${attempt}.json`);
524
- // Anchored review: readonly (reviewer) steps receive the adjudicated
525
- // findings history so a fresh reviewer inherits settled verdicts instead
526
- // of re-litigating them; writing steps get the latest review findings
527
- // with their adjudication status (the fixer's worklist).
528
- const input = { workId: ledger.state.run.work, runId: ledger.state.run.id, step, attempt };
529
- const anchor = buildAnchor(context.repoRoot, ledger.state.run.work);
530
- if (anchor && stepDef.readonly) {
531
- input.anchor = anchor;
532
- } else if (anchor) {
533
- const lastReview = [...ledger.state.evidence]
534
- .reverse()
535
- .find((item) => item.kind === "review");
536
- if (lastReview?.findings?.length) {
537
- const adjudicated = latestAdjudications(
538
- readFindingsAccount(context.repoRoot, ledger.state.run.work),
539
- );
540
- input.findings = lastReview.findings.map((finding) => ({
541
- severity: finding.severity,
542
- summary: finding.summary,
543
- fingerprint: fingerprintFinding(finding),
544
- adjudication: adjudicated.get(fingerprintFinding(finding))?.action ?? "open",
545
- }));
546
- }
547
- }
548
- // Envelope (C6): the worker's prompt, materialised into the run
549
- // directory and handed over as BUILDBEAT_PROMPT / input.envelope.
550
- const prompt = materialisePrompt({
551
- envelope: context.envelope,
552
- worker: stepDef.worker,
553
- runtimeDir: context.runtimeDir,
554
- runId: ledger.state.run.id,
690
+ ledger.append({
691
+ type: "STEP_FINISHED",
692
+ actor: KERNEL,
693
+ ts: now(),
694
+ data: { step, attempt, status: "blocked" },
695
+ });
696
+ context.waitHuman(`resume-${step}`, [
697
+ `read-only step ${step} modified the workspace; human triage required`,
698
+ ]);
699
+ return null;
700
+ }
701
+
702
+ let envelopeRaw = exec.envelope;
703
+ if (exec.envelope !== undefined && exec.envelope !== null) {
704
+ writeFileSync(outputPath, `${JSON.stringify(exec.envelope, null, 2)}\n`, "utf8");
705
+ } else if (existsSync(outputPath)) {
706
+ envelopeRaw = readFileSync(outputPath, "utf8");
707
+ }
708
+ const { envelope, error: envelopeError } = parseEnvelope(envelopeRaw);
709
+
710
+ let stepStatus;
711
+ if (exec.spawnError) {
712
+ stepStatus = "crashed";
713
+ } else if (exec.timedOut) {
714
+ stepStatus = "timeout";
715
+ } else if (exec.signal) {
716
+ stepStatus = "crashed";
717
+ } else if (exec.exitCode !== 0) {
718
+ stepStatus = "failed";
719
+ } else if (envelopeError) {
720
+ stepStatus = "invalid-output";
721
+ } else {
722
+ stepStatus = "succeeded";
723
+ }
724
+ // Infrastructure failure vs candidate failure. A timeout, a crash,
725
+ // garbage output or the worker's own "environment unavailable" signal
726
+ // (exit 75, EX_TEMPFAIL) says nothing about the candidate: no failure
727
+ // fingerprint, no fixer, the attempt is refunded, and a human decides
728
+ // when the backend is back. Real incidents: a worker backend outage
729
+ // (review exit 97) and non-JSON reviewer output killed five runs in two
730
+ // days as "no transition for (review, failed)"; PATH, port and host-load
731
+ // verify failures dispatched fixers five times.
732
+ const infra =
733
+ stepStatus === "timeout" ||
734
+ stepStatus === "crashed" ||
735
+ stepStatus === "invalid-output" ||
736
+ (stepStatus === "failed" && exec.exitCode === 75);
737
+ const free = stepStatus === "succeeded" && stepDef.readonly !== true;
738
+ ledger.append({
739
+ type: "STEP_FINISHED",
740
+ actor: KERNEL,
741
+ ts: now(),
742
+ data: { step, attempt, status: stepStatus, exitCode: exec.exitCode,
743
+ ...(infra ? { infra: true } : {}), ...(free ? { free: true } : {}) },
744
+ });
745
+ ledger.append({
746
+ type: "BUDGET_CONSUMED",
747
+ actor: KERNEL,
748
+ ts: now(),
749
+ data: {
750
+ kind: "attempts",
751
+ amount: infra || free ? 0 : 1,
752
+ remaining: context.maxAttemptsFor(step) - attempt,
753
+ },
754
+ });
755
+ if (infra) {
756
+ const cause =
757
+ stepStatus === "failed"
758
+ ? "exit 75 (worker reports its environment unavailable)"
759
+ : stepStatus === "invalid-output"
760
+ ? "output is not a worker envelope"
761
+ : stepStatus;
762
+ context.waitHuman(
763
+ `resume-${step}`,
764
+ [
765
+ `worker infrastructure failure at ${step}: ${cause}; not a candidate defect, attempt not charged`,
766
+ ...(tree.dirty ? [`the failed worker left the worktree dirty; inspect before rerunning`] : []),
767
+ `approve resume-${step} to rerun once the backend/environment is back; reject to end the run`,
768
+ ],
769
+ "infra",
770
+ );
771
+ return null;
772
+ }
773
+
774
+ let blockingFindings = [];
775
+ if (envelope?.findings) {
776
+ recordReviewFindings(context.repoRoot, ledger.state.run.work, {
777
+ run: ledger.state.run.id,
555
778
  step,
556
779
  attempt,
557
- repoRoot: context.repoRoot,
780
+ findings: envelope.findings,
781
+ ts: now(),
558
782
  });
559
- if (prompt) {
560
- input.envelope = { promptRef: prompt.ref, file: prompt.file, digest: context.envelope.digest, vars: context.envelope.vars };
561
- }
562
- // Incremental review (C7): tell a reviewer which candidate the last
563
- // review saw when it is an ancestor of this one.
564
- if (stepDef.readonly) {
565
- const head = before.head;
566
- const lastReviewed = lastReviewedCandidate(context.repoRoot, ledger.state.run.work, workspace.worktreePath, head);
567
- if (lastReviewed) {
568
- input.lastReviewed = lastReviewed;
569
- }
570
- }
571
- // Verification reuse (C7): same tree + same worker + same envelope that
572
- // already passed is referenced, not re-run. Failures always re-run.
573
- let stepCacheKey = null;
574
- let reused = null;
575
- if (context.cache[step] === "tree") {
576
- const current = readback(workspace.worktreePath);
577
- if (!current.dirty) {
578
- stepCacheKey = cacheKey({
579
- tree: treeHash(workspace.worktreePath),
580
- worker: stepDef.worker,
581
- adapterSpec: context.adapterConfigs[stepDef.worker] ?? null,
582
- adapterName: adapter.name,
583
- envelopeDigest: context.envelope?.digest ?? null,
584
- });
585
- reused = findReusableEvidence(context.repoRoot, stepCacheKey);
783
+ // A fingerprint a human dismissed stays visible in the evidence but no
784
+ // longer blocks: settled verdicts do not reopen without a human.
785
+ const adjudicated = latestAdjudications(
786
+ readFindingsAccount(context.repoRoot, ledger.state.run.work),
787
+ );
788
+ const suppressed = [];
789
+ blockingFindings = envelope.findings.filter((finding) => {
790
+ if (adjudicated.get(fingerprintFinding(finding))?.action === "dismiss") {
791
+ suppressed.push(fingerprintFinding(finding));
792
+ return false;
586
793
  }
587
- }
588
- let exec;
589
- if (reused) {
590
- const at = now();
591
- exec = {
592
- adapter: "cache",
593
- command: `reuse ${reused.run} ${reused.evidenceRef}`,
594
- exitCode: 0,
595
- signal: null,
596
- stdout: `REUSED: identical tree/worker/envelope already passed in ${reused.run} (${reused.evidenceRef}, ${reused.digest}); not re-run`,
597
- stderr: "",
598
- timedOut: false,
599
- spawnError: null,
600
- startedAt: at,
601
- finishedAt: at,
602
- };
603
- } else {
604
- exec = adapter.execute({
605
- step,
606
- worker: stepDef.worker,
607
- workspacePath: workspace.worktreePath,
608
- input,
609
- timeoutMs: context.stepTimeoutMs,
610
- outputPath,
611
- // Live output streams + marker land in the run directory so `status`
612
- // can answer "is it still doing something" while the step runs.
613
- liveDir: join(context.runtimeDir, "runs", ledger.state.run.id),
614
- promptPath: prompt?.path ?? null,
615
- vars: context.envelope?.vars ?? null,
616
- });
617
- }
618
- const tree = readback(workspace.worktreePath);
619
- const evidence = collectCommandEvidence({
620
- runtimeDir: context.runtimeDir,
621
- runId: ledger.state.run.id,
622
- step,
623
- attempt,
624
- execResult: exec,
625
- subject: tree.head,
626
- grade: reused ? reused.grade : stepDef.grade ?? "L2",
627
- redact: context.redact,
794
+ return finding.severity === "P0" || finding.severity === "P1";
628
795
  });
629
796
  ledger.append({
630
797
  type: "EVIDENCE_RECORDED",
631
798
  actor: KERNEL,
632
799
  ts: now(),
633
800
  data: {
634
- evidenceRef: toRepoRef(context.repoRoot, evidence.location),
635
- kind: evidence.kind,
636
- subject: evidence.subject,
637
- digest: evidence.digest,
638
- status: evidence.status,
639
- grade: evidence.grade,
640
- ...(stepCacheKey ? { cacheKey: stepCacheKey } : {}),
641
- ...(reused ? { reused: { run: reused.run, evidenceRef: reused.evidenceRef, digest: reused.digest } } : {}),
801
+ evidenceRef: toRepoRef(context.repoRoot, outputPath),
802
+ kind: "review",
803
+ subject: tree.head,
804
+ digest: sha256(canonicalJson(envelope)),
805
+ status: blockingFindings.length > 0 ? "failed" : "passed",
806
+ grade: "L2",
807
+ findings: envelope.findings,
808
+ ...(suppressed.length > 0 ? { suppressedFingerprints: suppressed } : {}),
642
809
  },
643
810
  });
811
+ }
644
812
 
645
- // Read-only enforcement: a reviewer that changed the workspace is a
646
- // policy violation, not a candidate (invariants 9/17).
647
- if (stepDef.readonly && (tree.head !== before.head || tree.dirty !== before.dirty)) {
813
+ // Scope enforcement (B §10: out-of-scope changes stop the loop): any
814
+ // path changed outside the allowed set means this candidate cannot
815
+ // proceed, whatever the exit code said.
816
+ if (!stepDef.readonly && context.allowedPaths) {
817
+ const changed = listChangedPaths(workspace.worktreePath, workspace.base);
818
+ const violations = changed.filter(
819
+ (path) =>
820
+ !context.allowedPaths.some(
821
+ (prefix) =>
822
+ path === prefix || path.startsWith(prefix.endsWith("/") ? prefix : `${prefix}/`),
823
+ ),
824
+ );
825
+ if (violations.length > 0) {
648
826
  ledger.append({
649
827
  type: "POLICY_EVALUATED",
650
828
  actor: KERNEL,
651
829
  ts: now(),
652
830
  data: {
653
- policy: "step.readonly",
831
+ policy: "workspace.scope",
654
832
  phase: "action",
655
833
  result: "BLOCK",
656
834
  enforcement: "LOCAL_ENFORCED",
657
- reason: `read-only step ${step} modified the workspace`,
835
+ reason: `out-of-scope changes: ${violations.slice(0, 5).join(", ")}`,
658
836
  },
659
837
  });
660
- ledger.append({
661
- type: "STEP_FINISHED",
662
- actor: KERNEL,
663
- ts: now(),
664
- data: { step, attempt, status: "blocked" },
665
- });
666
838
  context.waitHuman(`resume-${step}`, [
667
- `read-only step ${step} modified the workspace; human triage required`,
839
+ `worker changed paths outside the allowed scope: ${violations.slice(0, 5).join(", ")}`,
668
840
  ]);
669
- return;
670
- }
671
-
672
- let envelopeRaw = exec.envelope;
673
- if (exec.envelope !== undefined && exec.envelope !== null) {
674
- writeFileSync(outputPath, `${JSON.stringify(exec.envelope, null, 2)}\n`, "utf8");
675
- } else if (existsSync(outputPath)) {
676
- envelopeRaw = readFileSync(outputPath, "utf8");
841
+ return null;
677
842
  }
678
- const { envelope, error: envelopeError } = parseEnvelope(envelopeRaw);
843
+ }
679
844
 
680
- let stepStatus;
681
- if (exec.spawnError) {
682
- stepStatus = "crashed";
683
- } else if (exec.timedOut) {
684
- stepStatus = "timeout";
685
- } else if (exec.signal) {
686
- stepStatus = "crashed";
687
- } else if (exec.exitCode !== 0) {
688
- stepStatus = "failed";
689
- } else if (envelopeError) {
690
- stepStatus = "invalid-output";
691
- } else {
692
- stepStatus = "succeeded";
693
- }
694
- // Infrastructure failure vs candidate failure. A timeout, a crash,
695
- // garbage output or the worker's own "environment unavailable" signal
696
- // (exit 75, EX_TEMPFAIL) says nothing about the candidate: no failure
697
- // fingerprint, no fixer, the attempt is refunded, and a human decides
698
- // when the backend is back. Real incidents: a worker backend outage
699
- // (review exit 97) and non-JSON reviewer output killed five runs in two
700
- // days as "no transition for (review, failed)"; PATH, port and host-load
701
- // verify failures dispatched fixers five times.
702
- const infra =
703
- stepStatus === "timeout" ||
704
- stepStatus === "crashed" ||
705
- stepStatus === "invalid-output" ||
706
- (stepStatus === "failed" && exec.exitCode === 75);
707
- const free = stepStatus === "succeeded" && stepDef.readonly !== true;
708
- ledger.append({
709
- type: "STEP_FINISHED",
710
- actor: KERNEL,
711
- ts: now(),
712
- data: { step, attempt, status: stepStatus, exitCode: exec.exitCode,
713
- ...(infra ? { infra: true } : {}), ...(free ? { free: true } : {}) },
714
- });
715
- ledger.append({
716
- type: "BUDGET_CONSUMED",
717
- actor: KERNEL,
718
- ts: now(),
719
- data: {
720
- kind: "attempts",
721
- amount: infra || free ? 0 : 1,
722
- remaining: context.maxAttemptsFor(step) - attempt,
723
- },
724
- });
725
- if (infra) {
726
- const cause =
727
- stepStatus === "failed"
728
- ? "exit 75 (worker reports its environment unavailable)"
729
- : stepStatus === "invalid-output"
730
- ? "output is not a worker envelope"
731
- : stepStatus;
732
- context.waitHuman(
733
- `resume-${step}`,
734
- [
735
- `worker infrastructure failure at ${step}: ${cause}; not a candidate defect, attempt not charged`,
736
- ...(tree.dirty ? [`the failed worker left the worktree dirty; inspect before rerunning`] : []),
737
- `approve resume-${step} to rerun once the backend/environment is back; reject to end the run`,
738
- ],
739
- "infra",
740
- );
741
- return;
845
+ if (stepStatus === "succeeded" && !stepDef.readonly) {
846
+ if (tree.dirty) {
847
+ context.waitHuman(`resume-${step}`, [
848
+ `step ${step} left a dirty worktree; a candidate must be a committed state`,
849
+ ]);
850
+ return null;
742
851
  }
743
-
744
- let blockingFindings = [];
745
- if (envelope?.findings) {
746
- recordReviewFindings(context.repoRoot, ledger.state.run.work, {
747
- run: ledger.state.run.id,
748
- step,
749
- attempt,
750
- findings: envelope.findings,
751
- ts: now(),
752
- });
753
- // A fingerprint a human dismissed stays visible in the evidence but no
754
- // longer blocks: settled verdicts do not reopen without a human.
755
- const adjudicated = latestAdjudications(
756
- readFindingsAccount(context.repoRoot, ledger.state.run.work),
757
- );
758
- const suppressed = [];
759
- blockingFindings = envelope.findings.filter((finding) => {
760
- if (adjudicated.get(fingerprintFinding(finding))?.action === "dismiss") {
761
- suppressed.push(fingerprintFinding(finding));
762
- return false;
763
- }
764
- return finding.severity === "P0" || finding.severity === "P1";
765
- });
852
+ const pinned = ledger.state.workspaces[workspace.workspaceId]?.candidate;
853
+ if (tree.head !== (pinned ?? workspace.base)) {
766
854
  ledger.append({
767
- type: "EVIDENCE_RECORDED",
855
+ type: "CANDIDATE_PINNED",
768
856
  actor: KERNEL,
769
857
  ts: now(),
770
858
  data: {
771
- evidenceRef: toRepoRef(context.repoRoot, outputPath),
772
- kind: "review",
773
- subject: tree.head,
774
- digest: sha256(canonicalJson(envelope)),
775
- status: blockingFindings.length > 0 ? "failed" : "passed",
776
- grade: "L2",
777
- findings: envelope.findings,
778
- ...(suppressed.length > 0 ? { suppressedFingerprints: suppressed } : {}),
859
+ workspaceId: workspace.workspaceId,
860
+ base: workspace.base,
861
+ candidate: tree.head,
779
862
  },
780
863
  });
781
864
  }
865
+ }
782
866
 
783
- // Scope enforcement (B §10: out-of-scope changes stop the loop): any
784
- // path changed outside the allowed set means this candidate cannot
785
- // proceed, whatever the exit code said.
786
- if (!stepDef.readonly && context.allowedPaths) {
787
- const changed = listChangedPaths(workspace.worktreePath, workspace.base);
788
- const violations = changed.filter(
789
- (path) =>
790
- !context.allowedPaths.some(
791
- (prefix) =>
792
- path === prefix || path.startsWith(prefix.endsWith("/") ? prefix : `${prefix}/`),
793
- ),
794
- );
795
- if (violations.length > 0) {
796
- ledger.append({
797
- type: "POLICY_EVALUATED",
798
- actor: KERNEL,
799
- ts: now(),
800
- data: {
801
- policy: "workspace.scope",
802
- phase: "action",
803
- result: "BLOCK",
804
- enforcement: "LOCAL_ENFORCED",
805
- reason: `out-of-scope changes: ${violations.slice(0, 5).join(", ")}`,
806
- },
807
- });
808
- context.waitHuman(`resume-${step}`, [
809
- `worker changed paths outside the allowed scope: ${violations.slice(0, 5).join(", ")}`,
810
- ]);
811
- return;
812
- }
867
+ if (stepStatus === "succeeded") {
868
+ const postGate = runPolicyGate(context, "post", step);
869
+ if (postGate.action === "block") {
870
+ ledger.append({
871
+ type: "RUN_TERMINAL",
872
+ actor: KERNEL,
873
+ ts: now(),
874
+ data: { status: "FAILED", reason: `post policy blocked ${step}` },
875
+ });
876
+ writeRunRecord({ repoRoot: context.repoRoot, ledger, ts: now() });
877
+ return null;
813
878
  }
814
-
815
- if (stepStatus === "succeeded" && !stepDef.readonly) {
816
- if (tree.dirty) {
817
- context.waitHuman(`resume-${step}`, [
818
- `step ${step} left a dirty worktree; a candidate must be a committed state`,
819
- ]);
820
- return;
821
- }
822
- const pinned = ledger.state.workspaces[workspace.workspaceId]?.candidate;
823
- if (tree.head !== (pinned ?? workspace.base)) {
824
- ledger.append({
825
- type: "CANDIDATE_PINNED",
826
- actor: KERNEL,
827
- ts: now(),
828
- data: {
829
- workspaceId: workspace.workspaceId,
830
- base: workspace.base,
831
- candidate: tree.head,
832
- },
833
- });
834
- }
879
+ if (postGate.action === "wait") {
880
+ context.waitHuman(`resume-${step}`, policyReasons(postGate.rows));
881
+ return null;
835
882
  }
883
+ }
836
884
 
837
- if (stepStatus === "succeeded") {
838
- const postGate = runPolicyGate(context, "post", step);
839
- if (postGate.action === "block") {
840
- ledger.append({
841
- type: "RUN_TERMINAL",
842
- actor: KERNEL,
843
- ts: now(),
844
- data: { status: "FAILED", reason: `post policy blocked ${step}` },
845
- });
846
- writeRunRecord({ repoRoot: context.repoRoot, ledger, ts: now() });
847
- return;
848
- }
849
- if (postGate.action === "wait") {
850
- context.waitHuman(`resume-${step}`, policyReasons(postGate.rows));
851
- return;
852
- }
885
+ return { stepStatus, blockingFindings, tree, exec };
886
+ }
887
+
888
+ // Settles the outcome and picks the next step; blocking findings stop once
889
+ // for triage or for the review budget before any fixer runs. Returns the
890
+ // next step, or null when the run stopped.
891
+ function routeAfterStep(context, step, stepDef, { stepStatus, blockingFindings, tree, exec }) {
892
+ let outcome;
893
+ if (stepStatus !== "succeeded") {
894
+ outcome = "failed";
895
+ } else if (blockingFindings.length > 0) {
896
+ outcome = "findings-blocking";
897
+ } else {
898
+ outcome = "succeeded";
899
+ }
900
+ const routed = settleOutcome(context, step, outcome, tree, exec);
901
+ // Ask before spending fix/verify workers: one approval covers the next
902
+ // round and both review caps, with the grant bound to this request.
903
+ if (routed && outcome === "findings-blocking") {
904
+ const grants = isReviewStep(step, stepDef) ? budgetGrants(context, step) : [];
905
+ const triage = context.reviewTriage === "required";
906
+ if (triage || grants.length) {
907
+ const { used, limit, failures } = budgetUsage(context, step);
908
+ const workBudget = workReviewBudget(context, step);
909
+ context.waitHuman(
910
+ `enter-${routed}`,
911
+ [
912
+ ...(grants.length ? [
913
+ `${step} budget exhausted: ${used}/${limit} review round(s) used in this run${workBudget ? `, ${workBudget.rounds}/${workBudget.allowed} across the work` : ""}, ${failures} real failure(s); approve enter-${routed} = fix + re-verify + one more review round; reject = end this run and decide the merge on the evidence you have`,
914
+ ] : []),
915
+ `review found ${blockingFindings.length} blocking finding(s); ${triage ? "triage" : "approve another round"} before ${routed} runs`,
916
+ ...blockingFindings.slice(0, 5).map((finding) =>
917
+ `[${finding.severity} ${fingerprintFinding(finding)}] ${finding.summary.slice(0, 200)}`),
918
+ `adjudicate fingerprints (findings adjudicate), then approve enter-${routed} or reject the run`,
919
+ ],
920
+ triage ? "finding-triage" : "budget",
921
+ grants,
922
+ );
923
+ return null;
853
924
  }
925
+ }
926
+ return routed;
927
+ }
854
928
 
855
- let outcome;
856
- if (stepStatus !== "succeeded") {
857
- outcome = "failed";
858
- } else if (blockingFindings.length > 0) {
859
- outcome = "findings-blocking";
860
- } else {
861
- outcome = "succeeded";
929
+ function drive(context, startStep, { skipBoundaryOnce = false } = {}) {
930
+ let step = startStep;
931
+ let firstStep = true;
932
+ while (step) {
933
+ const entry = checkBeforeStep(context, step, { skipBoundary: skipBoundaryOnce && firstStep });
934
+ firstStep = false;
935
+ if (!entry) {
936
+ return;
862
937
  }
863
- const routed = settleOutcome(context, step, outcome, tree, exec);
864
- // Ask before spending fix/verify workers: one approval covers the next
865
- // round and both review caps, with the grant bound to this request.
866
- if (routed && outcome === "findings-blocking") {
867
- const isReview = step === "review" || stepDef.worker === "reviewer";
868
- const grants = isReview ? budgetGrants(context, step) : [];
869
- const triage = context.reviewTriage === "required";
870
- if (triage || grants.length) {
871
- const { used, limit, failures } = budgetUsage(context, step);
872
- const workBudget = workReviewBudget(context, step);
873
- context.waitHuman(
874
- `enter-${routed}`,
875
- [
876
- ...(grants.length ? [
877
- `${step} budget exhausted: ${used}/${limit} review round(s) used in this run${workBudget ? `, ${workBudget.rounds}/${workBudget.allowed} across the work` : ""}, ${failures} real failure(s); approve enter-${routed} = fix + re-verify + one more review round; reject = end this run and decide the merge on the evidence you have`,
878
- ] : []),
879
- `review found ${blockingFindings.length} blocking finding(s); ${triage ? "triage" : "approve another round"} before ${routed} runs`,
880
- ...blockingFindings.slice(0, 5).map((finding) =>
881
- `[${finding.severity} ${fingerprintFinding(finding)}] ${finding.summary.slice(0, 200)}`),
882
- `adjudicate fingerprints (findings adjudicate), then approve enter-${routed} or reject the run`,
883
- ],
884
- triage ? "finding-triage" : "budget",
885
- grants,
886
- );
887
- return;
888
- }
938
+ const { stepDef, adapter, attempt } = entry;
939
+ const started = beginStep(context, step, stepDef, attempt);
940
+ const ran = executeOrReuse(context, step, stepDef, adapter, attempt, started);
941
+ const result = recordStepResult(context, step, stepDef, attempt, started, ran);
942
+ if (!result) {
943
+ return;
889
944
  }
890
- step = routed;
945
+ step = routeAfterStep(context, step, stepDef, result);
891
946
  }
892
947
  }
893
948