harnery 0.28.0 → 0.30.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (59) hide show
  1. package/dist/commands/work.d.ts +2 -1
  2. package/dist/commands/work.d.ts.map +1 -1
  3. package/dist/commands/work.js +44 -4
  4. package/dist/core/harnesses/types.d.ts +5 -0
  5. package/dist/core/harnesses/types.d.ts.map +1 -1
  6. package/dist/core/supervisor/plan-read.js +7 -1
  7. package/dist/core/supervisor/plan-types.d.ts +10 -0
  8. package/dist/core/supervisor/plan-types.d.ts.map +1 -1
  9. package/dist/core/supervisor/planning.d.ts.map +1 -1
  10. package/dist/core/supervisor/planning.js +18 -0
  11. package/dist/core/supervisor/state.d.ts.map +1 -1
  12. package/dist/core/supervisor/state.js +65 -11
  13. package/dist/core/work/runner.d.ts.map +1 -1
  14. package/dist/core/work/runner.js +11 -2
  15. package/dist/core/work/state.d.ts +17 -0
  16. package/dist/core/work/state.d.ts.map +1 -1
  17. package/dist/core/work/state.js +102 -3
  18. package/dist/core/workflow/engine.js +13 -0
  19. package/dist/core/workflow/index.d.ts +3 -2
  20. package/dist/core/workflow/index.d.ts.map +1 -1
  21. package/dist/core/workflow/index.js +2 -1
  22. package/dist/core/workflow/proof.d.ts +18 -1
  23. package/dist/core/workflow/proof.d.ts.map +1 -1
  24. package/dist/core/workflow/proof.js +34 -0
  25. package/dist/core/workflow/spawn-claude.d.ts.map +1 -1
  26. package/dist/core/workflow/spawn-claude.js +19 -2
  27. package/dist/core/workflow/spawn-codex.d.ts.map +1 -1
  28. package/dist/core/workflow/spawn-codex.js +17 -2
  29. package/dist/core/workflow/spawn-cursor.d.ts.map +1 -1
  30. package/dist/core/workflow/spawn-cursor.js +18 -2
  31. package/dist/core/workflow/spawn-failure.d.ts +14 -0
  32. package/dist/core/workflow/spawn-failure.d.ts.map +1 -1
  33. package/dist/core/workflow/spawn-failure.js +34 -0
  34. package/dist/core/workflow/types.d.ts +26 -0
  35. package/dist/core/workflow/types.d.ts.map +1 -1
  36. package/dist/core/workflow/workspaces/local-git.d.ts.map +1 -1
  37. package/dist/core/workflow/workspaces/local-git.js +101 -20
  38. package/dist/lib/exec.d.ts +7 -0
  39. package/dist/lib/exec.d.ts.map +1 -1
  40. package/dist/lib/exec.js +6 -3
  41. package/package.json +1 -1
  42. package/src/commands/work.ts +51 -4
  43. package/src/core/harnesses/types.ts +5 -0
  44. package/src/core/supervisor/plan-read.ts +11 -2
  45. package/src/core/supervisor/plan-types.ts +11 -0
  46. package/src/core/supervisor/planning.ts +21 -0
  47. package/src/core/supervisor/state.ts +68 -11
  48. package/src/core/work/runner.ts +11 -2
  49. package/src/core/work/state.ts +125 -3
  50. package/src/core/workflow/engine.ts +11 -0
  51. package/src/core/workflow/index.ts +3 -0
  52. package/src/core/workflow/proof.ts +36 -0
  53. package/src/core/workflow/spawn-claude.ts +21 -2
  54. package/src/core/workflow/spawn-codex.ts +17 -2
  55. package/src/core/workflow/spawn-cursor.ts +18 -2
  56. package/src/core/workflow/spawn-failure.ts +40 -0
  57. package/src/core/workflow/types.ts +27 -0
  58. package/src/core/workflow/workspaces/local-git.ts +112 -22
  59. package/src/lib/exec.ts +13 -3
@@ -38,6 +38,13 @@ const MAX_INTENT_BYTES = 256 * 1024;
38
38
  const MAX_TEMPLATES = 20;
39
39
  const TEMPLATE_ID = /^[A-Za-z0-9][A-Za-z0-9._-]{0,63}$/;
40
40
  const FOREIGN_LEASE_STALE_MS = 24 * 60 * 60 * 1_000;
41
+ /** Ceiling on CONSECUTIVE uncharged (upstream) replans (ADR 0046). Because an
42
+ * uncharged replan does not spend `max_replans`, an unending vendor outage would
43
+ * otherwise replan forever; this bound stops it and names the outside service.
44
+ * Kept low — each replan is a real planner round-trip — and a module constant
45
+ * rather than a policy field to keep the frozen replanning schema unchanged; an
46
+ * environment failure self-bounds at one because it stops immediately. */
47
+ const MAX_UNCHARGED_REPLANS = 3;
41
48
 
42
49
  export interface SupervisorLimits {
43
50
  max_cycles: number;
@@ -322,6 +329,18 @@ function deriveProjection(
322
329
  ): SupervisorProjection {
323
330
  const root = work.find((record) => record.intent.id === plans.active_root_work_id);
324
331
  const attemptsUsed = governed.reduce((sum, record) => sum + record.projection.attempts_used, 0);
332
+ // The goal budget, like the per-item budget, counts only charged attempts
333
+ // (ADR 0046): an uncharged environment/upstream attempt on any child does not
334
+ // spend the goal's total. attempts_used stays the raw count for display.
335
+ const chargedUsed = governed.reduce((sum, record) => sum + record.projection.charged_attempts, 0);
336
+ // Replans, like per-item attempts (ADR 0046), are budgeted by the CHARGED
337
+ // count: an uncharged planner failure (environment/upstream) does not spend
338
+ // max_replans. replans_used stays the raw count for display.
339
+ const unchargedReplans = plans.plans.filter(
340
+ (plan) =>
341
+ plan.status === "failed" && (plan.class === "environment" || plan.class === "upstream"),
342
+ ).length;
343
+ const chargedReplans = plans.plans.length - unchargedReplans;
325
344
  const readyWork = work
326
345
  .filter((record) => record.projection.state === "ready")
327
346
  .map((record) => record.intent.id);
@@ -382,11 +401,11 @@ function deriveProjection(
382
401
  retryable_work: retryableWork,
383
402
  attention_work: attentionWork,
384
403
  attempts_used: attemptsUsed,
385
- attempts_remaining: Math.max(0, intent.limits.max_total_attempts - attemptsUsed),
404
+ attempts_remaining: Math.max(0, intent.limits.max_total_attempts - chargedUsed),
386
405
  specialists: Object.keys(intent.specialists),
387
406
  plan_generation: plans.generation,
388
407
  replans_used: plans.plans.length,
389
- replans_remaining: Math.max(0, (intent.replanning?.max_replans ?? 0) - plans.plans.length),
408
+ replans_remaining: Math.max(0, (intent.replanning?.max_replans ?? 0) - chargedReplans),
390
409
  milestones_completed: milestonesCompleted,
391
410
  milestones_remaining: Math.max(0, (intent.mission?.max_milestones ?? 0) - milestonesCompleted),
392
411
  pending_plan_id: plans.latest?.status === "proposed" ? plans.latest.request.id : undefined,
@@ -433,6 +452,46 @@ function deriveProjection(
433
452
  next_action: "replan",
434
453
  };
435
454
  }
455
+ // ADR 0046: a planner failure that never touched the plan is handled before
456
+ // the replan path. Otherwise a "failed" plan falls through to canReplan and
457
+ // the runner replans an unchanged environment on every tick — the failure mode
458
+ // that burned the measured 19 "codex not found" replans.
459
+ if (plans.latest?.status === "failed" && plans.latest.class === "environment") {
460
+ // A missing precondition (the planner binary was absent). Retrying an
461
+ // unchanged environment cannot help, so the goal STOPS and names it. A human
462
+ // who installs the binary can re-run; the failure was uncharged, so the
463
+ // replan budget is intact.
464
+ return {
465
+ ...base,
466
+ state: "blocked",
467
+ reason: `planning could not start; a required precondition is missing: ${plans.latest.reason ?? "the planner binary was absent"}`,
468
+ next_action: "none",
469
+ };
470
+ }
471
+ if (plans.latest?.status === "failed" && plans.latest.class === "upstream") {
472
+ // The vendor refused. Uncharged replans do not spend max_replans, so the
473
+ // only brake on an unending outage is this consecutive bound; at the limit
474
+ // the goal stops and names the outside service, distinct from work-blocked.
475
+ let trailingUncharged = 0;
476
+ for (let index = plans.plans.length - 1; index >= 0; index--) {
477
+ const plan = plans.plans[index]!;
478
+ if (plan.status !== "failed" || (plan.class !== "environment" && plan.class !== "upstream")) {
479
+ break;
480
+ }
481
+ trailingUncharged++;
482
+ }
483
+ if (trailingUncharged >= MAX_UNCHARGED_REPLANS) {
484
+ return {
485
+ ...base,
486
+ state: "blocked",
487
+ reason: `planning is blocked waiting on an outside service after ${trailingUncharged} consecutive uncharged attempt(s): ${plans.latest.reason ?? "the vendor refused"}`,
488
+ next_action: "none",
489
+ };
490
+ }
491
+ // Under the bound: fall through so the goal replans when nothing else is
492
+ // dispatchable — the vendor may have recovered. canReplan reads
493
+ // chargedReplans, so these uncharged attempts do not exhaust max_replans.
494
+ }
436
495
  const triggerFingerprint = supervisorGraphFingerprint({
437
496
  rootWorkId: plans.active_root_work_id,
438
497
  generation: plans.generation,
@@ -449,13 +508,11 @@ function deriveProjection(
449
508
  reason: plans.latest.reason ?? `plan ${plans.latest.request.id} requires attention`,
450
509
  attention_plan_id: plans.latest.request.id,
451
510
  next_action:
452
- intent.replanning && plans.plans.length < intent.replanning.max_replans
453
- ? "retry_plan"
454
- : "none",
511
+ intent.replanning && chargedReplans < intent.replanning.max_replans ? "retry_plan" : "none",
455
512
  };
456
513
  }
457
514
  if (!root && intent.mission) {
458
- if (!intent.replanning || plans.plans.length >= intent.replanning.max_replans) {
515
+ if (!intent.replanning || chargedReplans >= intent.replanning.max_replans) {
459
516
  return {
460
517
  ...base,
461
518
  state: "budget_exhausted",
@@ -472,7 +529,7 @@ function deriveProjection(
472
529
  }
473
530
  if (!root) throw new Error(`supervisor ${intent.id} root work is missing`);
474
531
  if (root.projection.state === "succeeded" && intent.mission) {
475
- if (!intent.replanning || plans.plans.length >= intent.replanning.max_replans) {
532
+ if (!intent.replanning || chargedReplans >= intent.replanning.max_replans) {
476
533
  return {
477
534
  ...base,
478
535
  state: "budget_exhausted",
@@ -517,7 +574,7 @@ function deriveProjection(
517
574
  next_action: "run",
518
575
  };
519
576
  }
520
- if (attemptsUsed >= intent.limits.max_total_attempts && attemptDispatchable.length > 0) {
577
+ if (chargedUsed >= intent.limits.max_total_attempts && attemptDispatchable.length > 0) {
521
578
  return {
522
579
  ...base,
523
580
  state: "budget_exhausted",
@@ -565,7 +622,7 @@ function deriveProjection(
565
622
  next_action: "retry",
566
623
  };
567
624
  }
568
- if (attemptsUsed >= intent.limits.max_total_attempts && cancelled.length === 0) {
625
+ if (chargedUsed >= intent.limits.max_total_attempts && cancelled.length === 0) {
569
626
  return {
570
627
  ...base,
571
628
  state: "budget_exhausted",
@@ -575,7 +632,7 @@ function deriveProjection(
575
632
  }
576
633
  const canReplan =
577
634
  intent.replanning !== undefined &&
578
- plans.plans.length < intent.replanning.max_replans &&
635
+ chargedReplans < intent.replanning.max_replans &&
579
636
  cancelled.length === 0 &&
580
637
  !latestHandledSameGraph;
581
638
  if (canReplan) {
@@ -590,7 +647,7 @@ function deriveProjection(
590
647
  }
591
648
  if (
592
649
  intent.replanning &&
593
- plans.plans.length >= intent.replanning.max_replans &&
650
+ chargedReplans >= intent.replanning.max_replans &&
594
651
  cancelled.length === 0
595
652
  ) {
596
653
  return {
@@ -68,7 +68,11 @@ export async function runWorkItem(input: RunWorkItemInput): Promise<RunReport> {
68
68
  if (!(record.projection.state === "ready" || record.projection.state === "blocked")) {
69
69
  throw new Error(`work item ${input.workId} cannot run from state ${record.projection.state}`);
70
70
  }
71
- if (record.projection.attempts_used >= record.intent.max_attempts) {
71
+ // Budget is spent by CHARGED attempts (ADR 0046); uncharged environment or
72
+ // upstream attempts don't count against it. The attempt NUMBER still comes
73
+ // from attempts_used so the history stays a gapless 1..N and the validator
74
+ // is satisfied.
75
+ if (record.projection.charged_attempts >= record.intent.max_attempts) {
72
76
  throw new Error(
73
77
  `work item ${input.workId} exhausted its ${record.intent.max_attempts} attempts`,
74
78
  );
@@ -132,7 +136,12 @@ function priorContext(
132
136
  coordRoot: string,
133
137
  prior: WorkAttempt,
134
138
  ): NonNullable<WorkflowAttemptContext["prior"]> {
135
- if (!prior.proof_path) {
139
+ // An uncharged attempt (ADR 0046) produced no information about the work, so
140
+ // it is presented to the retry the same way a lost attempt is — carrying no
141
+ // proof-derived cause. Reporting its 5xx/circuit-open text as a
142
+ // "workflow_error" would tell the model "your code errored" for something that
143
+ // never ran the work.
144
+ if (prior.uncharged || !prior.proof_path) {
136
145
  return {
137
146
  run_id: prior.run_id,
138
147
  causes: ["lost"],
@@ -39,6 +39,12 @@ const MAX_EVENTS_BYTES = 4 * 1024 * 1024;
39
39
  const MAX_EVENT_BYTES = 16 * 1024;
40
40
  const MAX_EVENTS = 1_000;
41
41
  const FOREIGN_LEASE_STALE_MS = 24 * 60 * 60 * 1_000;
42
+ /** Default ceiling on consecutive uncharged attempts (ADR 0046). Kept low: each
43
+ * one is a real vendor round-trip, so a handful gives a transient outage room
44
+ * to recover before the item stops and names the outside service. There is no
45
+ * backoff between them beyond the cadence of the supervisor and the vendor's own
46
+ * responses, which is why the bound stays small. */
47
+ const DEFAULT_MAX_UNCHARGED_ATTEMPTS = 3;
42
48
 
43
49
  export type WorkState =
44
50
  | "waiting"
@@ -71,6 +77,12 @@ export interface WorkIntent {
71
77
  dependencies: string[];
72
78
  workflow: { path: string; sha256: string };
73
79
  max_attempts: number;
80
+ /** Ceiling on CONSECUTIVE uncharged attempts (ADR 0046), separate from
81
+ * max_attempts. An upstream outage produces uncharged attempt after uncharged
82
+ * attempt; without a bound it would retry forever. At the bound the item stops
83
+ * and reports it is blocked on an outside service. Optional for back-compat:
84
+ * an intent written before ADR 0046 has none and falls back to the default. */
85
+ max_uncharged_attempts?: number;
74
86
  source?: { kind: "human" | "workflow" | "external"; ref?: string };
75
87
  created_at: string;
76
88
  }
@@ -128,6 +140,10 @@ export interface WorkAttempt {
128
140
  approval_id?: string;
129
141
  proof_path?: string;
130
142
  journal_error?: string;
143
+ /** Why this failed attempt was uninformative about the work (ADR 0046), read
144
+ * from the proof's run.class. Absent ⇒ the attempt is charged, exactly as
145
+ * before ADR 0046 (which is also how a proof without a class reads). */
146
+ uncharged?: "environment" | "upstream";
131
147
  }
132
148
 
133
149
  export interface WorkProjection {
@@ -138,7 +154,13 @@ export interface WorkProjection {
138
154
  next_action: WorkNextAction;
139
155
  unresolved_dependencies: string[];
140
156
  attempts: WorkAttempt[];
157
+ /** Every attempt started, charged or not. Drives the next attempt number and
158
+ * history ordering, so it counts uncharged attempts too. */
141
159
  attempts_used: number;
160
+ /** Attempts that were informative about the work (ADR 0046). This — not
161
+ * attempts_used — is what max_attempts budgets, so an uncharged environment or
162
+ * upstream attempt does not consume the retry budget. */
163
+ charged_attempts: number;
142
164
  attempts_remaining: number;
143
165
  latest_run_id?: string;
144
166
  approval_id?: string;
@@ -160,6 +182,7 @@ export interface CreateWorkItemInput {
160
182
  acceptance?: string[];
161
183
  dependencies?: string[];
162
184
  maxAttempts?: number;
185
+ maxUnchargedAttempts?: number;
163
186
  source?: WorkIntent["source"];
164
187
  id?: string;
165
188
  actor?: string;
@@ -188,6 +211,14 @@ export function createWorkItem(input: CreateWorkItemInput): WorkRecord {
188
211
  if (!Number.isSafeInteger(maxAttempts) || maxAttempts < 1 || maxAttempts > 100) {
189
212
  throw new Error("work maxAttempts must be an integer from 1 to 100");
190
213
  }
214
+ const maxUnchargedAttempts = input.maxUnchargedAttempts ?? DEFAULT_MAX_UNCHARGED_ATTEMPTS;
215
+ if (
216
+ !Number.isSafeInteger(maxUnchargedAttempts) ||
217
+ maxUnchargedAttempts < 1 ||
218
+ maxUnchargedAttempts > 100
219
+ ) {
220
+ throw new Error("work maxUnchargedAttempts must be an integer from 1 to 100");
221
+ }
191
222
  const acceptance = (input.acceptance ?? []).map((value, index) =>
192
223
  boundedString(value, `work acceptance[${index}]`, MAX_ACCEPTANCE_ITEM),
193
224
  );
@@ -204,6 +235,7 @@ export function createWorkItem(input: CreateWorkItemInput): WorkRecord {
204
235
  dependencies,
205
236
  workflow: { path: workflowPath, sha256: workflowScriptDigest(workflowPath) },
206
237
  max_attempts: maxAttempts,
238
+ max_uncharged_attempts: maxUnchargedAttempts,
207
239
  source,
208
240
  created_at: new Date().toISOString(),
209
241
  };
@@ -375,13 +407,18 @@ function deriveWorkProjection(
375
407
  const attempts = attemptEvents.map((event, index) =>
376
408
  inspectAttempt(coordRoot, event, intent, attemptEvents[index - 1]?.run_id),
377
409
  );
410
+ // Charged attempts — not the raw count — are what max_attempts budgets
411
+ // (ADR 0046). An uncharged environment/upstream attempt still increments
412
+ // attempts_used (ordering + next number) but not the budget.
413
+ const chargedAttempts = attempts.filter((attempt) => attempt.uncharged === undefined).length;
378
414
  const base = {
379
415
  id: intent.id,
380
416
  title: intent.title,
381
417
  unresolved_dependencies: [] as string[],
382
418
  attempts,
383
419
  attempts_used: attempts.length,
384
- attempts_remaining: Math.max(0, intent.max_attempts - attempts.length),
420
+ charged_attempts: chargedAttempts,
421
+ attempts_remaining: Math.max(0, intent.max_attempts - chargedAttempts),
385
422
  latest_run_id: attempts.at(-1)?.run_id,
386
423
  approval_id: attempts.at(-1)?.approval_id,
387
424
  proof_path: attempts.at(-1)?.proof_path,
@@ -462,7 +499,7 @@ function deriveWorkProjection(
462
499
  next_action: "review",
463
500
  };
464
501
  }
465
- const attemptsRemaining = intent.max_attempts - attempts.length;
502
+ const attemptsRemaining = intent.max_attempts - chargedAttempts;
466
503
  if (latest.status === "journal_unreadable") {
467
504
  return {
468
505
  ...base,
@@ -473,6 +510,49 @@ function deriveWorkProjection(
473
510
  next_action: attemptsRemaining > 0 ? "retry" : "none",
474
511
  };
475
512
  }
513
+ // Uncharged failures (ADR 0046) are handled before the ordinary work-failure
514
+ // path: they did not touch the work, so they neither spent the budget nor
515
+ // signal that retrying the work would help.
516
+ if (latest.status === "failed" && latest.uncharged === "environment") {
517
+ // A missing precondition. The operator chose to STOP the item immediately
518
+ // rather than retry an unchanged environment (ADR 0046); next_action "none"
519
+ // stops the supervisor. A human who fixes the environment can still force a
520
+ // retry — the attempt was uncharged, so the budget is intact.
521
+ return {
522
+ ...base,
523
+ state: "blocked",
524
+ reason: environmentBlockedReason(latest),
525
+ next_action: "none",
526
+ };
527
+ }
528
+ if (latest.status === "failed" && latest.uncharged === "upstream") {
529
+ // Count trailing consecutive uncharged attempts in the current window. The
530
+ // bound is the only brake on an outage that never ends, so at the limit the
531
+ // item stops and names the outside service — distinct from work-blocked.
532
+ const maxUncharged = intent.max_uncharged_attempts ?? DEFAULT_MAX_UNCHARGED_ATTEMPTS;
533
+ let trailingUncharged = 0;
534
+ for (let index = currentAttempts.length - 1; index >= 0; index--) {
535
+ if (currentAttempts[index]!.uncharged === undefined) break;
536
+ trailingUncharged++;
537
+ }
538
+ if (trailingUncharged >= maxUncharged || attemptsRemaining <= 0) {
539
+ return {
540
+ ...base,
541
+ state: "blocked",
542
+ reason:
543
+ trailingUncharged >= maxUncharged
544
+ ? `blocked waiting on an outside service after ${trailingUncharged} consecutive uncharged attempt(s): ${upstreamReason(latest)}`
545
+ : `workflow attempt ${latest.number} was uncharged (${upstreamReason(latest)}) but the work attempt budget is exhausted`,
546
+ next_action: "none",
547
+ };
548
+ }
549
+ return {
550
+ ...base,
551
+ state: "blocked",
552
+ reason: `workflow attempt ${latest.number} did not touch the work; an outside service refused: ${upstreamReason(latest)}`,
553
+ next_action: "retry",
554
+ };
555
+ }
476
556
  return {
477
557
  ...base,
478
558
  state: "blocked",
@@ -484,6 +564,33 @@ function deriveWorkProjection(
484
564
  };
485
565
  }
486
566
 
567
+ /** The failure reason for an uncharged attempt, read from the proof when
568
+ * present. Bounded so a verbose vendor transcript cannot bloat the projection
569
+ * reason (which is persisted verbatim into the reconciliation event). */
570
+ function unchargedProofError(attempt: WorkAttempt): string | undefined {
571
+ if (!attempt.proof_path) return undefined;
572
+ try {
573
+ const proof = parseObject(readFileSync(attempt.proof_path, "utf8"), "workflow proof");
574
+ const run = proof.run;
575
+ const error =
576
+ run && typeof run === "object" ? (run as Record<string, unknown>).error : undefined;
577
+ return typeof error === "string" && error.trim() ? boundedJournalError(error) : undefined;
578
+ } catch {
579
+ return undefined;
580
+ }
581
+ }
582
+
583
+ function environmentBlockedReason(attempt: WorkAttempt): string {
584
+ const detail = unchargedProofError(attempt);
585
+ return detail
586
+ ? `workflow attempt ${attempt.number} could not start; a required precondition is missing: ${detail}`
587
+ : `workflow attempt ${attempt.number} could not start; a required precondition is missing`;
588
+ }
589
+
590
+ function upstreamReason(attempt: WorkAttempt): string {
591
+ return unchargedProofError(attempt) ?? "the vendor was reached and refused";
592
+ }
593
+
487
594
  function inspectAttempt(
488
595
  coordRoot: string,
489
596
  event: WorkEvent,
@@ -547,6 +654,15 @@ function inspectAttempt(
547
654
  proof.acceptance.summary.unknown === 0
548
655
  ? "succeeded"
549
656
  : "failed";
657
+ // A failed attempt the run classified as uninformative about the work is
658
+ // uncharged (ADR 0046). Absent class ⇒ charged, as before. Read only for a
659
+ // failed attempt: a succeeded run never carries a class.
660
+ if (
661
+ attempt.status === "failed" &&
662
+ (proof.run.class === "environment" || proof.run.class === "upstream")
663
+ ) {
664
+ attempt.uncharged = proof.run.class;
665
+ }
550
666
  return attempt;
551
667
  }
552
668
  const journalPath = workflowJournalPath(coordRoot, runId);
@@ -882,7 +998,13 @@ function validateWorkIntent(intent: WorkIntent, workId: string): void {
882
998
  !/^[a-f0-9]{64}$/.test(intent.workflow.sha256) ||
883
999
  !Number.isSafeInteger(intent.max_attempts) ||
884
1000
  intent.max_attempts < 1 ||
885
- intent.max_attempts > 100
1001
+ intent.max_attempts > 100 ||
1002
+ // Optional for back-compat: absent on pre-ADR-0046 intents. When present it
1003
+ // must be a valid bound.
1004
+ (intent.max_uncharged_attempts !== undefined &&
1005
+ (!Number.isSafeInteger(intent.max_uncharged_attempts) ||
1006
+ intent.max_uncharged_attempts < 1 ||
1007
+ intent.max_uncharged_attempts > 100))
886
1008
  ) {
887
1009
  throw new Error(`work intent ${workId} has an unsupported or mismatched schema`);
888
1010
  }
@@ -911,6 +911,11 @@ async function executeWorkflow(
911
911
 
912
912
  if (!last.ok) {
913
913
  journal("agent.attempt_failed", { id, attempt, error: last.error });
914
+ // ADR 0046: an environment failure (the binary was absent) cannot be
915
+ // helped by retrying an unchanged environment, so stop the in-agent
916
+ // retry too — not just the outer attempt/replan budget. An upstream
917
+ // refusal keeps retrying here: the vendor may recover mid-loop.
918
+ if (last.class === "environment") break;
914
919
  continue; // spawn-level failure: retry with the original prompt
915
920
  }
916
921
  if (!agentOpts.schema) {
@@ -982,6 +987,12 @@ async function executeWorkflow(
982
987
  agentCostUsd > 0 || last?.costUsd !== undefined ? agentCostUsd : undefined;
983
988
  agentProof.session_id = last?.sessionId;
984
989
  agentProof.error = reason;
990
+ // Carry the spawn class (environment/upstream) onto the proof only when the
991
+ // final outcome was a spawn failure. A schema failure after a spawn that
992
+ // reached the model is a work failure — left unclassed (charged). An
993
+ // earlier attempt that transiently failed and then succeeded returned
994
+ // above, so this only fires when the agent genuinely failed.
995
+ if (last && !last.ok && last.class) agentProof.class = last.class;
985
996
  journal("agent.failed", { id, error: reason });
986
997
  throw new Error(`agent ${proofLabel}: ${reason}`);
987
998
  } catch (error) {
@@ -25,6 +25,7 @@ export { runWorkflow, WorkflowParkedError, WorkflowRunError } from "./engine.ts"
25
25
  export {
26
26
  buildWorkflowProof,
27
27
  createEvidenceRecord,
28
+ deriveRunFailureClass,
28
29
  digestResult,
29
30
  normalizeWorkflowMeta,
30
31
  readWorkflowProof,
@@ -43,6 +44,7 @@ export {
43
44
  workflowScriptDigest,
44
45
  writeWorkflowRunManifest,
45
46
  } from "./run-state.ts";
47
+ export { isUpstreamFailureText, vendorFailureText } from "./spawn-failure.ts";
46
48
  export type {
47
49
  AcceptanceCriterion,
48
50
  AcceptanceResult,
@@ -59,6 +61,7 @@ export type {
59
61
  ResultDigest,
60
62
  RunReport,
61
63
  Spawner,
64
+ SpawnFailureClass,
62
65
  SpawnRequest,
63
66
  SpawnResult,
64
67
  StageSchema,
@@ -20,6 +20,7 @@ import type {
20
20
  HarnessEvidenceCapability,
21
21
  HarnessEvidenceCoverage,
22
22
  ResultDigest,
23
+ SpawnFailureClass,
23
24
  WorkflowAgentProof,
24
25
  WorkflowAttemptContext,
25
26
  WorkflowEvidenceInput,
@@ -218,6 +219,36 @@ export function digestResult(value: unknown, kind?: "text" | "json"): ResultDige
218
219
  };
219
220
  }
220
221
 
222
+ /**
223
+ * The run-level failure class (ADR 0046), derived from the agents rather than a
224
+ * single terminal throw so it survives a script's `parallel()` swallowing the
225
+ * rejection — a swallowed agent's proof is still recorded with its class.
226
+ *
227
+ * Rules, in order, all in service of "default to charging":
228
+ * 1. A succeeded run is never classed (there is nothing uncharged about it).
229
+ * 2. If ANY agent produced a result (succeeded or replayed from cache) the
230
+ * attempt was informative about the work — charge it. This also keeps a
231
+ * resumed run whose earlier segment did real work from being written off by
232
+ * a later environment failure.
233
+ * 3. Otherwise, among the failed agents, environment wins over upstream: a
234
+ * missing binary means nothing ran at all, and it is the operator-chosen
235
+ * hard stop.
236
+ * 4. Anything else is undefined ⇒ a charged work failure, exactly as today.
237
+ */
238
+ export function deriveRunFailureClass(
239
+ status: "succeeded" | "failed",
240
+ agents: readonly Pick<WorkflowAgentProof, "status" | "class">[],
241
+ ): SpawnFailureClass | undefined {
242
+ if (status !== "failed") return undefined;
243
+ if (agents.some((agent) => agent.status === "succeeded" || agent.status === "cached")) {
244
+ return undefined;
245
+ }
246
+ const failed = agents.filter((agent) => agent.status === "failed");
247
+ if (failed.some((agent) => agent.class === "environment")) return "environment";
248
+ if (failed.some((agent) => agent.class === "upstream")) return "upstream";
249
+ return undefined;
250
+ }
251
+
221
252
  export function buildWorkflowProof(input: BuildWorkflowProofInput): WorkflowProof {
222
253
  const acceptance = rollupAcceptance(input.meta.acceptance, input.evidence);
223
254
  const repository = buildRepoEvidence(input.before, input.after);
@@ -231,6 +262,7 @@ export function buildWorkflowProof(input: BuildWorkflowProofInput): WorkflowProo
231
262
  }));
232
263
  const harnesses = buildHarnessCoverage(agents, input.harnessEvidence, input.harnessAttestations);
233
264
  const unknowns = buildUnknowns(agents, harnesses, repository);
265
+ const runClass = deriveRunFailureClass(input.status, agents);
234
266
  const journal = readFileSync(input.journalPath);
235
267
  return {
236
268
  schema_version: WORKFLOW_PROOF_SCHEMA_VERSION,
@@ -247,6 +279,7 @@ export function buildWorkflowProof(input: BuildWorkflowProofInput): WorkflowProo
247
279
  objective: input.meta.objective,
248
280
  error: clippedOptional(input.error, MAX_SUMMARY_CHARS),
249
281
  result: input.result === undefined ? undefined : digestResult(input.result),
282
+ ...(runClass ? { class: runClass } : {}),
250
283
  },
251
284
  acceptance,
252
285
  agents,
@@ -326,6 +359,9 @@ export function readWorkflowProof(coordRoot: string, runId: string): WorkflowPro
326
359
  if (
327
360
  proof.schema_version !== WORKFLOW_PROOF_SCHEMA_VERSION ||
328
361
  proof.run?.id !== runId ||
362
+ (proof.run.class !== undefined &&
363
+ proof.run.class !== "environment" &&
364
+ proof.run.class !== "upstream") ||
329
365
  (proof.run.work_context !== undefined &&
330
366
  (!proof.run.work_item_id ||
331
367
  proof.run.work_context.id !== proof.run.work_item_id ||
@@ -24,7 +24,7 @@ import type { HarnessInvocation, HarnessRawResult } from "../harnesses/types.ts"
24
24
  import { buildChildEnv } from "./child-env.ts";
25
25
  import { notFoundError } from "./harnesses.ts";
26
26
  import { resolveSandboxProjection } from "./sandbox-projection.ts";
27
- import { vendorFailureText } from "./spawn-failure.ts";
27
+ import { isUpstreamFailureText, vendorFailureText } from "./spawn-failure.ts";
28
28
  import type { Spawner, SpawnRequest, SpawnResult } from "./types.ts";
29
29
 
30
30
  interface ClaudeEnvelope {
@@ -70,7 +70,20 @@ export function normalizeClaudeResult(raw: HarnessRawResult): SpawnResult {
70
70
  error: `claude timed out after ${raw.durationMs}ms and was killed`,
71
71
  };
72
72
  }
73
+ // Structural environment signal: the binary was never there (spawned directly,
74
+ // so a missing binary surfaces as ENOENT). Uncharged and not retried.
75
+ if (raw.spawnErrno === "ENOENT") {
76
+ return {
77
+ ok: false,
78
+ text: "",
79
+ durationMs: raw.durationMs,
80
+ error: notFoundError("claude-code"),
81
+ class: "environment",
82
+ };
83
+ }
73
84
  if (raw.exitCode === 127) {
85
+ // A bare 127 with no errno is a shell/vendor 127, indistinguishable from a
86
+ // legitimate one — charged as work rather than classed environment.
74
87
  return {
75
88
  ok: false,
76
89
  text: "",
@@ -79,11 +92,13 @@ export function normalizeClaudeResult(raw: HarnessRawResult): SpawnResult {
79
92
  };
80
93
  }
81
94
  if (raw.exitCode !== 0) {
95
+ const failureText = vendorFailureText(raw);
82
96
  return {
83
97
  ok: false,
84
98
  text: "",
85
99
  durationMs: raw.durationMs,
86
- error: `claude exited ${raw.exitCode}: ${vendorFailureText(raw)}`,
100
+ error: `claude exited ${raw.exitCode}: ${failureText}`,
101
+ ...(isUpstreamFailureText(failureText) ? { class: "upstream" as const } : {}),
87
102
  };
88
103
  }
89
104
 
@@ -100,6 +115,9 @@ export function normalizeClaudeResult(raw: HarnessRawResult): SpawnResult {
100
115
  }
101
116
 
102
117
  if (envelope.is_error) {
118
+ const envelopeError = `${envelope.subtype ?? ""} ${(envelope.errors ?? []).join("; ")} ${String(
119
+ envelope.result ?? "",
120
+ )}`;
103
121
  return {
104
122
  ok: false,
105
123
  text: String(envelope.result ?? ""),
@@ -107,6 +125,7 @@ export function normalizeClaudeResult(raw: HarnessRawResult): SpawnResult {
107
125
  costUsd: envelope.total_cost_usd,
108
126
  durationMs: raw.durationMs,
109
127
  error: `harness error (${envelope.subtype ?? "unknown"}): ${(envelope.errors ?? []).join("; ") || "see envelope"}`,
128
+ ...(isUpstreamFailureText(envelopeError) ? { class: "upstream" as const } : {}),
110
129
  };
111
130
  }
112
131
 
@@ -25,7 +25,7 @@ import type { HarnessInvocation, HarnessRawResult } from "../harnesses/types.ts"
25
25
  import { buildChildEnv } from "./child-env.ts";
26
26
  import { notFoundError } from "./harnesses.ts";
27
27
  import { resolveSandboxProjection } from "./sandbox-projection.ts";
28
- import { vendorFailureText } from "./spawn-failure.ts";
28
+ import { isUpstreamFailureText, vendorFailureText } from "./spawn-failure.ts";
29
29
  import type { Spawner, SpawnRequest, SpawnResult } from "./types.ts";
30
30
 
31
31
  export function buildCodexInvocation(req: SpawnRequest, resultFile?: string): HarnessInvocation {
@@ -69,15 +69,30 @@ export function normalizeCodexResult(raw: HarnessRawResult): SpawnResult {
69
69
  error: `codex timed out after ${raw.durationMs}ms and was killed`,
70
70
  };
71
71
  }
72
+ // Structural environment signal: the binary was never there (spawned directly,
73
+ // so a missing binary surfaces as ENOENT). Uncharged and not retried.
74
+ if (raw.spawnErrno === "ENOENT") {
75
+ return {
76
+ ok: false,
77
+ text: "",
78
+ durationMs: raw.durationMs,
79
+ error: notFoundError("codex"),
80
+ class: "environment",
81
+ };
82
+ }
72
83
  if (raw.exitCode === 127) {
84
+ // A bare 127 with no errno is a shell/vendor 127, indistinguishable from a
85
+ // legitimate one — charged as work rather than classed environment.
73
86
  return { ok: false, text: "", durationMs: raw.durationMs, error: notFoundError("codex") };
74
87
  }
75
88
  if (raw.exitCode !== 0) {
89
+ const failureText = vendorFailureText(raw);
76
90
  return {
77
91
  ok: false,
78
92
  text: "",
79
93
  durationMs: raw.durationMs,
80
- error: `codex exited ${raw.exitCode}: ${vendorFailureText(raw)}`,
94
+ error: `codex exited ${raw.exitCode}: ${failureText}`,
95
+ ...(isUpstreamFailureText(failureText) ? { class: "upstream" as const } : {}),
81
96
  };
82
97
  }
83
98
  return {
@@ -25,7 +25,7 @@ import type { HarnessInvocation, HarnessRawResult } from "../harnesses/types.ts"
25
25
  import { buildChildEnv } from "./child-env.ts";
26
26
  import { notFoundError } from "./harnesses.ts";
27
27
  import { resolveSandboxProjection } from "./sandbox-projection.ts";
28
- import { vendorFailureText } from "./spawn-failure.ts";
28
+ import { isUpstreamFailureText, vendorFailureText } from "./spawn-failure.ts";
29
29
  import type { Spawner, SpawnRequest, SpawnResult } from "./types.ts";
30
30
 
31
31
  interface CursorEnvelope {
@@ -82,15 +82,30 @@ export function normalizeCursorResult(raw: HarnessRawResult): SpawnResult {
82
82
  error: `cursor timed out after ${raw.durationMs}ms and was killed`,
83
83
  };
84
84
  }
85
+ // Structural environment signal: the binary was never there (spawned directly,
86
+ // so a missing binary surfaces as ENOENT). Uncharged and not retried.
87
+ if (raw.spawnErrno === "ENOENT") {
88
+ return {
89
+ ok: false,
90
+ text: "",
91
+ durationMs: raw.durationMs,
92
+ error: notFoundError("cursor"),
93
+ class: "environment",
94
+ };
95
+ }
85
96
  if (raw.exitCode === 127) {
97
+ // A bare 127 with no errno is a shell/vendor 127, indistinguishable from a
98
+ // legitimate one — charged as work rather than classed environment.
86
99
  return { ok: false, text: "", durationMs: raw.durationMs, error: notFoundError("cursor") };
87
100
  }
88
101
  if (raw.exitCode !== 0) {
102
+ const failureText = vendorFailureText(raw);
89
103
  return {
90
104
  ok: false,
91
105
  text: "",
92
106
  durationMs: raw.durationMs,
93
- error: `cursor-agent exited ${raw.exitCode}: ${vendorFailureText(raw)}`,
107
+ error: `cursor-agent exited ${raw.exitCode}: ${failureText}`,
108
+ ...(isUpstreamFailureText(failureText) ? { class: "upstream" as const } : {}),
94
109
  };
95
110
  }
96
111
 
@@ -102,6 +117,7 @@ export function normalizeCursorResult(raw: HarnessRawResult): SpawnResult {
102
117
  sessionId: parsed.sessionId,
103
118
  durationMs: raw.durationMs,
104
119
  error: `cursor-agent reported is_error: ${parsed.text.slice(0, 300)}`,
120
+ ...(isUpstreamFailureText(parsed.text) ? { class: "upstream" as const } : {}),
105
121
  };
106
122
  }
107
123
  return {