harnery 0.28.0 → 0.30.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/commands/work.d.ts +2 -1
- package/dist/commands/work.d.ts.map +1 -1
- package/dist/commands/work.js +44 -4
- package/dist/core/harnesses/types.d.ts +5 -0
- package/dist/core/harnesses/types.d.ts.map +1 -1
- package/dist/core/supervisor/plan-read.js +7 -1
- package/dist/core/supervisor/plan-types.d.ts +10 -0
- package/dist/core/supervisor/plan-types.d.ts.map +1 -1
- package/dist/core/supervisor/planning.d.ts.map +1 -1
- package/dist/core/supervisor/planning.js +18 -0
- package/dist/core/supervisor/state.d.ts.map +1 -1
- package/dist/core/supervisor/state.js +65 -11
- package/dist/core/work/runner.d.ts.map +1 -1
- package/dist/core/work/runner.js +11 -2
- package/dist/core/work/state.d.ts +17 -0
- package/dist/core/work/state.d.ts.map +1 -1
- package/dist/core/work/state.js +102 -3
- package/dist/core/workflow/engine.js +13 -0
- package/dist/core/workflow/index.d.ts +3 -2
- package/dist/core/workflow/index.d.ts.map +1 -1
- package/dist/core/workflow/index.js +2 -1
- package/dist/core/workflow/proof.d.ts +18 -1
- package/dist/core/workflow/proof.d.ts.map +1 -1
- package/dist/core/workflow/proof.js +34 -0
- package/dist/core/workflow/spawn-claude.d.ts.map +1 -1
- package/dist/core/workflow/spawn-claude.js +19 -2
- package/dist/core/workflow/spawn-codex.d.ts.map +1 -1
- package/dist/core/workflow/spawn-codex.js +17 -2
- package/dist/core/workflow/spawn-cursor.d.ts.map +1 -1
- package/dist/core/workflow/spawn-cursor.js +18 -2
- package/dist/core/workflow/spawn-failure.d.ts +14 -0
- package/dist/core/workflow/spawn-failure.d.ts.map +1 -1
- package/dist/core/workflow/spawn-failure.js +34 -0
- package/dist/core/workflow/types.d.ts +26 -0
- package/dist/core/workflow/types.d.ts.map +1 -1
- package/dist/core/workflow/workspaces/local-git.d.ts.map +1 -1
- package/dist/core/workflow/workspaces/local-git.js +101 -20
- package/dist/lib/exec.d.ts +7 -0
- package/dist/lib/exec.d.ts.map +1 -1
- package/dist/lib/exec.js +6 -3
- package/package.json +1 -1
- package/src/commands/work.ts +51 -4
- package/src/core/harnesses/types.ts +5 -0
- package/src/core/supervisor/plan-read.ts +11 -2
- package/src/core/supervisor/plan-types.ts +11 -0
- package/src/core/supervisor/planning.ts +21 -0
- package/src/core/supervisor/state.ts +68 -11
- package/src/core/work/runner.ts +11 -2
- package/src/core/work/state.ts +125 -3
- package/src/core/workflow/engine.ts +11 -0
- package/src/core/workflow/index.ts +3 -0
- package/src/core/workflow/proof.ts +36 -0
- package/src/core/workflow/spawn-claude.ts +21 -2
- package/src/core/workflow/spawn-codex.ts +17 -2
- package/src/core/workflow/spawn-cursor.ts +18 -2
- package/src/core/workflow/spawn-failure.ts +40 -0
- package/src/core/workflow/types.ts +27 -0
- package/src/core/workflow/workspaces/local-git.ts +112 -22
- package/src/lib/exec.ts +13 -3
|
@@ -38,6 +38,13 @@ const MAX_INTENT_BYTES = 256 * 1024;
|
|
|
38
38
|
const MAX_TEMPLATES = 20;
|
|
39
39
|
const TEMPLATE_ID = /^[A-Za-z0-9][A-Za-z0-9._-]{0,63}$/;
|
|
40
40
|
const FOREIGN_LEASE_STALE_MS = 24 * 60 * 60 * 1_000;
|
|
41
|
+
/** Ceiling on CONSECUTIVE uncharged (upstream) replans (ADR 0046). Because an
|
|
42
|
+
* uncharged replan does not spend `max_replans`, an unending vendor outage would
|
|
43
|
+
* otherwise replan forever; this bound stops it and names the outside service.
|
|
44
|
+
* Kept low — each replan is a real planner round-trip — and a module constant
|
|
45
|
+
* rather than a policy field to keep the frozen replanning schema unchanged; an
|
|
46
|
+
* environment failure self-bounds at one because it stops immediately. */
|
|
47
|
+
const MAX_UNCHARGED_REPLANS = 3;
|
|
41
48
|
|
|
42
49
|
export interface SupervisorLimits {
|
|
43
50
|
max_cycles: number;
|
|
@@ -322,6 +329,18 @@ function deriveProjection(
|
|
|
322
329
|
): SupervisorProjection {
|
|
323
330
|
const root = work.find((record) => record.intent.id === plans.active_root_work_id);
|
|
324
331
|
const attemptsUsed = governed.reduce((sum, record) => sum + record.projection.attempts_used, 0);
|
|
332
|
+
// The goal budget, like the per-item budget, counts only charged attempts
|
|
333
|
+
// (ADR 0046): an uncharged environment/upstream attempt on any child does not
|
|
334
|
+
// spend the goal's total. attempts_used stays the raw count for display.
|
|
335
|
+
const chargedUsed = governed.reduce((sum, record) => sum + record.projection.charged_attempts, 0);
|
|
336
|
+
// Replans, like per-item attempts (ADR 0046), are budgeted by the CHARGED
|
|
337
|
+
// count: an uncharged planner failure (environment/upstream) does not spend
|
|
338
|
+
// max_replans. replans_used stays the raw count for display.
|
|
339
|
+
const unchargedReplans = plans.plans.filter(
|
|
340
|
+
(plan) =>
|
|
341
|
+
plan.status === "failed" && (plan.class === "environment" || plan.class === "upstream"),
|
|
342
|
+
).length;
|
|
343
|
+
const chargedReplans = plans.plans.length - unchargedReplans;
|
|
325
344
|
const readyWork = work
|
|
326
345
|
.filter((record) => record.projection.state === "ready")
|
|
327
346
|
.map((record) => record.intent.id);
|
|
@@ -382,11 +401,11 @@ function deriveProjection(
|
|
|
382
401
|
retryable_work: retryableWork,
|
|
383
402
|
attention_work: attentionWork,
|
|
384
403
|
attempts_used: attemptsUsed,
|
|
385
|
-
attempts_remaining: Math.max(0, intent.limits.max_total_attempts -
|
|
404
|
+
attempts_remaining: Math.max(0, intent.limits.max_total_attempts - chargedUsed),
|
|
386
405
|
specialists: Object.keys(intent.specialists),
|
|
387
406
|
plan_generation: plans.generation,
|
|
388
407
|
replans_used: plans.plans.length,
|
|
389
|
-
replans_remaining: Math.max(0, (intent.replanning?.max_replans ?? 0) -
|
|
408
|
+
replans_remaining: Math.max(0, (intent.replanning?.max_replans ?? 0) - chargedReplans),
|
|
390
409
|
milestones_completed: milestonesCompleted,
|
|
391
410
|
milestones_remaining: Math.max(0, (intent.mission?.max_milestones ?? 0) - milestonesCompleted),
|
|
392
411
|
pending_plan_id: plans.latest?.status === "proposed" ? plans.latest.request.id : undefined,
|
|
@@ -433,6 +452,46 @@ function deriveProjection(
|
|
|
433
452
|
next_action: "replan",
|
|
434
453
|
};
|
|
435
454
|
}
|
|
455
|
+
// ADR 0046: a planner failure that never touched the plan is handled before
|
|
456
|
+
// the replan path. Otherwise a "failed" plan falls through to canReplan and
|
|
457
|
+
// the runner replans an unchanged environment on every tick — the failure mode
|
|
458
|
+
// that burned the measured 19 "codex not found" replans.
|
|
459
|
+
if (plans.latest?.status === "failed" && plans.latest.class === "environment") {
|
|
460
|
+
// A missing precondition (the planner binary was absent). Retrying an
|
|
461
|
+
// unchanged environment cannot help, so the goal STOPS and names it. A human
|
|
462
|
+
// who installs the binary can re-run; the failure was uncharged, so the
|
|
463
|
+
// replan budget is intact.
|
|
464
|
+
return {
|
|
465
|
+
...base,
|
|
466
|
+
state: "blocked",
|
|
467
|
+
reason: `planning could not start; a required precondition is missing: ${plans.latest.reason ?? "the planner binary was absent"}`,
|
|
468
|
+
next_action: "none",
|
|
469
|
+
};
|
|
470
|
+
}
|
|
471
|
+
if (plans.latest?.status === "failed" && plans.latest.class === "upstream") {
|
|
472
|
+
// The vendor refused. Uncharged replans do not spend max_replans, so the
|
|
473
|
+
// only brake on an unending outage is this consecutive bound; at the limit
|
|
474
|
+
// the goal stops and names the outside service, distinct from work-blocked.
|
|
475
|
+
let trailingUncharged = 0;
|
|
476
|
+
for (let index = plans.plans.length - 1; index >= 0; index--) {
|
|
477
|
+
const plan = plans.plans[index]!;
|
|
478
|
+
if (plan.status !== "failed" || (plan.class !== "environment" && plan.class !== "upstream")) {
|
|
479
|
+
break;
|
|
480
|
+
}
|
|
481
|
+
trailingUncharged++;
|
|
482
|
+
}
|
|
483
|
+
if (trailingUncharged >= MAX_UNCHARGED_REPLANS) {
|
|
484
|
+
return {
|
|
485
|
+
...base,
|
|
486
|
+
state: "blocked",
|
|
487
|
+
reason: `planning is blocked waiting on an outside service after ${trailingUncharged} consecutive uncharged attempt(s): ${plans.latest.reason ?? "the vendor refused"}`,
|
|
488
|
+
next_action: "none",
|
|
489
|
+
};
|
|
490
|
+
}
|
|
491
|
+
// Under the bound: fall through so the goal replans when nothing else is
|
|
492
|
+
// dispatchable — the vendor may have recovered. canReplan reads
|
|
493
|
+
// chargedReplans, so these uncharged attempts do not exhaust max_replans.
|
|
494
|
+
}
|
|
436
495
|
const triggerFingerprint = supervisorGraphFingerprint({
|
|
437
496
|
rootWorkId: plans.active_root_work_id,
|
|
438
497
|
generation: plans.generation,
|
|
@@ -449,13 +508,11 @@ function deriveProjection(
|
|
|
449
508
|
reason: plans.latest.reason ?? `plan ${plans.latest.request.id} requires attention`,
|
|
450
509
|
attention_plan_id: plans.latest.request.id,
|
|
451
510
|
next_action:
|
|
452
|
-
intent.replanning &&
|
|
453
|
-
? "retry_plan"
|
|
454
|
-
: "none",
|
|
511
|
+
intent.replanning && chargedReplans < intent.replanning.max_replans ? "retry_plan" : "none",
|
|
455
512
|
};
|
|
456
513
|
}
|
|
457
514
|
if (!root && intent.mission) {
|
|
458
|
-
if (!intent.replanning ||
|
|
515
|
+
if (!intent.replanning || chargedReplans >= intent.replanning.max_replans) {
|
|
459
516
|
return {
|
|
460
517
|
...base,
|
|
461
518
|
state: "budget_exhausted",
|
|
@@ -472,7 +529,7 @@ function deriveProjection(
|
|
|
472
529
|
}
|
|
473
530
|
if (!root) throw new Error(`supervisor ${intent.id} root work is missing`);
|
|
474
531
|
if (root.projection.state === "succeeded" && intent.mission) {
|
|
475
|
-
if (!intent.replanning ||
|
|
532
|
+
if (!intent.replanning || chargedReplans >= intent.replanning.max_replans) {
|
|
476
533
|
return {
|
|
477
534
|
...base,
|
|
478
535
|
state: "budget_exhausted",
|
|
@@ -517,7 +574,7 @@ function deriveProjection(
|
|
|
517
574
|
next_action: "run",
|
|
518
575
|
};
|
|
519
576
|
}
|
|
520
|
-
if (
|
|
577
|
+
if (chargedUsed >= intent.limits.max_total_attempts && attemptDispatchable.length > 0) {
|
|
521
578
|
return {
|
|
522
579
|
...base,
|
|
523
580
|
state: "budget_exhausted",
|
|
@@ -565,7 +622,7 @@ function deriveProjection(
|
|
|
565
622
|
next_action: "retry",
|
|
566
623
|
};
|
|
567
624
|
}
|
|
568
|
-
if (
|
|
625
|
+
if (chargedUsed >= intent.limits.max_total_attempts && cancelled.length === 0) {
|
|
569
626
|
return {
|
|
570
627
|
...base,
|
|
571
628
|
state: "budget_exhausted",
|
|
@@ -575,7 +632,7 @@ function deriveProjection(
|
|
|
575
632
|
}
|
|
576
633
|
const canReplan =
|
|
577
634
|
intent.replanning !== undefined &&
|
|
578
|
-
|
|
635
|
+
chargedReplans < intent.replanning.max_replans &&
|
|
579
636
|
cancelled.length === 0 &&
|
|
580
637
|
!latestHandledSameGraph;
|
|
581
638
|
if (canReplan) {
|
|
@@ -590,7 +647,7 @@ function deriveProjection(
|
|
|
590
647
|
}
|
|
591
648
|
if (
|
|
592
649
|
intent.replanning &&
|
|
593
|
-
|
|
650
|
+
chargedReplans >= intent.replanning.max_replans &&
|
|
594
651
|
cancelled.length === 0
|
|
595
652
|
) {
|
|
596
653
|
return {
|
package/src/core/work/runner.ts
CHANGED
|
@@ -68,7 +68,11 @@ export async function runWorkItem(input: RunWorkItemInput): Promise<RunReport> {
|
|
|
68
68
|
if (!(record.projection.state === "ready" || record.projection.state === "blocked")) {
|
|
69
69
|
throw new Error(`work item ${input.workId} cannot run from state ${record.projection.state}`);
|
|
70
70
|
}
|
|
71
|
-
|
|
71
|
+
// Budget is spent by CHARGED attempts (ADR 0046); uncharged environment or
|
|
72
|
+
// upstream attempts don't count against it. The attempt NUMBER still comes
|
|
73
|
+
// from attempts_used so the history stays a gapless 1..N and the validator
|
|
74
|
+
// is satisfied.
|
|
75
|
+
if (record.projection.charged_attempts >= record.intent.max_attempts) {
|
|
72
76
|
throw new Error(
|
|
73
77
|
`work item ${input.workId} exhausted its ${record.intent.max_attempts} attempts`,
|
|
74
78
|
);
|
|
@@ -132,7 +136,12 @@ function priorContext(
|
|
|
132
136
|
coordRoot: string,
|
|
133
137
|
prior: WorkAttempt,
|
|
134
138
|
): NonNullable<WorkflowAttemptContext["prior"]> {
|
|
135
|
-
|
|
139
|
+
// An uncharged attempt (ADR 0046) produced no information about the work, so
|
|
140
|
+
// it is presented to the retry the same way a lost attempt is — carrying no
|
|
141
|
+
// proof-derived cause. Reporting its 5xx/circuit-open text as a
|
|
142
|
+
// "workflow_error" would tell the model "your code errored" for something that
|
|
143
|
+
// never ran the work.
|
|
144
|
+
if (prior.uncharged || !prior.proof_path) {
|
|
136
145
|
return {
|
|
137
146
|
run_id: prior.run_id,
|
|
138
147
|
causes: ["lost"],
|
package/src/core/work/state.ts
CHANGED
|
@@ -39,6 +39,12 @@ const MAX_EVENTS_BYTES = 4 * 1024 * 1024;
|
|
|
39
39
|
const MAX_EVENT_BYTES = 16 * 1024;
|
|
40
40
|
const MAX_EVENTS = 1_000;
|
|
41
41
|
const FOREIGN_LEASE_STALE_MS = 24 * 60 * 60 * 1_000;
|
|
42
|
+
/** Default ceiling on consecutive uncharged attempts (ADR 0046). Kept low: each
|
|
43
|
+
* one is a real vendor round-trip, so a handful gives a transient outage room
|
|
44
|
+
* to recover before the item stops and names the outside service. There is no
|
|
45
|
+
* backoff between them beyond the cadence of the supervisor and the vendor's own
|
|
46
|
+
* responses, which is why the bound stays small. */
|
|
47
|
+
const DEFAULT_MAX_UNCHARGED_ATTEMPTS = 3;
|
|
42
48
|
|
|
43
49
|
export type WorkState =
|
|
44
50
|
| "waiting"
|
|
@@ -71,6 +77,12 @@ export interface WorkIntent {
|
|
|
71
77
|
dependencies: string[];
|
|
72
78
|
workflow: { path: string; sha256: string };
|
|
73
79
|
max_attempts: number;
|
|
80
|
+
/** Ceiling on CONSECUTIVE uncharged attempts (ADR 0046), separate from
|
|
81
|
+
* max_attempts. An upstream outage produces uncharged attempt after uncharged
|
|
82
|
+
* attempt; without a bound it would retry forever. At the bound the item stops
|
|
83
|
+
* and reports it is blocked on an outside service. Optional for back-compat:
|
|
84
|
+
* an intent written before ADR 0046 has none and falls back to the default. */
|
|
85
|
+
max_uncharged_attempts?: number;
|
|
74
86
|
source?: { kind: "human" | "workflow" | "external"; ref?: string };
|
|
75
87
|
created_at: string;
|
|
76
88
|
}
|
|
@@ -128,6 +140,10 @@ export interface WorkAttempt {
|
|
|
128
140
|
approval_id?: string;
|
|
129
141
|
proof_path?: string;
|
|
130
142
|
journal_error?: string;
|
|
143
|
+
/** Why this failed attempt was uninformative about the work (ADR 0046), read
|
|
144
|
+
* from the proof's run.class. Absent ⇒ the attempt is charged, exactly as
|
|
145
|
+
* before ADR 0046 (which is also how a proof without a class reads). */
|
|
146
|
+
uncharged?: "environment" | "upstream";
|
|
131
147
|
}
|
|
132
148
|
|
|
133
149
|
export interface WorkProjection {
|
|
@@ -138,7 +154,13 @@ export interface WorkProjection {
|
|
|
138
154
|
next_action: WorkNextAction;
|
|
139
155
|
unresolved_dependencies: string[];
|
|
140
156
|
attempts: WorkAttempt[];
|
|
157
|
+
/** Every attempt started, charged or not. Drives the next attempt number and
|
|
158
|
+
* history ordering, so it counts uncharged attempts too. */
|
|
141
159
|
attempts_used: number;
|
|
160
|
+
/** Attempts that were informative about the work (ADR 0046). This — not
|
|
161
|
+
* attempts_used — is what max_attempts budgets, so an uncharged environment or
|
|
162
|
+
* upstream attempt does not consume the retry budget. */
|
|
163
|
+
charged_attempts: number;
|
|
142
164
|
attempts_remaining: number;
|
|
143
165
|
latest_run_id?: string;
|
|
144
166
|
approval_id?: string;
|
|
@@ -160,6 +182,7 @@ export interface CreateWorkItemInput {
|
|
|
160
182
|
acceptance?: string[];
|
|
161
183
|
dependencies?: string[];
|
|
162
184
|
maxAttempts?: number;
|
|
185
|
+
maxUnchargedAttempts?: number;
|
|
163
186
|
source?: WorkIntent["source"];
|
|
164
187
|
id?: string;
|
|
165
188
|
actor?: string;
|
|
@@ -188,6 +211,14 @@ export function createWorkItem(input: CreateWorkItemInput): WorkRecord {
|
|
|
188
211
|
if (!Number.isSafeInteger(maxAttempts) || maxAttempts < 1 || maxAttempts > 100) {
|
|
189
212
|
throw new Error("work maxAttempts must be an integer from 1 to 100");
|
|
190
213
|
}
|
|
214
|
+
const maxUnchargedAttempts = input.maxUnchargedAttempts ?? DEFAULT_MAX_UNCHARGED_ATTEMPTS;
|
|
215
|
+
if (
|
|
216
|
+
!Number.isSafeInteger(maxUnchargedAttempts) ||
|
|
217
|
+
maxUnchargedAttempts < 1 ||
|
|
218
|
+
maxUnchargedAttempts > 100
|
|
219
|
+
) {
|
|
220
|
+
throw new Error("work maxUnchargedAttempts must be an integer from 1 to 100");
|
|
221
|
+
}
|
|
191
222
|
const acceptance = (input.acceptance ?? []).map((value, index) =>
|
|
192
223
|
boundedString(value, `work acceptance[${index}]`, MAX_ACCEPTANCE_ITEM),
|
|
193
224
|
);
|
|
@@ -204,6 +235,7 @@ export function createWorkItem(input: CreateWorkItemInput): WorkRecord {
|
|
|
204
235
|
dependencies,
|
|
205
236
|
workflow: { path: workflowPath, sha256: workflowScriptDigest(workflowPath) },
|
|
206
237
|
max_attempts: maxAttempts,
|
|
238
|
+
max_uncharged_attempts: maxUnchargedAttempts,
|
|
207
239
|
source,
|
|
208
240
|
created_at: new Date().toISOString(),
|
|
209
241
|
};
|
|
@@ -375,13 +407,18 @@ function deriveWorkProjection(
|
|
|
375
407
|
const attempts = attemptEvents.map((event, index) =>
|
|
376
408
|
inspectAttempt(coordRoot, event, intent, attemptEvents[index - 1]?.run_id),
|
|
377
409
|
);
|
|
410
|
+
// Charged attempts — not the raw count — are what max_attempts budgets
|
|
411
|
+
// (ADR 0046). An uncharged environment/upstream attempt still increments
|
|
412
|
+
// attempts_used (ordering + next number) but not the budget.
|
|
413
|
+
const chargedAttempts = attempts.filter((attempt) => attempt.uncharged === undefined).length;
|
|
378
414
|
const base = {
|
|
379
415
|
id: intent.id,
|
|
380
416
|
title: intent.title,
|
|
381
417
|
unresolved_dependencies: [] as string[],
|
|
382
418
|
attempts,
|
|
383
419
|
attempts_used: attempts.length,
|
|
384
|
-
|
|
420
|
+
charged_attempts: chargedAttempts,
|
|
421
|
+
attempts_remaining: Math.max(0, intent.max_attempts - chargedAttempts),
|
|
385
422
|
latest_run_id: attempts.at(-1)?.run_id,
|
|
386
423
|
approval_id: attempts.at(-1)?.approval_id,
|
|
387
424
|
proof_path: attempts.at(-1)?.proof_path,
|
|
@@ -462,7 +499,7 @@ function deriveWorkProjection(
|
|
|
462
499
|
next_action: "review",
|
|
463
500
|
};
|
|
464
501
|
}
|
|
465
|
-
const attemptsRemaining = intent.max_attempts -
|
|
502
|
+
const attemptsRemaining = intent.max_attempts - chargedAttempts;
|
|
466
503
|
if (latest.status === "journal_unreadable") {
|
|
467
504
|
return {
|
|
468
505
|
...base,
|
|
@@ -473,6 +510,49 @@ function deriveWorkProjection(
|
|
|
473
510
|
next_action: attemptsRemaining > 0 ? "retry" : "none",
|
|
474
511
|
};
|
|
475
512
|
}
|
|
513
|
+
// Uncharged failures (ADR 0046) are handled before the ordinary work-failure
|
|
514
|
+
// path: they did not touch the work, so they neither spent the budget nor
|
|
515
|
+
// signal that retrying the work would help.
|
|
516
|
+
if (latest.status === "failed" && latest.uncharged === "environment") {
|
|
517
|
+
// A missing precondition. The operator chose to STOP the item immediately
|
|
518
|
+
// rather than retry an unchanged environment (ADR 0046); next_action "none"
|
|
519
|
+
// stops the supervisor. A human who fixes the environment can still force a
|
|
520
|
+
// retry — the attempt was uncharged, so the budget is intact.
|
|
521
|
+
return {
|
|
522
|
+
...base,
|
|
523
|
+
state: "blocked",
|
|
524
|
+
reason: environmentBlockedReason(latest),
|
|
525
|
+
next_action: "none",
|
|
526
|
+
};
|
|
527
|
+
}
|
|
528
|
+
if (latest.status === "failed" && latest.uncharged === "upstream") {
|
|
529
|
+
// Count trailing consecutive uncharged attempts in the current window. The
|
|
530
|
+
// bound is the only brake on an outage that never ends, so at the limit the
|
|
531
|
+
// item stops and names the outside service — distinct from work-blocked.
|
|
532
|
+
const maxUncharged = intent.max_uncharged_attempts ?? DEFAULT_MAX_UNCHARGED_ATTEMPTS;
|
|
533
|
+
let trailingUncharged = 0;
|
|
534
|
+
for (let index = currentAttempts.length - 1; index >= 0; index--) {
|
|
535
|
+
if (currentAttempts[index]!.uncharged === undefined) break;
|
|
536
|
+
trailingUncharged++;
|
|
537
|
+
}
|
|
538
|
+
if (trailingUncharged >= maxUncharged || attemptsRemaining <= 0) {
|
|
539
|
+
return {
|
|
540
|
+
...base,
|
|
541
|
+
state: "blocked",
|
|
542
|
+
reason:
|
|
543
|
+
trailingUncharged >= maxUncharged
|
|
544
|
+
? `blocked waiting on an outside service after ${trailingUncharged} consecutive uncharged attempt(s): ${upstreamReason(latest)}`
|
|
545
|
+
: `workflow attempt ${latest.number} was uncharged (${upstreamReason(latest)}) but the work attempt budget is exhausted`,
|
|
546
|
+
next_action: "none",
|
|
547
|
+
};
|
|
548
|
+
}
|
|
549
|
+
return {
|
|
550
|
+
...base,
|
|
551
|
+
state: "blocked",
|
|
552
|
+
reason: `workflow attempt ${latest.number} did not touch the work; an outside service refused: ${upstreamReason(latest)}`,
|
|
553
|
+
next_action: "retry",
|
|
554
|
+
};
|
|
555
|
+
}
|
|
476
556
|
return {
|
|
477
557
|
...base,
|
|
478
558
|
state: "blocked",
|
|
@@ -484,6 +564,33 @@ function deriveWorkProjection(
|
|
|
484
564
|
};
|
|
485
565
|
}
|
|
486
566
|
|
|
567
|
+
/** The failure reason for an uncharged attempt, read from the proof when
|
|
568
|
+
* present. Bounded so a verbose vendor transcript cannot bloat the projection
|
|
569
|
+
* reason (which is persisted verbatim into the reconciliation event). */
|
|
570
|
+
function unchargedProofError(attempt: WorkAttempt): string | undefined {
|
|
571
|
+
if (!attempt.proof_path) return undefined;
|
|
572
|
+
try {
|
|
573
|
+
const proof = parseObject(readFileSync(attempt.proof_path, "utf8"), "workflow proof");
|
|
574
|
+
const run = proof.run;
|
|
575
|
+
const error =
|
|
576
|
+
run && typeof run === "object" ? (run as Record<string, unknown>).error : undefined;
|
|
577
|
+
return typeof error === "string" && error.trim() ? boundedJournalError(error) : undefined;
|
|
578
|
+
} catch {
|
|
579
|
+
return undefined;
|
|
580
|
+
}
|
|
581
|
+
}
|
|
582
|
+
|
|
583
|
+
function environmentBlockedReason(attempt: WorkAttempt): string {
|
|
584
|
+
const detail = unchargedProofError(attempt);
|
|
585
|
+
return detail
|
|
586
|
+
? `workflow attempt ${attempt.number} could not start; a required precondition is missing: ${detail}`
|
|
587
|
+
: `workflow attempt ${attempt.number} could not start; a required precondition is missing`;
|
|
588
|
+
}
|
|
589
|
+
|
|
590
|
+
function upstreamReason(attempt: WorkAttempt): string {
|
|
591
|
+
return unchargedProofError(attempt) ?? "the vendor was reached and refused";
|
|
592
|
+
}
|
|
593
|
+
|
|
487
594
|
function inspectAttempt(
|
|
488
595
|
coordRoot: string,
|
|
489
596
|
event: WorkEvent,
|
|
@@ -547,6 +654,15 @@ function inspectAttempt(
|
|
|
547
654
|
proof.acceptance.summary.unknown === 0
|
|
548
655
|
? "succeeded"
|
|
549
656
|
: "failed";
|
|
657
|
+
// A failed attempt the run classified as uninformative about the work is
|
|
658
|
+
// uncharged (ADR 0046). Absent class ⇒ charged, as before. Read only for a
|
|
659
|
+
// failed attempt: a succeeded run never carries a class.
|
|
660
|
+
if (
|
|
661
|
+
attempt.status === "failed" &&
|
|
662
|
+
(proof.run.class === "environment" || proof.run.class === "upstream")
|
|
663
|
+
) {
|
|
664
|
+
attempt.uncharged = proof.run.class;
|
|
665
|
+
}
|
|
550
666
|
return attempt;
|
|
551
667
|
}
|
|
552
668
|
const journalPath = workflowJournalPath(coordRoot, runId);
|
|
@@ -882,7 +998,13 @@ function validateWorkIntent(intent: WorkIntent, workId: string): void {
|
|
|
882
998
|
!/^[a-f0-9]{64}$/.test(intent.workflow.sha256) ||
|
|
883
999
|
!Number.isSafeInteger(intent.max_attempts) ||
|
|
884
1000
|
intent.max_attempts < 1 ||
|
|
885
|
-
intent.max_attempts > 100
|
|
1001
|
+
intent.max_attempts > 100 ||
|
|
1002
|
+
// Optional for back-compat: absent on pre-ADR-0046 intents. When present it
|
|
1003
|
+
// must be a valid bound.
|
|
1004
|
+
(intent.max_uncharged_attempts !== undefined &&
|
|
1005
|
+
(!Number.isSafeInteger(intent.max_uncharged_attempts) ||
|
|
1006
|
+
intent.max_uncharged_attempts < 1 ||
|
|
1007
|
+
intent.max_uncharged_attempts > 100))
|
|
886
1008
|
) {
|
|
887
1009
|
throw new Error(`work intent ${workId} has an unsupported or mismatched schema`);
|
|
888
1010
|
}
|
|
@@ -911,6 +911,11 @@ async function executeWorkflow(
|
|
|
911
911
|
|
|
912
912
|
if (!last.ok) {
|
|
913
913
|
journal("agent.attempt_failed", { id, attempt, error: last.error });
|
|
914
|
+
// ADR 0046: an environment failure (the binary was absent) cannot be
|
|
915
|
+
// helped by retrying an unchanged environment, so stop the in-agent
|
|
916
|
+
// retry too — not just the outer attempt/replan budget. An upstream
|
|
917
|
+
// refusal keeps retrying here: the vendor may recover mid-loop.
|
|
918
|
+
if (last.class === "environment") break;
|
|
914
919
|
continue; // spawn-level failure: retry with the original prompt
|
|
915
920
|
}
|
|
916
921
|
if (!agentOpts.schema) {
|
|
@@ -982,6 +987,12 @@ async function executeWorkflow(
|
|
|
982
987
|
agentCostUsd > 0 || last?.costUsd !== undefined ? agentCostUsd : undefined;
|
|
983
988
|
agentProof.session_id = last?.sessionId;
|
|
984
989
|
agentProof.error = reason;
|
|
990
|
+
// Carry the spawn class (environment/upstream) onto the proof only when the
|
|
991
|
+
// final outcome was a spawn failure. A schema failure after a spawn that
|
|
992
|
+
// reached the model is a work failure — left unclassed (charged). An
|
|
993
|
+
// earlier attempt that transiently failed and then succeeded returned
|
|
994
|
+
// above, so this only fires when the agent genuinely failed.
|
|
995
|
+
if (last && !last.ok && last.class) agentProof.class = last.class;
|
|
985
996
|
journal("agent.failed", { id, error: reason });
|
|
986
997
|
throw new Error(`agent ${proofLabel}: ${reason}`);
|
|
987
998
|
} catch (error) {
|
|
@@ -25,6 +25,7 @@ export { runWorkflow, WorkflowParkedError, WorkflowRunError } from "./engine.ts"
|
|
|
25
25
|
export {
|
|
26
26
|
buildWorkflowProof,
|
|
27
27
|
createEvidenceRecord,
|
|
28
|
+
deriveRunFailureClass,
|
|
28
29
|
digestResult,
|
|
29
30
|
normalizeWorkflowMeta,
|
|
30
31
|
readWorkflowProof,
|
|
@@ -43,6 +44,7 @@ export {
|
|
|
43
44
|
workflowScriptDigest,
|
|
44
45
|
writeWorkflowRunManifest,
|
|
45
46
|
} from "./run-state.ts";
|
|
47
|
+
export { isUpstreamFailureText, vendorFailureText } from "./spawn-failure.ts";
|
|
46
48
|
export type {
|
|
47
49
|
AcceptanceCriterion,
|
|
48
50
|
AcceptanceResult,
|
|
@@ -59,6 +61,7 @@ export type {
|
|
|
59
61
|
ResultDigest,
|
|
60
62
|
RunReport,
|
|
61
63
|
Spawner,
|
|
64
|
+
SpawnFailureClass,
|
|
62
65
|
SpawnRequest,
|
|
63
66
|
SpawnResult,
|
|
64
67
|
StageSchema,
|
|
@@ -20,6 +20,7 @@ import type {
|
|
|
20
20
|
HarnessEvidenceCapability,
|
|
21
21
|
HarnessEvidenceCoverage,
|
|
22
22
|
ResultDigest,
|
|
23
|
+
SpawnFailureClass,
|
|
23
24
|
WorkflowAgentProof,
|
|
24
25
|
WorkflowAttemptContext,
|
|
25
26
|
WorkflowEvidenceInput,
|
|
@@ -218,6 +219,36 @@ export function digestResult(value: unknown, kind?: "text" | "json"): ResultDige
|
|
|
218
219
|
};
|
|
219
220
|
}
|
|
220
221
|
|
|
222
|
+
/**
|
|
223
|
+
* The run-level failure class (ADR 0046), derived from the agents rather than a
|
|
224
|
+
* single terminal throw so it survives a script's `parallel()` swallowing the
|
|
225
|
+
* rejection — a swallowed agent's proof is still recorded with its class.
|
|
226
|
+
*
|
|
227
|
+
* Rules, in order, all in service of "default to charging":
|
|
228
|
+
* 1. A succeeded run is never classed (there is nothing uncharged about it).
|
|
229
|
+
* 2. If ANY agent produced a result (succeeded or replayed from cache) the
|
|
230
|
+
* attempt was informative about the work — charge it. This also keeps a
|
|
231
|
+
* resumed run whose earlier segment did real work from being written off by
|
|
232
|
+
* a later environment failure.
|
|
233
|
+
* 3. Otherwise, among the failed agents, environment wins over upstream: a
|
|
234
|
+
* missing binary means nothing ran at all, and it is the operator-chosen
|
|
235
|
+
* hard stop.
|
|
236
|
+
* 4. Anything else is undefined ⇒ a charged work failure, exactly as today.
|
|
237
|
+
*/
|
|
238
|
+
export function deriveRunFailureClass(
|
|
239
|
+
status: "succeeded" | "failed",
|
|
240
|
+
agents: readonly Pick<WorkflowAgentProof, "status" | "class">[],
|
|
241
|
+
): SpawnFailureClass | undefined {
|
|
242
|
+
if (status !== "failed") return undefined;
|
|
243
|
+
if (agents.some((agent) => agent.status === "succeeded" || agent.status === "cached")) {
|
|
244
|
+
return undefined;
|
|
245
|
+
}
|
|
246
|
+
const failed = agents.filter((agent) => agent.status === "failed");
|
|
247
|
+
if (failed.some((agent) => agent.class === "environment")) return "environment";
|
|
248
|
+
if (failed.some((agent) => agent.class === "upstream")) return "upstream";
|
|
249
|
+
return undefined;
|
|
250
|
+
}
|
|
251
|
+
|
|
221
252
|
export function buildWorkflowProof(input: BuildWorkflowProofInput): WorkflowProof {
|
|
222
253
|
const acceptance = rollupAcceptance(input.meta.acceptance, input.evidence);
|
|
223
254
|
const repository = buildRepoEvidence(input.before, input.after);
|
|
@@ -231,6 +262,7 @@ export function buildWorkflowProof(input: BuildWorkflowProofInput): WorkflowProo
|
|
|
231
262
|
}));
|
|
232
263
|
const harnesses = buildHarnessCoverage(agents, input.harnessEvidence, input.harnessAttestations);
|
|
233
264
|
const unknowns = buildUnknowns(agents, harnesses, repository);
|
|
265
|
+
const runClass = deriveRunFailureClass(input.status, agents);
|
|
234
266
|
const journal = readFileSync(input.journalPath);
|
|
235
267
|
return {
|
|
236
268
|
schema_version: WORKFLOW_PROOF_SCHEMA_VERSION,
|
|
@@ -247,6 +279,7 @@ export function buildWorkflowProof(input: BuildWorkflowProofInput): WorkflowProo
|
|
|
247
279
|
objective: input.meta.objective,
|
|
248
280
|
error: clippedOptional(input.error, MAX_SUMMARY_CHARS),
|
|
249
281
|
result: input.result === undefined ? undefined : digestResult(input.result),
|
|
282
|
+
...(runClass ? { class: runClass } : {}),
|
|
250
283
|
},
|
|
251
284
|
acceptance,
|
|
252
285
|
agents,
|
|
@@ -326,6 +359,9 @@ export function readWorkflowProof(coordRoot: string, runId: string): WorkflowPro
|
|
|
326
359
|
if (
|
|
327
360
|
proof.schema_version !== WORKFLOW_PROOF_SCHEMA_VERSION ||
|
|
328
361
|
proof.run?.id !== runId ||
|
|
362
|
+
(proof.run.class !== undefined &&
|
|
363
|
+
proof.run.class !== "environment" &&
|
|
364
|
+
proof.run.class !== "upstream") ||
|
|
329
365
|
(proof.run.work_context !== undefined &&
|
|
330
366
|
(!proof.run.work_item_id ||
|
|
331
367
|
proof.run.work_context.id !== proof.run.work_item_id ||
|
|
@@ -24,7 +24,7 @@ import type { HarnessInvocation, HarnessRawResult } from "../harnesses/types.ts"
|
|
|
24
24
|
import { buildChildEnv } from "./child-env.ts";
|
|
25
25
|
import { notFoundError } from "./harnesses.ts";
|
|
26
26
|
import { resolveSandboxProjection } from "./sandbox-projection.ts";
|
|
27
|
-
import { vendorFailureText } from "./spawn-failure.ts";
|
|
27
|
+
import { isUpstreamFailureText, vendorFailureText } from "./spawn-failure.ts";
|
|
28
28
|
import type { Spawner, SpawnRequest, SpawnResult } from "./types.ts";
|
|
29
29
|
|
|
30
30
|
interface ClaudeEnvelope {
|
|
@@ -70,7 +70,20 @@ export function normalizeClaudeResult(raw: HarnessRawResult): SpawnResult {
|
|
|
70
70
|
error: `claude timed out after ${raw.durationMs}ms and was killed`,
|
|
71
71
|
};
|
|
72
72
|
}
|
|
73
|
+
// Structural environment signal: the binary was never there (spawned directly,
|
|
74
|
+
// so a missing binary surfaces as ENOENT). Uncharged and not retried.
|
|
75
|
+
if (raw.spawnErrno === "ENOENT") {
|
|
76
|
+
return {
|
|
77
|
+
ok: false,
|
|
78
|
+
text: "",
|
|
79
|
+
durationMs: raw.durationMs,
|
|
80
|
+
error: notFoundError("claude-code"),
|
|
81
|
+
class: "environment",
|
|
82
|
+
};
|
|
83
|
+
}
|
|
73
84
|
if (raw.exitCode === 127) {
|
|
85
|
+
// A bare 127 with no errno is a shell/vendor 127, indistinguishable from a
|
|
86
|
+
// legitimate one — charged as work rather than classed environment.
|
|
74
87
|
return {
|
|
75
88
|
ok: false,
|
|
76
89
|
text: "",
|
|
@@ -79,11 +92,13 @@ export function normalizeClaudeResult(raw: HarnessRawResult): SpawnResult {
|
|
|
79
92
|
};
|
|
80
93
|
}
|
|
81
94
|
if (raw.exitCode !== 0) {
|
|
95
|
+
const failureText = vendorFailureText(raw);
|
|
82
96
|
return {
|
|
83
97
|
ok: false,
|
|
84
98
|
text: "",
|
|
85
99
|
durationMs: raw.durationMs,
|
|
86
|
-
error: `claude exited ${raw.exitCode}: ${
|
|
100
|
+
error: `claude exited ${raw.exitCode}: ${failureText}`,
|
|
101
|
+
...(isUpstreamFailureText(failureText) ? { class: "upstream" as const } : {}),
|
|
87
102
|
};
|
|
88
103
|
}
|
|
89
104
|
|
|
@@ -100,6 +115,9 @@ export function normalizeClaudeResult(raw: HarnessRawResult): SpawnResult {
|
|
|
100
115
|
}
|
|
101
116
|
|
|
102
117
|
if (envelope.is_error) {
|
|
118
|
+
const envelopeError = `${envelope.subtype ?? ""} ${(envelope.errors ?? []).join("; ")} ${String(
|
|
119
|
+
envelope.result ?? "",
|
|
120
|
+
)}`;
|
|
103
121
|
return {
|
|
104
122
|
ok: false,
|
|
105
123
|
text: String(envelope.result ?? ""),
|
|
@@ -107,6 +125,7 @@ export function normalizeClaudeResult(raw: HarnessRawResult): SpawnResult {
|
|
|
107
125
|
costUsd: envelope.total_cost_usd,
|
|
108
126
|
durationMs: raw.durationMs,
|
|
109
127
|
error: `harness error (${envelope.subtype ?? "unknown"}): ${(envelope.errors ?? []).join("; ") || "see envelope"}`,
|
|
128
|
+
...(isUpstreamFailureText(envelopeError) ? { class: "upstream" as const } : {}),
|
|
110
129
|
};
|
|
111
130
|
}
|
|
112
131
|
|
|
@@ -25,7 +25,7 @@ import type { HarnessInvocation, HarnessRawResult } from "../harnesses/types.ts"
|
|
|
25
25
|
import { buildChildEnv } from "./child-env.ts";
|
|
26
26
|
import { notFoundError } from "./harnesses.ts";
|
|
27
27
|
import { resolveSandboxProjection } from "./sandbox-projection.ts";
|
|
28
|
-
import { vendorFailureText } from "./spawn-failure.ts";
|
|
28
|
+
import { isUpstreamFailureText, vendorFailureText } from "./spawn-failure.ts";
|
|
29
29
|
import type { Spawner, SpawnRequest, SpawnResult } from "./types.ts";
|
|
30
30
|
|
|
31
31
|
export function buildCodexInvocation(req: SpawnRequest, resultFile?: string): HarnessInvocation {
|
|
@@ -69,15 +69,30 @@ export function normalizeCodexResult(raw: HarnessRawResult): SpawnResult {
|
|
|
69
69
|
error: `codex timed out after ${raw.durationMs}ms and was killed`,
|
|
70
70
|
};
|
|
71
71
|
}
|
|
72
|
+
// Structural environment signal: the binary was never there (spawned directly,
|
|
73
|
+
// so a missing binary surfaces as ENOENT). Uncharged and not retried.
|
|
74
|
+
if (raw.spawnErrno === "ENOENT") {
|
|
75
|
+
return {
|
|
76
|
+
ok: false,
|
|
77
|
+
text: "",
|
|
78
|
+
durationMs: raw.durationMs,
|
|
79
|
+
error: notFoundError("codex"),
|
|
80
|
+
class: "environment",
|
|
81
|
+
};
|
|
82
|
+
}
|
|
72
83
|
if (raw.exitCode === 127) {
|
|
84
|
+
// A bare 127 with no errno is a shell/vendor 127, indistinguishable from a
|
|
85
|
+
// legitimate one — charged as work rather than classed environment.
|
|
73
86
|
return { ok: false, text: "", durationMs: raw.durationMs, error: notFoundError("codex") };
|
|
74
87
|
}
|
|
75
88
|
if (raw.exitCode !== 0) {
|
|
89
|
+
const failureText = vendorFailureText(raw);
|
|
76
90
|
return {
|
|
77
91
|
ok: false,
|
|
78
92
|
text: "",
|
|
79
93
|
durationMs: raw.durationMs,
|
|
80
|
-
error: `codex exited ${raw.exitCode}: ${
|
|
94
|
+
error: `codex exited ${raw.exitCode}: ${failureText}`,
|
|
95
|
+
...(isUpstreamFailureText(failureText) ? { class: "upstream" as const } : {}),
|
|
81
96
|
};
|
|
82
97
|
}
|
|
83
98
|
return {
|
|
@@ -25,7 +25,7 @@ import type { HarnessInvocation, HarnessRawResult } from "../harnesses/types.ts"
|
|
|
25
25
|
import { buildChildEnv } from "./child-env.ts";
|
|
26
26
|
import { notFoundError } from "./harnesses.ts";
|
|
27
27
|
import { resolveSandboxProjection } from "./sandbox-projection.ts";
|
|
28
|
-
import { vendorFailureText } from "./spawn-failure.ts";
|
|
28
|
+
import { isUpstreamFailureText, vendorFailureText } from "./spawn-failure.ts";
|
|
29
29
|
import type { Spawner, SpawnRequest, SpawnResult } from "./types.ts";
|
|
30
30
|
|
|
31
31
|
interface CursorEnvelope {
|
|
@@ -82,15 +82,30 @@ export function normalizeCursorResult(raw: HarnessRawResult): SpawnResult {
|
|
|
82
82
|
error: `cursor timed out after ${raw.durationMs}ms and was killed`,
|
|
83
83
|
};
|
|
84
84
|
}
|
|
85
|
+
// Structural environment signal: the binary was never there (spawned directly,
|
|
86
|
+
// so a missing binary surfaces as ENOENT). Uncharged and not retried.
|
|
87
|
+
if (raw.spawnErrno === "ENOENT") {
|
|
88
|
+
return {
|
|
89
|
+
ok: false,
|
|
90
|
+
text: "",
|
|
91
|
+
durationMs: raw.durationMs,
|
|
92
|
+
error: notFoundError("cursor"),
|
|
93
|
+
class: "environment",
|
|
94
|
+
};
|
|
95
|
+
}
|
|
85
96
|
if (raw.exitCode === 127) {
|
|
97
|
+
// A bare 127 with no errno is a shell/vendor 127, indistinguishable from a
|
|
98
|
+
// legitimate one — charged as work rather than classed environment.
|
|
86
99
|
return { ok: false, text: "", durationMs: raw.durationMs, error: notFoundError("cursor") };
|
|
87
100
|
}
|
|
88
101
|
if (raw.exitCode !== 0) {
|
|
102
|
+
const failureText = vendorFailureText(raw);
|
|
89
103
|
return {
|
|
90
104
|
ok: false,
|
|
91
105
|
text: "",
|
|
92
106
|
durationMs: raw.durationMs,
|
|
93
|
-
error: `cursor-agent exited ${raw.exitCode}: ${
|
|
107
|
+
error: `cursor-agent exited ${raw.exitCode}: ${failureText}`,
|
|
108
|
+
...(isUpstreamFailureText(failureText) ? { class: "upstream" as const } : {}),
|
|
94
109
|
};
|
|
95
110
|
}
|
|
96
111
|
|
|
@@ -102,6 +117,7 @@ export function normalizeCursorResult(raw: HarnessRawResult): SpawnResult {
|
|
|
102
117
|
sessionId: parsed.sessionId,
|
|
103
118
|
durationMs: raw.durationMs,
|
|
104
119
|
error: `cursor-agent reported is_error: ${parsed.text.slice(0, 300)}`,
|
|
120
|
+
...(isUpstreamFailureText(parsed.text) ? { class: "upstream" as const } : {}),
|
|
105
121
|
};
|
|
106
122
|
}
|
|
107
123
|
return {
|