harnery 0.29.0 → 0.30.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/commands/work.d.ts +2 -1
- package/dist/commands/work.d.ts.map +1 -1
- package/dist/commands/work.js +17 -3
- package/dist/core/harnesses/types.d.ts +5 -0
- package/dist/core/harnesses/types.d.ts.map +1 -1
- package/dist/core/supervisor/plan-read.js +7 -1
- package/dist/core/supervisor/plan-types.d.ts +10 -0
- package/dist/core/supervisor/plan-types.d.ts.map +1 -1
- package/dist/core/supervisor/planning.d.ts.map +1 -1
- package/dist/core/supervisor/planning.js +18 -0
- package/dist/core/supervisor/state.d.ts.map +1 -1
- package/dist/core/supervisor/state.js +65 -11
- package/dist/core/work/runner.d.ts.map +1 -1
- package/dist/core/work/runner.js +11 -2
- package/dist/core/work/state.d.ts +17 -0
- package/dist/core/work/state.d.ts.map +1 -1
- package/dist/core/work/state.js +102 -3
- package/dist/core/workflow/engine.js +13 -0
- package/dist/core/workflow/index.d.ts +3 -2
- package/dist/core/workflow/index.d.ts.map +1 -1
- package/dist/core/workflow/index.js +2 -1
- package/dist/core/workflow/proof.d.ts +18 -1
- package/dist/core/workflow/proof.d.ts.map +1 -1
- package/dist/core/workflow/proof.js +34 -0
- package/dist/core/workflow/spawn-claude.d.ts.map +1 -1
- package/dist/core/workflow/spawn-claude.js +19 -2
- package/dist/core/workflow/spawn-codex.d.ts.map +1 -1
- package/dist/core/workflow/spawn-codex.js +17 -2
- package/dist/core/workflow/spawn-cursor.d.ts.map +1 -1
- package/dist/core/workflow/spawn-cursor.js +18 -2
- package/dist/core/workflow/spawn-failure.d.ts +14 -0
- package/dist/core/workflow/spawn-failure.d.ts.map +1 -1
- package/dist/core/workflow/spawn-failure.js +34 -0
- package/dist/core/workflow/types.d.ts +26 -0
- package/dist/core/workflow/types.d.ts.map +1 -1
- package/dist/lib/exec.d.ts +7 -0
- package/dist/lib/exec.d.ts.map +1 -1
- package/dist/lib/exec.js +6 -3
- package/package.json +1 -1
- package/src/commands/work.ts +22 -3
- package/src/core/harnesses/types.ts +5 -0
- package/src/core/supervisor/plan-read.ts +11 -2
- package/src/core/supervisor/plan-types.ts +11 -0
- package/src/core/supervisor/planning.ts +21 -0
- package/src/core/supervisor/state.ts +68 -11
- package/src/core/work/runner.ts +11 -2
- package/src/core/work/state.ts +125 -3
- package/src/core/workflow/engine.ts +11 -0
- package/src/core/workflow/index.ts +3 -0
- package/src/core/workflow/proof.ts +36 -0
- package/src/core/workflow/spawn-claude.ts +21 -2
- package/src/core/workflow/spawn-codex.ts +17 -2
- package/src/core/workflow/spawn-cursor.ts +18 -2
- package/src/core/workflow/spawn-failure.ts +40 -0
- package/src/core/workflow/types.ts +27 -0
- package/src/lib/exec.ts +13 -3
package/src/core/work/state.ts
CHANGED
|
@@ -39,6 +39,12 @@ const MAX_EVENTS_BYTES = 4 * 1024 * 1024;
|
|
|
39
39
|
const MAX_EVENT_BYTES = 16 * 1024;
|
|
40
40
|
const MAX_EVENTS = 1_000;
|
|
41
41
|
const FOREIGN_LEASE_STALE_MS = 24 * 60 * 60 * 1_000;
|
|
42
|
+
/** Default ceiling on consecutive uncharged attempts (ADR 0046). Kept low: each
|
|
43
|
+
* one is a real vendor round-trip, so a handful gives a transient outage room
|
|
44
|
+
* to recover before the item stops and names the outside service. There is no
|
|
45
|
+
* backoff between them beyond the cadence of the supervisor and the vendor's own
|
|
46
|
+
* responses, which is why the bound stays small. */
|
|
47
|
+
const DEFAULT_MAX_UNCHARGED_ATTEMPTS = 3;
|
|
42
48
|
|
|
43
49
|
export type WorkState =
|
|
44
50
|
| "waiting"
|
|
@@ -71,6 +77,12 @@ export interface WorkIntent {
|
|
|
71
77
|
dependencies: string[];
|
|
72
78
|
workflow: { path: string; sha256: string };
|
|
73
79
|
max_attempts: number;
|
|
80
|
+
/** Ceiling on CONSECUTIVE uncharged attempts (ADR 0046), separate from
|
|
81
|
+
* max_attempts. An upstream outage produces uncharged attempt after uncharged
|
|
82
|
+
* attempt; without a bound it would retry forever. At the bound the item stops
|
|
83
|
+
* and reports it is blocked on an outside service. Optional for back-compat:
|
|
84
|
+
* an intent written before ADR 0046 has none and falls back to the default. */
|
|
85
|
+
max_uncharged_attempts?: number;
|
|
74
86
|
source?: { kind: "human" | "workflow" | "external"; ref?: string };
|
|
75
87
|
created_at: string;
|
|
76
88
|
}
|
|
@@ -128,6 +140,10 @@ export interface WorkAttempt {
|
|
|
128
140
|
approval_id?: string;
|
|
129
141
|
proof_path?: string;
|
|
130
142
|
journal_error?: string;
|
|
143
|
+
/** Why this failed attempt was uninformative about the work (ADR 0046), read
|
|
144
|
+
* from the proof's run.class. Absent ⇒ the attempt is charged, exactly as
|
|
145
|
+
* before ADR 0046 (which is also how a proof without a class reads). */
|
|
146
|
+
uncharged?: "environment" | "upstream";
|
|
131
147
|
}
|
|
132
148
|
|
|
133
149
|
export interface WorkProjection {
|
|
@@ -138,7 +154,13 @@ export interface WorkProjection {
|
|
|
138
154
|
next_action: WorkNextAction;
|
|
139
155
|
unresolved_dependencies: string[];
|
|
140
156
|
attempts: WorkAttempt[];
|
|
157
|
+
/** Every attempt started, charged or not. Drives the next attempt number and
|
|
158
|
+
* history ordering, so it counts uncharged attempts too. */
|
|
141
159
|
attempts_used: number;
|
|
160
|
+
/** Attempts that were informative about the work (ADR 0046). This — not
|
|
161
|
+
* attempts_used — is what max_attempts budgets, so an uncharged environment or
|
|
162
|
+
* upstream attempt does not consume the retry budget. */
|
|
163
|
+
charged_attempts: number;
|
|
142
164
|
attempts_remaining: number;
|
|
143
165
|
latest_run_id?: string;
|
|
144
166
|
approval_id?: string;
|
|
@@ -160,6 +182,7 @@ export interface CreateWorkItemInput {
|
|
|
160
182
|
acceptance?: string[];
|
|
161
183
|
dependencies?: string[];
|
|
162
184
|
maxAttempts?: number;
|
|
185
|
+
maxUnchargedAttempts?: number;
|
|
163
186
|
source?: WorkIntent["source"];
|
|
164
187
|
id?: string;
|
|
165
188
|
actor?: string;
|
|
@@ -188,6 +211,14 @@ export function createWorkItem(input: CreateWorkItemInput): WorkRecord {
|
|
|
188
211
|
if (!Number.isSafeInteger(maxAttempts) || maxAttempts < 1 || maxAttempts > 100) {
|
|
189
212
|
throw new Error("work maxAttempts must be an integer from 1 to 100");
|
|
190
213
|
}
|
|
214
|
+
const maxUnchargedAttempts = input.maxUnchargedAttempts ?? DEFAULT_MAX_UNCHARGED_ATTEMPTS;
|
|
215
|
+
if (
|
|
216
|
+
!Number.isSafeInteger(maxUnchargedAttempts) ||
|
|
217
|
+
maxUnchargedAttempts < 1 ||
|
|
218
|
+
maxUnchargedAttempts > 100
|
|
219
|
+
) {
|
|
220
|
+
throw new Error("work maxUnchargedAttempts must be an integer from 1 to 100");
|
|
221
|
+
}
|
|
191
222
|
const acceptance = (input.acceptance ?? []).map((value, index) =>
|
|
192
223
|
boundedString(value, `work acceptance[${index}]`, MAX_ACCEPTANCE_ITEM),
|
|
193
224
|
);
|
|
@@ -204,6 +235,7 @@ export function createWorkItem(input: CreateWorkItemInput): WorkRecord {
|
|
|
204
235
|
dependencies,
|
|
205
236
|
workflow: { path: workflowPath, sha256: workflowScriptDigest(workflowPath) },
|
|
206
237
|
max_attempts: maxAttempts,
|
|
238
|
+
max_uncharged_attempts: maxUnchargedAttempts,
|
|
207
239
|
source,
|
|
208
240
|
created_at: new Date().toISOString(),
|
|
209
241
|
};
|
|
@@ -375,13 +407,18 @@ function deriveWorkProjection(
|
|
|
375
407
|
const attempts = attemptEvents.map((event, index) =>
|
|
376
408
|
inspectAttempt(coordRoot, event, intent, attemptEvents[index - 1]?.run_id),
|
|
377
409
|
);
|
|
410
|
+
// Charged attempts — not the raw count — are what max_attempts budgets
|
|
411
|
+
// (ADR 0046). An uncharged environment/upstream attempt still increments
|
|
412
|
+
// attempts_used (ordering + next number) but not the budget.
|
|
413
|
+
const chargedAttempts = attempts.filter((attempt) => attempt.uncharged === undefined).length;
|
|
378
414
|
const base = {
|
|
379
415
|
id: intent.id,
|
|
380
416
|
title: intent.title,
|
|
381
417
|
unresolved_dependencies: [] as string[],
|
|
382
418
|
attempts,
|
|
383
419
|
attempts_used: attempts.length,
|
|
384
|
-
|
|
420
|
+
charged_attempts: chargedAttempts,
|
|
421
|
+
attempts_remaining: Math.max(0, intent.max_attempts - chargedAttempts),
|
|
385
422
|
latest_run_id: attempts.at(-1)?.run_id,
|
|
386
423
|
approval_id: attempts.at(-1)?.approval_id,
|
|
387
424
|
proof_path: attempts.at(-1)?.proof_path,
|
|
@@ -462,7 +499,7 @@ function deriveWorkProjection(
|
|
|
462
499
|
next_action: "review",
|
|
463
500
|
};
|
|
464
501
|
}
|
|
465
|
-
const attemptsRemaining = intent.max_attempts -
|
|
502
|
+
const attemptsRemaining = intent.max_attempts - chargedAttempts;
|
|
466
503
|
if (latest.status === "journal_unreadable") {
|
|
467
504
|
return {
|
|
468
505
|
...base,
|
|
@@ -473,6 +510,49 @@ function deriveWorkProjection(
|
|
|
473
510
|
next_action: attemptsRemaining > 0 ? "retry" : "none",
|
|
474
511
|
};
|
|
475
512
|
}
|
|
513
|
+
// Uncharged failures (ADR 0046) are handled before the ordinary work-failure
|
|
514
|
+
// path: they did not touch the work, so they neither spent the budget nor
|
|
515
|
+
// signal that retrying the work would help.
|
|
516
|
+
if (latest.status === "failed" && latest.uncharged === "environment") {
|
|
517
|
+
// A missing precondition. The operator chose to STOP the item immediately
|
|
518
|
+
// rather than retry an unchanged environment (ADR 0046); next_action "none"
|
|
519
|
+
// stops the supervisor. A human who fixes the environment can still force a
|
|
520
|
+
// retry — the attempt was uncharged, so the budget is intact.
|
|
521
|
+
return {
|
|
522
|
+
...base,
|
|
523
|
+
state: "blocked",
|
|
524
|
+
reason: environmentBlockedReason(latest),
|
|
525
|
+
next_action: "none",
|
|
526
|
+
};
|
|
527
|
+
}
|
|
528
|
+
if (latest.status === "failed" && latest.uncharged === "upstream") {
|
|
529
|
+
// Count trailing consecutive uncharged attempts in the current window. The
|
|
530
|
+
// bound is the only brake on an outage that never ends, so at the limit the
|
|
531
|
+
// item stops and names the outside service — distinct from work-blocked.
|
|
532
|
+
const maxUncharged = intent.max_uncharged_attempts ?? DEFAULT_MAX_UNCHARGED_ATTEMPTS;
|
|
533
|
+
let trailingUncharged = 0;
|
|
534
|
+
for (let index = currentAttempts.length - 1; index >= 0; index--) {
|
|
535
|
+
if (currentAttempts[index]!.uncharged === undefined) break;
|
|
536
|
+
trailingUncharged++;
|
|
537
|
+
}
|
|
538
|
+
if (trailingUncharged >= maxUncharged || attemptsRemaining <= 0) {
|
|
539
|
+
return {
|
|
540
|
+
...base,
|
|
541
|
+
state: "blocked",
|
|
542
|
+
reason:
|
|
543
|
+
trailingUncharged >= maxUncharged
|
|
544
|
+
? `blocked waiting on an outside service after ${trailingUncharged} consecutive uncharged attempt(s): ${upstreamReason(latest)}`
|
|
545
|
+
: `workflow attempt ${latest.number} was uncharged (${upstreamReason(latest)}) but the work attempt budget is exhausted`,
|
|
546
|
+
next_action: "none",
|
|
547
|
+
};
|
|
548
|
+
}
|
|
549
|
+
return {
|
|
550
|
+
...base,
|
|
551
|
+
state: "blocked",
|
|
552
|
+
reason: `workflow attempt ${latest.number} did not touch the work; an outside service refused: ${upstreamReason(latest)}`,
|
|
553
|
+
next_action: "retry",
|
|
554
|
+
};
|
|
555
|
+
}
|
|
476
556
|
return {
|
|
477
557
|
...base,
|
|
478
558
|
state: "blocked",
|
|
@@ -484,6 +564,33 @@ function deriveWorkProjection(
|
|
|
484
564
|
};
|
|
485
565
|
}
|
|
486
566
|
|
|
567
|
+
/** The failure reason for an uncharged attempt, read from the proof when
|
|
568
|
+
* present. Bounded so a verbose vendor transcript cannot bloat the projection
|
|
569
|
+
* reason (which is persisted verbatim into the reconciliation event). */
|
|
570
|
+
function unchargedProofError(attempt: WorkAttempt): string | undefined {
|
|
571
|
+
if (!attempt.proof_path) return undefined;
|
|
572
|
+
try {
|
|
573
|
+
const proof = parseObject(readFileSync(attempt.proof_path, "utf8"), "workflow proof");
|
|
574
|
+
const run = proof.run;
|
|
575
|
+
const error =
|
|
576
|
+
run && typeof run === "object" ? (run as Record<string, unknown>).error : undefined;
|
|
577
|
+
return typeof error === "string" && error.trim() ? boundedJournalError(error) : undefined;
|
|
578
|
+
} catch {
|
|
579
|
+
return undefined;
|
|
580
|
+
}
|
|
581
|
+
}
|
|
582
|
+
|
|
583
|
+
function environmentBlockedReason(attempt: WorkAttempt): string {
|
|
584
|
+
const detail = unchargedProofError(attempt);
|
|
585
|
+
return detail
|
|
586
|
+
? `workflow attempt ${attempt.number} could not start; a required precondition is missing: ${detail}`
|
|
587
|
+
: `workflow attempt ${attempt.number} could not start; a required precondition is missing`;
|
|
588
|
+
}
|
|
589
|
+
|
|
590
|
+
function upstreamReason(attempt: WorkAttempt): string {
|
|
591
|
+
return unchargedProofError(attempt) ?? "the vendor was reached and refused";
|
|
592
|
+
}
|
|
593
|
+
|
|
487
594
|
function inspectAttempt(
|
|
488
595
|
coordRoot: string,
|
|
489
596
|
event: WorkEvent,
|
|
@@ -547,6 +654,15 @@ function inspectAttempt(
|
|
|
547
654
|
proof.acceptance.summary.unknown === 0
|
|
548
655
|
? "succeeded"
|
|
549
656
|
: "failed";
|
|
657
|
+
// A failed attempt the run classified as uninformative about the work is
|
|
658
|
+
// uncharged (ADR 0046). Absent class ⇒ charged, as before. Read only for a
|
|
659
|
+
// failed attempt: a succeeded run never carries a class.
|
|
660
|
+
if (
|
|
661
|
+
attempt.status === "failed" &&
|
|
662
|
+
(proof.run.class === "environment" || proof.run.class === "upstream")
|
|
663
|
+
) {
|
|
664
|
+
attempt.uncharged = proof.run.class;
|
|
665
|
+
}
|
|
550
666
|
return attempt;
|
|
551
667
|
}
|
|
552
668
|
const journalPath = workflowJournalPath(coordRoot, runId);
|
|
@@ -882,7 +998,13 @@ function validateWorkIntent(intent: WorkIntent, workId: string): void {
|
|
|
882
998
|
!/^[a-f0-9]{64}$/.test(intent.workflow.sha256) ||
|
|
883
999
|
!Number.isSafeInteger(intent.max_attempts) ||
|
|
884
1000
|
intent.max_attempts < 1 ||
|
|
885
|
-
intent.max_attempts > 100
|
|
1001
|
+
intent.max_attempts > 100 ||
|
|
1002
|
+
// Optional for back-compat: absent on pre-ADR-0046 intents. When present it
|
|
1003
|
+
// must be a valid bound.
|
|
1004
|
+
(intent.max_uncharged_attempts !== undefined &&
|
|
1005
|
+
(!Number.isSafeInteger(intent.max_uncharged_attempts) ||
|
|
1006
|
+
intent.max_uncharged_attempts < 1 ||
|
|
1007
|
+
intent.max_uncharged_attempts > 100))
|
|
886
1008
|
) {
|
|
887
1009
|
throw new Error(`work intent ${workId} has an unsupported or mismatched schema`);
|
|
888
1010
|
}
|
|
@@ -911,6 +911,11 @@ async function executeWorkflow(
|
|
|
911
911
|
|
|
912
912
|
if (!last.ok) {
|
|
913
913
|
journal("agent.attempt_failed", { id, attempt, error: last.error });
|
|
914
|
+
// ADR 0046: an environment failure (the binary was absent) cannot be
|
|
915
|
+
// helped by retrying an unchanged environment, so stop the in-agent
|
|
916
|
+
// retry too — not just the outer attempt/replan budget. An upstream
|
|
917
|
+
// refusal keeps retrying here: the vendor may recover mid-loop.
|
|
918
|
+
if (last.class === "environment") break;
|
|
914
919
|
continue; // spawn-level failure: retry with the original prompt
|
|
915
920
|
}
|
|
916
921
|
if (!agentOpts.schema) {
|
|
@@ -982,6 +987,12 @@ async function executeWorkflow(
|
|
|
982
987
|
agentCostUsd > 0 || last?.costUsd !== undefined ? agentCostUsd : undefined;
|
|
983
988
|
agentProof.session_id = last?.sessionId;
|
|
984
989
|
agentProof.error = reason;
|
|
990
|
+
// Carry the spawn class (environment/upstream) onto the proof only when the
|
|
991
|
+
// final outcome was a spawn failure. A schema failure after a spawn that
|
|
992
|
+
// reached the model is a work failure — left unclassed (charged). An
|
|
993
|
+
// earlier attempt that transiently failed and then succeeded returned
|
|
994
|
+
// above, so this only fires when the agent genuinely failed.
|
|
995
|
+
if (last && !last.ok && last.class) agentProof.class = last.class;
|
|
985
996
|
journal("agent.failed", { id, error: reason });
|
|
986
997
|
throw new Error(`agent ${proofLabel}: ${reason}`);
|
|
987
998
|
} catch (error) {
|
|
@@ -25,6 +25,7 @@ export { runWorkflow, WorkflowParkedError, WorkflowRunError } from "./engine.ts"
|
|
|
25
25
|
export {
|
|
26
26
|
buildWorkflowProof,
|
|
27
27
|
createEvidenceRecord,
|
|
28
|
+
deriveRunFailureClass,
|
|
28
29
|
digestResult,
|
|
29
30
|
normalizeWorkflowMeta,
|
|
30
31
|
readWorkflowProof,
|
|
@@ -43,6 +44,7 @@ export {
|
|
|
43
44
|
workflowScriptDigest,
|
|
44
45
|
writeWorkflowRunManifest,
|
|
45
46
|
} from "./run-state.ts";
|
|
47
|
+
export { isUpstreamFailureText, vendorFailureText } from "./spawn-failure.ts";
|
|
46
48
|
export type {
|
|
47
49
|
AcceptanceCriterion,
|
|
48
50
|
AcceptanceResult,
|
|
@@ -59,6 +61,7 @@ export type {
|
|
|
59
61
|
ResultDigest,
|
|
60
62
|
RunReport,
|
|
61
63
|
Spawner,
|
|
64
|
+
SpawnFailureClass,
|
|
62
65
|
SpawnRequest,
|
|
63
66
|
SpawnResult,
|
|
64
67
|
StageSchema,
|
|
@@ -20,6 +20,7 @@ import type {
|
|
|
20
20
|
HarnessEvidenceCapability,
|
|
21
21
|
HarnessEvidenceCoverage,
|
|
22
22
|
ResultDigest,
|
|
23
|
+
SpawnFailureClass,
|
|
23
24
|
WorkflowAgentProof,
|
|
24
25
|
WorkflowAttemptContext,
|
|
25
26
|
WorkflowEvidenceInput,
|
|
@@ -218,6 +219,36 @@ export function digestResult(value: unknown, kind?: "text" | "json"): ResultDige
|
|
|
218
219
|
};
|
|
219
220
|
}
|
|
220
221
|
|
|
222
|
+
/**
|
|
223
|
+
* The run-level failure class (ADR 0046), derived from the agents rather than a
|
|
224
|
+
* single terminal throw so it survives a script's `parallel()` swallowing the
|
|
225
|
+
* rejection — a swallowed agent's proof is still recorded with its class.
|
|
226
|
+
*
|
|
227
|
+
* Rules, in order, all in service of "default to charging":
|
|
228
|
+
* 1. A succeeded run is never classed (there is nothing uncharged about it).
|
|
229
|
+
* 2. If ANY agent produced a result (succeeded or replayed from cache) the
|
|
230
|
+
* attempt was informative about the work — charge it. This also keeps a
|
|
231
|
+
* resumed run whose earlier segment did real work from being written off by
|
|
232
|
+
* a later environment failure.
|
|
233
|
+
* 3. Otherwise, among the failed agents, environment wins over upstream: a
|
|
234
|
+
* missing binary means nothing ran at all, and it is the operator-chosen
|
|
235
|
+
* hard stop.
|
|
236
|
+
* 4. Anything else is undefined ⇒ a charged work failure, exactly as today.
|
|
237
|
+
*/
|
|
238
|
+
export function deriveRunFailureClass(
|
|
239
|
+
status: "succeeded" | "failed",
|
|
240
|
+
agents: readonly Pick<WorkflowAgentProof, "status" | "class">[],
|
|
241
|
+
): SpawnFailureClass | undefined {
|
|
242
|
+
if (status !== "failed") return undefined;
|
|
243
|
+
if (agents.some((agent) => agent.status === "succeeded" || agent.status === "cached")) {
|
|
244
|
+
return undefined;
|
|
245
|
+
}
|
|
246
|
+
const failed = agents.filter((agent) => agent.status === "failed");
|
|
247
|
+
if (failed.some((agent) => agent.class === "environment")) return "environment";
|
|
248
|
+
if (failed.some((agent) => agent.class === "upstream")) return "upstream";
|
|
249
|
+
return undefined;
|
|
250
|
+
}
|
|
251
|
+
|
|
221
252
|
export function buildWorkflowProof(input: BuildWorkflowProofInput): WorkflowProof {
|
|
222
253
|
const acceptance = rollupAcceptance(input.meta.acceptance, input.evidence);
|
|
223
254
|
const repository = buildRepoEvidence(input.before, input.after);
|
|
@@ -231,6 +262,7 @@ export function buildWorkflowProof(input: BuildWorkflowProofInput): WorkflowProo
|
|
|
231
262
|
}));
|
|
232
263
|
const harnesses = buildHarnessCoverage(agents, input.harnessEvidence, input.harnessAttestations);
|
|
233
264
|
const unknowns = buildUnknowns(agents, harnesses, repository);
|
|
265
|
+
const runClass = deriveRunFailureClass(input.status, agents);
|
|
234
266
|
const journal = readFileSync(input.journalPath);
|
|
235
267
|
return {
|
|
236
268
|
schema_version: WORKFLOW_PROOF_SCHEMA_VERSION,
|
|
@@ -247,6 +279,7 @@ export function buildWorkflowProof(input: BuildWorkflowProofInput): WorkflowProo
|
|
|
247
279
|
objective: input.meta.objective,
|
|
248
280
|
error: clippedOptional(input.error, MAX_SUMMARY_CHARS),
|
|
249
281
|
result: input.result === undefined ? undefined : digestResult(input.result),
|
|
282
|
+
...(runClass ? { class: runClass } : {}),
|
|
250
283
|
},
|
|
251
284
|
acceptance,
|
|
252
285
|
agents,
|
|
@@ -326,6 +359,9 @@ export function readWorkflowProof(coordRoot: string, runId: string): WorkflowPro
|
|
|
326
359
|
if (
|
|
327
360
|
proof.schema_version !== WORKFLOW_PROOF_SCHEMA_VERSION ||
|
|
328
361
|
proof.run?.id !== runId ||
|
|
362
|
+
(proof.run.class !== undefined &&
|
|
363
|
+
proof.run.class !== "environment" &&
|
|
364
|
+
proof.run.class !== "upstream") ||
|
|
329
365
|
(proof.run.work_context !== undefined &&
|
|
330
366
|
(!proof.run.work_item_id ||
|
|
331
367
|
proof.run.work_context.id !== proof.run.work_item_id ||
|
|
@@ -24,7 +24,7 @@ import type { HarnessInvocation, HarnessRawResult } from "../harnesses/types.ts"
|
|
|
24
24
|
import { buildChildEnv } from "./child-env.ts";
|
|
25
25
|
import { notFoundError } from "./harnesses.ts";
|
|
26
26
|
import { resolveSandboxProjection } from "./sandbox-projection.ts";
|
|
27
|
-
import { vendorFailureText } from "./spawn-failure.ts";
|
|
27
|
+
import { isUpstreamFailureText, vendorFailureText } from "./spawn-failure.ts";
|
|
28
28
|
import type { Spawner, SpawnRequest, SpawnResult } from "./types.ts";
|
|
29
29
|
|
|
30
30
|
interface ClaudeEnvelope {
|
|
@@ -70,7 +70,20 @@ export function normalizeClaudeResult(raw: HarnessRawResult): SpawnResult {
|
|
|
70
70
|
error: `claude timed out after ${raw.durationMs}ms and was killed`,
|
|
71
71
|
};
|
|
72
72
|
}
|
|
73
|
+
// Structural environment signal: the binary was never there (spawned directly,
|
|
74
|
+
// so a missing binary surfaces as ENOENT). Uncharged and not retried.
|
|
75
|
+
if (raw.spawnErrno === "ENOENT") {
|
|
76
|
+
return {
|
|
77
|
+
ok: false,
|
|
78
|
+
text: "",
|
|
79
|
+
durationMs: raw.durationMs,
|
|
80
|
+
error: notFoundError("claude-code"),
|
|
81
|
+
class: "environment",
|
|
82
|
+
};
|
|
83
|
+
}
|
|
73
84
|
if (raw.exitCode === 127) {
|
|
85
|
+
// A bare 127 with no errno is a shell/vendor 127, indistinguishable from a
|
|
86
|
+
// legitimate one — charged as work rather than classed environment.
|
|
74
87
|
return {
|
|
75
88
|
ok: false,
|
|
76
89
|
text: "",
|
|
@@ -79,11 +92,13 @@ export function normalizeClaudeResult(raw: HarnessRawResult): SpawnResult {
|
|
|
79
92
|
};
|
|
80
93
|
}
|
|
81
94
|
if (raw.exitCode !== 0) {
|
|
95
|
+
const failureText = vendorFailureText(raw);
|
|
82
96
|
return {
|
|
83
97
|
ok: false,
|
|
84
98
|
text: "",
|
|
85
99
|
durationMs: raw.durationMs,
|
|
86
|
-
error: `claude exited ${raw.exitCode}: ${
|
|
100
|
+
error: `claude exited ${raw.exitCode}: ${failureText}`,
|
|
101
|
+
...(isUpstreamFailureText(failureText) ? { class: "upstream" as const } : {}),
|
|
87
102
|
};
|
|
88
103
|
}
|
|
89
104
|
|
|
@@ -100,6 +115,9 @@ export function normalizeClaudeResult(raw: HarnessRawResult): SpawnResult {
|
|
|
100
115
|
}
|
|
101
116
|
|
|
102
117
|
if (envelope.is_error) {
|
|
118
|
+
const envelopeError = `${envelope.subtype ?? ""} ${(envelope.errors ?? []).join("; ")} ${String(
|
|
119
|
+
envelope.result ?? "",
|
|
120
|
+
)}`;
|
|
103
121
|
return {
|
|
104
122
|
ok: false,
|
|
105
123
|
text: String(envelope.result ?? ""),
|
|
@@ -107,6 +125,7 @@ export function normalizeClaudeResult(raw: HarnessRawResult): SpawnResult {
|
|
|
107
125
|
costUsd: envelope.total_cost_usd,
|
|
108
126
|
durationMs: raw.durationMs,
|
|
109
127
|
error: `harness error (${envelope.subtype ?? "unknown"}): ${(envelope.errors ?? []).join("; ") || "see envelope"}`,
|
|
128
|
+
...(isUpstreamFailureText(envelopeError) ? { class: "upstream" as const } : {}),
|
|
110
129
|
};
|
|
111
130
|
}
|
|
112
131
|
|
|
@@ -25,7 +25,7 @@ import type { HarnessInvocation, HarnessRawResult } from "../harnesses/types.ts"
|
|
|
25
25
|
import { buildChildEnv } from "./child-env.ts";
|
|
26
26
|
import { notFoundError } from "./harnesses.ts";
|
|
27
27
|
import { resolveSandboxProjection } from "./sandbox-projection.ts";
|
|
28
|
-
import { vendorFailureText } from "./spawn-failure.ts";
|
|
28
|
+
import { isUpstreamFailureText, vendorFailureText } from "./spawn-failure.ts";
|
|
29
29
|
import type { Spawner, SpawnRequest, SpawnResult } from "./types.ts";
|
|
30
30
|
|
|
31
31
|
export function buildCodexInvocation(req: SpawnRequest, resultFile?: string): HarnessInvocation {
|
|
@@ -69,15 +69,30 @@ export function normalizeCodexResult(raw: HarnessRawResult): SpawnResult {
|
|
|
69
69
|
error: `codex timed out after ${raw.durationMs}ms and was killed`,
|
|
70
70
|
};
|
|
71
71
|
}
|
|
72
|
+
// Structural environment signal: the binary was never there (spawned directly,
|
|
73
|
+
// so a missing binary surfaces as ENOENT). Uncharged and not retried.
|
|
74
|
+
if (raw.spawnErrno === "ENOENT") {
|
|
75
|
+
return {
|
|
76
|
+
ok: false,
|
|
77
|
+
text: "",
|
|
78
|
+
durationMs: raw.durationMs,
|
|
79
|
+
error: notFoundError("codex"),
|
|
80
|
+
class: "environment",
|
|
81
|
+
};
|
|
82
|
+
}
|
|
72
83
|
if (raw.exitCode === 127) {
|
|
84
|
+
// A bare 127 with no errno is a shell/vendor 127, indistinguishable from a
|
|
85
|
+
// legitimate one — charged as work rather than classed environment.
|
|
73
86
|
return { ok: false, text: "", durationMs: raw.durationMs, error: notFoundError("codex") };
|
|
74
87
|
}
|
|
75
88
|
if (raw.exitCode !== 0) {
|
|
89
|
+
const failureText = vendorFailureText(raw);
|
|
76
90
|
return {
|
|
77
91
|
ok: false,
|
|
78
92
|
text: "",
|
|
79
93
|
durationMs: raw.durationMs,
|
|
80
|
-
error: `codex exited ${raw.exitCode}: ${
|
|
94
|
+
error: `codex exited ${raw.exitCode}: ${failureText}`,
|
|
95
|
+
...(isUpstreamFailureText(failureText) ? { class: "upstream" as const } : {}),
|
|
81
96
|
};
|
|
82
97
|
}
|
|
83
98
|
return {
|
|
@@ -25,7 +25,7 @@ import type { HarnessInvocation, HarnessRawResult } from "../harnesses/types.ts"
|
|
|
25
25
|
import { buildChildEnv } from "./child-env.ts";
|
|
26
26
|
import { notFoundError } from "./harnesses.ts";
|
|
27
27
|
import { resolveSandboxProjection } from "./sandbox-projection.ts";
|
|
28
|
-
import { vendorFailureText } from "./spawn-failure.ts";
|
|
28
|
+
import { isUpstreamFailureText, vendorFailureText } from "./spawn-failure.ts";
|
|
29
29
|
import type { Spawner, SpawnRequest, SpawnResult } from "./types.ts";
|
|
30
30
|
|
|
31
31
|
interface CursorEnvelope {
|
|
@@ -82,15 +82,30 @@ export function normalizeCursorResult(raw: HarnessRawResult): SpawnResult {
|
|
|
82
82
|
error: `cursor timed out after ${raw.durationMs}ms and was killed`,
|
|
83
83
|
};
|
|
84
84
|
}
|
|
85
|
+
// Structural environment signal: the binary was never there (spawned directly,
|
|
86
|
+
// so a missing binary surfaces as ENOENT). Uncharged and not retried.
|
|
87
|
+
if (raw.spawnErrno === "ENOENT") {
|
|
88
|
+
return {
|
|
89
|
+
ok: false,
|
|
90
|
+
text: "",
|
|
91
|
+
durationMs: raw.durationMs,
|
|
92
|
+
error: notFoundError("cursor"),
|
|
93
|
+
class: "environment",
|
|
94
|
+
};
|
|
95
|
+
}
|
|
85
96
|
if (raw.exitCode === 127) {
|
|
97
|
+
// A bare 127 with no errno is a shell/vendor 127, indistinguishable from a
|
|
98
|
+
// legitimate one — charged as work rather than classed environment.
|
|
86
99
|
return { ok: false, text: "", durationMs: raw.durationMs, error: notFoundError("cursor") };
|
|
87
100
|
}
|
|
88
101
|
if (raw.exitCode !== 0) {
|
|
102
|
+
const failureText = vendorFailureText(raw);
|
|
89
103
|
return {
|
|
90
104
|
ok: false,
|
|
91
105
|
text: "",
|
|
92
106
|
durationMs: raw.durationMs,
|
|
93
|
-
error: `cursor-agent exited ${raw.exitCode}: ${
|
|
107
|
+
error: `cursor-agent exited ${raw.exitCode}: ${failureText}`,
|
|
108
|
+
...(isUpstreamFailureText(failureText) ? { class: "upstream" as const } : {}),
|
|
94
109
|
};
|
|
95
110
|
}
|
|
96
111
|
|
|
@@ -102,6 +117,7 @@ export function normalizeCursorResult(raw: HarnessRawResult): SpawnResult {
|
|
|
102
117
|
sessionId: parsed.sessionId,
|
|
103
118
|
durationMs: raw.durationMs,
|
|
104
119
|
error: `cursor-agent reported is_error: ${parsed.text.slice(0, 300)}`,
|
|
120
|
+
...(isUpstreamFailureText(parsed.text) ? { class: "upstream" as const } : {}),
|
|
105
121
|
};
|
|
106
122
|
}
|
|
107
123
|
return {
|
|
@@ -26,3 +26,43 @@ export function vendorFailureText(
|
|
|
26
26
|
const combined = parts.join("\n");
|
|
27
27
|
return combined.length > maxChars ? `…${combined.slice(-maxChars)}` : combined;
|
|
28
28
|
}
|
|
29
|
+
|
|
30
|
+
// A 429 or 5xx status, but only where it sits in HTTP-status CONTEXT — not any
|
|
31
|
+
// digits that merely happen to fall in that range. A bare-number match treats a
|
|
32
|
+
// line number ("at line 500"), an item count ("got 429 items"), or a duration
|
|
33
|
+
// ("after 502 seconds") as an upstream refusal, which wrongly withholds a charge
|
|
34
|
+
// for a failure that was entirely our code. Three recognized shapes:
|
|
35
|
+
// 1. after an "HTTP" label: "HTTP 429", "HTTP/1.1 503 Service Unavailable"
|
|
36
|
+
// 2. after a "status" label: "status 500", "status: 429", "status code 502"
|
|
37
|
+
// 3. a bare status line: "503 Service Unavailable" (code at line start,
|
|
38
|
+
// then a reason phrase) — the classic "<code> <Reason-Phrase>" shape.
|
|
39
|
+
// The trailing `(?![0-9])` and the label/line-start anchors also keep a status
|
|
40
|
+
// embedded in a longer number ("4295"/"5031") out. All 5xx are server-side; 429
|
|
41
|
+
// is the only 4xx that means "reached and refused, retry later".
|
|
42
|
+
const STATUS = "(?:429|5[0-9]{2})";
|
|
43
|
+
const UPSTREAM_STATUS = new RegExp(
|
|
44
|
+
`(?:\\bHTTP\\b[/0-9. ]*|\\bstatus(?:\\s*code)?\\b[\\s:=]*)${STATUS}(?![0-9])` +
|
|
45
|
+
`|(?:^|\\n)\\s*${STATUS}\\s+[A-Za-z]`,
|
|
46
|
+
"i",
|
|
47
|
+
);
|
|
48
|
+
// Explicit vendor wording for the same conditions when a numeric status is
|
|
49
|
+
// absent. Deliberately short — see isUpstreamFailureText.
|
|
50
|
+
const UPSTREAM_PHRASES =
|
|
51
|
+
/circuit[ _-]?open|service unavailable|too many requests|rate[ _-]?limit|overloaded|bad gateway|gateway time-?out|internal server error/i;
|
|
52
|
+
|
|
53
|
+
/**
|
|
54
|
+
* Whether failure text names an UPSTREAM refusal — the vendor was reached and
|
|
55
|
+
* refused (a 5xx/429 status or circuit-open wording), as opposed to a work
|
|
56
|
+
* failure (the model ran and produced a wrong or incomplete result).
|
|
57
|
+
*
|
|
58
|
+
* There is no structural upstream signal the way ENOENT structurally marks a
|
|
59
|
+
* missing binary, so this is the one text match in the classifier. It is kept
|
|
60
|
+
* SHORT and documented on purpose: a per-vendor regex zoo would rot, and a
|
|
61
|
+
* false positive here wrongly withholds a charge, so both the phrase list and
|
|
62
|
+
* the numeric match stay tight — the number must sit in HTTP-status context, not
|
|
63
|
+
* merely fall in the 429/5xx range. Anything it does not match falls through to
|
|
64
|
+
* a charged work failure.
|
|
65
|
+
*/
|
|
66
|
+
export function isUpstreamFailureText(text: string): boolean {
|
|
67
|
+
return UPSTREAM_STATUS.test(text) || UPSTREAM_PHRASES.test(text);
|
|
68
|
+
}
|
|
@@ -99,6 +99,11 @@ export interface WorkflowAgentProof {
|
|
|
99
99
|
session_id?: string;
|
|
100
100
|
result?: ResultDigest;
|
|
101
101
|
error?: string;
|
|
102
|
+
/** Set on a failed agent whose spawn was classified uninformative about the
|
|
103
|
+
* work (ADR 0046): environment (binary absent) or upstream (vendor refused).
|
|
104
|
+
* Absent ⇒ a work failure. Recorded even when a script's parallel() swallows
|
|
105
|
+
* the rejection, so the run-level class can still be derived from proof. */
|
|
106
|
+
class?: SpawnFailureClass;
|
|
102
107
|
}
|
|
103
108
|
|
|
104
109
|
export interface WorkflowRepoSnapshot {
|
|
@@ -186,6 +191,12 @@ export interface WorkflowProof {
|
|
|
186
191
|
objective?: string;
|
|
187
192
|
error?: string;
|
|
188
193
|
result?: ResultDigest;
|
|
194
|
+
/** Set on a failed run that was uninformative about the work (ADR 0046):
|
|
195
|
+
* environment or upstream. Derived from the agents' classes when no agent
|
|
196
|
+
* produced a result; absent ⇒ the attempt is charged as before. The durable
|
|
197
|
+
* work projection reads this to decide charging, stopping, and the
|
|
198
|
+
* uncharged-attempt bound. */
|
|
199
|
+
class?: SpawnFailureClass;
|
|
189
200
|
};
|
|
190
201
|
acceptance: {
|
|
191
202
|
criteria: AcceptanceResult[];
|
|
@@ -298,6 +309,19 @@ export interface WorkflowSpecialistProfile {
|
|
|
298
309
|
maxTurns?: number;
|
|
299
310
|
}
|
|
300
311
|
|
|
312
|
+
/**
|
|
313
|
+
* Why a failed run was uninformative about the work (ADR 0046). Absent means
|
|
314
|
+
* the attempt produced information about the work (or succeeded), which is the
|
|
315
|
+
* default and is charged against the attempt budget exactly as before.
|
|
316
|
+
*
|
|
317
|
+
* - `environment`: the run never started — a precondition was missing (the
|
|
318
|
+
* vendor binary was absent). Uncharged AND not retried: retrying an unchanged
|
|
319
|
+
* environment cannot help, so the work item stops and names the precondition.
|
|
320
|
+
* - `upstream`: the vendor was reached and refused (5xx, 429, circuit open).
|
|
321
|
+
* Uncharged, but retry stays available (bounded by max_uncharged_attempts).
|
|
322
|
+
*/
|
|
323
|
+
export type SpawnFailureClass = "environment" | "upstream";
|
|
324
|
+
|
|
301
325
|
/** What a spawn adapter returns for one subagent run. */
|
|
302
326
|
export interface SpawnResult {
|
|
303
327
|
ok: boolean;
|
|
@@ -309,6 +333,9 @@ export interface SpawnResult {
|
|
|
309
333
|
durationMs: number;
|
|
310
334
|
/** Populated when ok=false. */
|
|
311
335
|
error?: string;
|
|
336
|
+
/** Set when ok=false and the failure was positively identified as
|
|
337
|
+
* uninformative about the work. Absent ⇒ a work failure (charged). */
|
|
338
|
+
class?: SpawnFailureClass;
|
|
312
339
|
}
|
|
313
340
|
|
|
314
341
|
export interface SpawnRequest {
|
package/src/lib/exec.ts
CHANGED
|
@@ -17,6 +17,13 @@ export interface ExecResult {
|
|
|
17
17
|
* child that handles the signal cleanly still exits 0, so the exit code alone
|
|
18
18
|
* cannot distinguish a kill from an ordinary finish. */
|
|
19
19
|
timedOut?: boolean;
|
|
20
|
+
/** The `code` from a spawn-level failure (`proc` "error" event), e.g.
|
|
21
|
+
* "ENOENT" when the binary is absent. Absent for an ordinary process exit.
|
|
22
|
+
* We collapse such failures to exitCode 127 so callers' exit-code branches
|
|
23
|
+
* still fire, but a shell can legitimately exit 127 too; only this field
|
|
24
|
+
* distinguishes "the binary was never there" from "the process ran and exited
|
|
25
|
+
* 127". The process that knows reports it (as ADR 0044 did for `timedOut`). */
|
|
26
|
+
spawnErrno?: string;
|
|
20
27
|
}
|
|
21
28
|
|
|
22
29
|
export interface ExecOpts {
|
|
@@ -56,7 +63,7 @@ export function exec(cmd: string[], opts: ExecOpts = {}): Promise<ExecResult> {
|
|
|
56
63
|
proc.kill();
|
|
57
64
|
}, timeout);
|
|
58
65
|
|
|
59
|
-
const finish = (exitCode: number, errOverride?: string): void => {
|
|
66
|
+
const finish = (exitCode: number, errOverride?: string, spawnErrno?: string): void => {
|
|
60
67
|
clearTimeout(timer);
|
|
61
68
|
const stdout = Buffer.concat(out).toString("utf-8");
|
|
62
69
|
const stderr = errOverride ?? Buffer.concat(err).toString("utf-8");
|
|
@@ -66,13 +73,16 @@ export function exec(cmd: string[], opts: ExecOpts = {}): Promise<ExecResult> {
|
|
|
66
73
|
stderr: shouldTrim ? stderr.trim() : stderr.replace(/\n$/, ""),
|
|
67
74
|
exitCode,
|
|
68
75
|
...(timedOut ? { timedOut: true } : {}),
|
|
76
|
+
...(spawnErrno ? { spawnErrno } : {}),
|
|
69
77
|
});
|
|
70
78
|
};
|
|
71
79
|
|
|
72
80
|
// ENOENT (binary not found) and similar spawn failures surface here rather
|
|
73
81
|
// than throwing: resolve with 127 + the message so callers' exitCode
|
|
74
|
-
// branches handle it instead of crashing on an unhandled rejection.
|
|
75
|
-
|
|
82
|
+
// branches handle it instead of crashing on an unhandled rejection. The
|
|
83
|
+
// errno (e.g. "ENOENT") is carried through so a real missing binary is
|
|
84
|
+
// distinguishable from a shell that merely exited 127.
|
|
85
|
+
proc.on("error", (e: Error) => finish(127, e.message, (e as NodeJS.ErrnoException).code));
|
|
76
86
|
proc.on("close", (code) => finish(code ?? 1));
|
|
77
87
|
});
|
|
78
88
|
}
|