cadet-agent 0.44.0 → 0.46.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +2 -2
- package/package.json +1 -1
- package/src/cli.mjs +184 -6
- package/src/harness/commands.mjs +9 -0
- package/src/harness/index.mjs +9 -0
- package/src/harness/policy.mjs +96 -1
- package/src/harness/reachability.mjs +431 -0
- package/src/harness/routing.mjs +161 -153
- package/src/harness/state.mjs +101 -5
- package/src/harness/verification.mjs +219 -5
|
@@ -7,13 +7,14 @@
|
|
|
7
7
|
* Contract: docs/core/HarnessContract.md §4 (retry classifier), §5 (commands), §6 (loop).
|
|
8
8
|
*/
|
|
9
9
|
|
|
10
|
-
import { spawn } from 'node:child_process';
|
|
10
|
+
import { spawn, spawnSync } from 'node:child_process';
|
|
11
11
|
import { mkdirSync, writeFileSync } from 'node:fs';
|
|
12
12
|
import { join } from 'node:path';
|
|
13
13
|
import { createEvidence, computeInputTreeHash } from './state.mjs';
|
|
14
14
|
import { hashCriteria, sha256Bytes, timestamp, newId } from './util.mjs';
|
|
15
15
|
import { BudgetTracker, budgetExhaustedResult, evaluateHardStop } from './budget.mjs';
|
|
16
16
|
import { redactString } from './redaction.mjs';
|
|
17
|
+
import { whichAll } from './routing.mjs';
|
|
17
18
|
|
|
18
19
|
export const RETRY_CLASSES = Object.freeze(['deterministic', 'transient', 'repair', 'unknown']);
|
|
19
20
|
export const RESULT_STATUSES = Object.freeze(['passed', 'failed', 'flaky', 'blocked', 'timed-out']);
|
|
@@ -52,6 +53,151 @@ const DETERMINISTIC_SIGNATURES = Object.freeze([
|
|
|
52
53
|
|
|
53
54
|
const TRANSIENT_SIGNATURES = DEFAULT_FLAKY_SIGNATURES;
|
|
54
55
|
|
|
56
|
+
/**
|
|
57
|
+
* A command that never launched produced no test result, so it can never be a red.
|
|
58
|
+
*
|
|
59
|
+
* These are shell- and launcher-level signatures for "the program was not found",
|
|
60
|
+
* as distinct from "the program ran and reported failure". They matter most on
|
|
61
|
+
* Windows, where `shell: true` hands the command string to `cmd.exe`: a bare `bash`
|
|
62
|
+
* resolves through Windows PATH to the Windows Subsystem for Linux stub, which
|
|
63
|
+
* fails without ever exec'ing a shell. The command looks like it ran and failed;
|
|
64
|
+
* nothing ran at all.
|
|
65
|
+
*/
|
|
66
|
+
export const LAUNCH_FAILURE_SIGNATURES = Object.freeze([
|
|
67
|
+
// cmd.exe: the shell could not find the program on PATH.
|
|
68
|
+
'is not recognized as an internal or external command',
|
|
69
|
+
// POSIX shells: the shell could not find the program.
|
|
70
|
+
'command not found',
|
|
71
|
+
// The WSL launcher could not exec the distribution's shell.
|
|
72
|
+
'execvpe(',
|
|
73
|
+
'has no installed distributions',
|
|
74
|
+
'wsl (',
|
|
75
|
+
]);
|
|
76
|
+
|
|
77
|
+
/** POSIX shell conventions: 126 = found but not executable, 127 = command not found. */
|
|
78
|
+
export const LAUNCH_FAILURE_EXIT_CODES = Object.freeze([126, 127]);
|
|
79
|
+
|
|
80
|
+
/**
|
|
81
|
+
* Decide whether a result means the command never actually ran. Returns
|
|
82
|
+
* `{ launchFailed, reason }`; `reason` is a concrete diagnostic, never generic.
|
|
83
|
+
*
|
|
84
|
+
* The bias is deliberate: a launch failure must be *blocked*, never *failed*,
|
|
85
|
+
* because `failed` is the record that satisfies red-before-green. When in doubt
|
|
86
|
+
* the safe direction is "no test result", not "the tests failed".
|
|
87
|
+
*/
|
|
88
|
+
export function detectLaunchFailure({ exitCode = null, timedOut = false, stdout = '', stderr = '', errorMessage = '' } = {}) {
|
|
89
|
+
// A timed-out process did launch; it is not a launch failure.
|
|
90
|
+
if (timedOut) return { launchFailed: false, reason: null };
|
|
91
|
+
|
|
92
|
+
// A spawn-level error means the process was never created at all.
|
|
93
|
+
if (errorMessage) {
|
|
94
|
+
return { launchFailed: true, reason: `the process was never created (${errorMessage})` };
|
|
95
|
+
}
|
|
96
|
+
|
|
97
|
+
const haystack = `${stderr}\n${stdout}`.toLowerCase();
|
|
98
|
+
const signature = LAUNCH_FAILURE_SIGNATURES.find((s) => haystack.includes(s));
|
|
99
|
+
if (signature) {
|
|
100
|
+
return { launchFailed: true, reason: `the shell reported "${signature}"` };
|
|
101
|
+
}
|
|
102
|
+
|
|
103
|
+
if (LAUNCH_FAILURE_EXIT_CODES.includes(exitCode)) {
|
|
104
|
+
return {
|
|
105
|
+
launchFailed: true,
|
|
106
|
+
reason: `exit ${exitCode} is the shell convention for an interpreter that could not be found or executed`,
|
|
107
|
+
};
|
|
108
|
+
}
|
|
109
|
+
|
|
110
|
+
return { launchFailed: false, reason: null };
|
|
111
|
+
}
|
|
112
|
+
|
|
113
|
+
/**
|
|
114
|
+
* POSIX interpreters whose bare name is ambiguous under `cmd.exe`: Windows ships a
|
|
115
|
+
* `bash.exe` stub that resolves but cannot run without a WSL distribution.
|
|
116
|
+
*/
|
|
117
|
+
export const POSIX_INTERPRETERS = Object.freeze(['bash', 'sh', 'dash', 'zsh', 'ksh']);
|
|
118
|
+
|
|
119
|
+
/** The command's first token, unwrapped from quotes. */
|
|
120
|
+
export function commandInterpreter(command) {
|
|
121
|
+
if (typeof command !== 'string') return null;
|
|
122
|
+
const trimmed = command.trim();
|
|
123
|
+
if (!trimmed) return null;
|
|
124
|
+
const m = trimmed.match(/^(?:"([^"]+)"|'([^']+)'|(\S+))/);
|
|
125
|
+
return m ? (m[1] ?? m[2] ?? m[3]) : null;
|
|
126
|
+
}
|
|
127
|
+
|
|
128
|
+
/** Whether a resolved path is one of the Windows shims for `bash`/`wsl`. */
|
|
129
|
+
export function isWslShim(path) {
|
|
130
|
+
if (typeof path !== 'string' || !path) return false;
|
|
131
|
+
const p = path.replace(/\//g, '\\').toLowerCase();
|
|
132
|
+
if (/(^|\\)system32\\bash\.exe$/.test(p)) return true;
|
|
133
|
+
if (/(^|\\)system32\\wsl\.exe$/.test(p)) return true;
|
|
134
|
+
return /(^|\\)windowsapps\\/.test(p);
|
|
135
|
+
}
|
|
136
|
+
|
|
137
|
+
function probeRunnable(path) {
|
|
138
|
+
// `--version` is not universal (dash's `sh` rejects it), so fall back to a
|
|
139
|
+
// trivial execution: the question is only whether this binary can run at all.
|
|
140
|
+
for (const args of [['--version'], ['-c', 'exit 0']]) {
|
|
141
|
+
try {
|
|
142
|
+
const res = spawnSync(path, args, { encoding: 'utf-8', windowsHide: true, shell: false, timeout: 10000 });
|
|
143
|
+
if (res.status === 0) return true;
|
|
144
|
+
} catch { /* try the next probe */ }
|
|
145
|
+
}
|
|
146
|
+
return false;
|
|
147
|
+
}
|
|
148
|
+
|
|
149
|
+
function defaultCandidates(name) {
|
|
150
|
+
return whichAll(name);
|
|
151
|
+
}
|
|
152
|
+
|
|
153
|
+
/**
|
|
154
|
+
* Resolve the command's interpreter before running it. Only a command led by a POSIX
|
|
155
|
+
* interpreter is checked, and only on Windows, so a `cmd` builtin or an ordinary
|
|
156
|
+
* executable is never second-guessed. The command string is not rewritten — the
|
|
157
|
+
* declaration stays the auditable record.
|
|
158
|
+
*
|
|
159
|
+
* `candidates` and `probe` are injectable so the decision is testable without
|
|
160
|
+
* depending on the host's PATH.
|
|
161
|
+
*/
|
|
162
|
+
export function resolveCommandInterpreter(command, {
|
|
163
|
+
platform = process.platform,
|
|
164
|
+
candidates = null,
|
|
165
|
+
probe = probeRunnable,
|
|
166
|
+
} = {}) {
|
|
167
|
+
const interpreter = commandInterpreter(command);
|
|
168
|
+
const raw = interpreter ? interpreter.split(/[\\/]/).pop().toLowerCase() : null;
|
|
169
|
+
// `bash.exe` is the same ambiguity as `bash`, and `where` reports the suffixed form.
|
|
170
|
+
const base = raw ? raw.replace(/\.exe$/, '') : null;
|
|
171
|
+
|
|
172
|
+
if (!base || !POSIX_INTERPRETERS.includes(base)) {
|
|
173
|
+
return { required: false, ok: true, interpreter, usable: [], shims: [], reason: null };
|
|
174
|
+
}
|
|
175
|
+
if (platform !== 'win32') {
|
|
176
|
+
return { required: true, ok: true, interpreter, usable: [], shims: [], reason: null };
|
|
177
|
+
}
|
|
178
|
+
|
|
179
|
+
const found = (candidates ?? defaultCandidates)(base);
|
|
180
|
+
const shims = found.filter(isWslShim);
|
|
181
|
+
const usable = found.filter((p) => !isWslShim(p) && probe(p));
|
|
182
|
+
if (usable.length) {
|
|
183
|
+
return { required: true, ok: true, interpreter, usable, shims, reason: null };
|
|
184
|
+
}
|
|
185
|
+
|
|
186
|
+
let detail;
|
|
187
|
+
if (found.length === 0) {
|
|
188
|
+
detail = 'no installed copy on PATH';
|
|
189
|
+
} else if (shims.length === found.length) {
|
|
190
|
+
detail = `only the Windows Subsystem for Linux stub (${shims.join(', ')})`;
|
|
191
|
+
} else {
|
|
192
|
+
detail = 'the copy on PATH is not runnable';
|
|
193
|
+
}
|
|
194
|
+
|
|
195
|
+
return {
|
|
196
|
+
required: true, ok: false, interpreter, usable, shims,
|
|
197
|
+
reason: `'${base}' cannot be run: ${detail}. Install Git for Windows (which provides a real bash) or declare a command that does not need a POSIX interpreter.`,
|
|
198
|
+
};
|
|
199
|
+
}
|
|
200
|
+
|
|
55
201
|
/**
|
|
56
202
|
* Classify a command result. The classifier is the single source of truth for
|
|
57
203
|
* retry behavior — nothing else may decide to retry.
|
|
@@ -103,7 +249,7 @@ export function classifyRepair({ failedEvidenceId, changedFiles = [] } = {}) {
|
|
|
103
249
|
* Run a command and capture bounded output. Never buffers unbounded output:
|
|
104
250
|
* output beyond `maxInlineBytes` is written to an artifact.
|
|
105
251
|
*
|
|
106
|
-
* Returns `{ exitCode, signal, timedOut, stdout, stderr, durationMs, outputBytes, artifactPath, artifactHash, preview }`.
|
|
252
|
+
* Returns `{ exitCode, signal, timedOut, stdout, stderr, durationMs, outputBytes, artifactPath, artifactHash, preview, launchFailed, launchFailureReason }`.
|
|
107
253
|
*/
|
|
108
254
|
export function runCommand(command, {
|
|
109
255
|
cwd = process.cwd(),
|
|
@@ -125,6 +271,7 @@ export function runCommand(command, {
|
|
|
125
271
|
exitCode: null, signal: null, timedOut: false, stdout: '', stderr: '',
|
|
126
272
|
durationMs: Date.now() - startedAt, outputBytes: 0, artifactPath: null,
|
|
127
273
|
artifactHash: null, preview: '', errorMessage: err.message,
|
|
274
|
+
launchFailed: true, launchFailureReason: `the process was never created (${err.message})`,
|
|
128
275
|
});
|
|
129
276
|
return;
|
|
130
277
|
}
|
|
@@ -177,9 +324,11 @@ export function runCommand(command, {
|
|
|
177
324
|
}
|
|
178
325
|
}
|
|
179
326
|
const preview = redactString((stdout + stderr).slice(0, previewBytes));
|
|
327
|
+
const failure = detectLaunchFailure({ exitCode, timedOut, stdout, stderr, errorMessage });
|
|
180
328
|
resolve({
|
|
181
329
|
exitCode, signal, timedOut, stdout, stderr, durationMs, outputBytes,
|
|
182
330
|
artifactPath, artifactHash, preview, errorMessage,
|
|
331
|
+
launchFailed: failure.launchFailed, launchFailureReason: failure.reason,
|
|
183
332
|
});
|
|
184
333
|
};
|
|
185
334
|
|
|
@@ -265,6 +414,7 @@ export async function runVerificationLoop({
|
|
|
265
414
|
flakySignatures = DEFAULT_FLAKY_SIGNATURES,
|
|
266
415
|
priorEvidence = [],
|
|
267
416
|
requireRedFirst = null,
|
|
417
|
+
resolveInterpreterImpl = resolveCommandInterpreter,
|
|
268
418
|
now = () => new Date(),
|
|
269
419
|
} = {}) {
|
|
270
420
|
const tracker = budgets || new BudgetTracker(policy);
|
|
@@ -273,6 +423,47 @@ export async function runVerificationLoop({
|
|
|
273
423
|
const criteriaHash = hashCriteria(criteria);
|
|
274
424
|
const perStepLimit = maxAttemptsOverride ?? (policy?.budgets?.maxRetriesPerStep?.hard ?? 2) + 1;
|
|
275
425
|
|
|
426
|
+
// Refuse a command whose interpreter cannot run, before executing anything. The
|
|
427
|
+
// declared command is not rewritten — the point is that a run which never starts
|
|
428
|
+
// must not leave behind a record that could later be read as a red.
|
|
429
|
+
const interpreter = resolveInterpreterImpl(command, {});
|
|
430
|
+
if (interpreter?.required && !interpreter.ok) {
|
|
431
|
+
const detail = `command never launched: ${interpreter.reason}`;
|
|
432
|
+
const evidence = createEvidence({
|
|
433
|
+
evidenceId: newId(),
|
|
434
|
+
workItemId,
|
|
435
|
+
acceptanceCriterionId,
|
|
436
|
+
phase,
|
|
437
|
+
gate,
|
|
438
|
+
status: 'blocked',
|
|
439
|
+
command,
|
|
440
|
+
result: detail,
|
|
441
|
+
exitCode: null,
|
|
442
|
+
inputTreeHash,
|
|
443
|
+
criteriaHash,
|
|
444
|
+
relevantFiles,
|
|
445
|
+
commit,
|
|
446
|
+
createdAt: now(),
|
|
447
|
+
source: 'automated',
|
|
448
|
+
});
|
|
449
|
+
attempts.push({
|
|
450
|
+
attempt: 1,
|
|
451
|
+
spanId: newId(),
|
|
452
|
+
tool,
|
|
453
|
+
command,
|
|
454
|
+
exitCode: null,
|
|
455
|
+
durationMs: 0,
|
|
456
|
+
outputBytes: 0,
|
|
457
|
+
artifactPath: null,
|
|
458
|
+
artifactHash: null,
|
|
459
|
+
status: 'blocked',
|
|
460
|
+
retryClass: 'deterministic',
|
|
461
|
+
retryReason: 'interpreter not runnable',
|
|
462
|
+
evidence,
|
|
463
|
+
});
|
|
464
|
+
return finalize({ status: 'blocked', attempts, tracker, inputTreeHash, criteriaHash, stopReason: 'launch-failed', diagnostic: detail });
|
|
465
|
+
}
|
|
466
|
+
|
|
276
467
|
// Red-before-green: a testable gate must not be satisfied by green evidence
|
|
277
468
|
// unless a prior failed (red) record exists for the same work item and gate.
|
|
278
469
|
// `requireRedFirst` may be set explicitly; otherwise it applies to
|
|
@@ -309,6 +500,12 @@ export async function runVerificationLoop({
|
|
|
309
500
|
// configured budget (e.g. cost) could not be measured.
|
|
310
501
|
if (hardStop.exhausted || hardStop.blocked) gatePassed = false;
|
|
311
502
|
|
|
503
|
+
// `runCommand` reports this directly; an injected runner may not, so fall back
|
|
504
|
+
// to the same detection over its result. A timed-out process did launch.
|
|
505
|
+
const launch = result.launchFailed === true
|
|
506
|
+
? { launchFailed: true, reason: result.launchFailureReason || 'the command did not launch' }
|
|
507
|
+
: detectLaunchFailure(result);
|
|
508
|
+
|
|
312
509
|
const evidence = createEvidence({
|
|
313
510
|
evidenceId: newId(),
|
|
314
511
|
workItemId,
|
|
@@ -317,11 +514,13 @@ export async function runVerificationLoop({
|
|
|
317
514
|
gate,
|
|
318
515
|
status: hardStop.exhausted || hardStop.blocked
|
|
319
516
|
? 'blocked'
|
|
320
|
-
: (gatePassed ? 'passed' : (result.timedOut ? 'blocked' : 'failed')),
|
|
517
|
+
: (gatePassed ? 'passed' : (result.timedOut ? 'blocked' : (launch.launchFailed ? 'blocked' : 'failed'))),
|
|
321
518
|
command,
|
|
322
519
|
result: hardStop.exhausted
|
|
323
520
|
? `budget-exhausted: ${hardStop.reason}`
|
|
324
|
-
: (hardStop.blocked
|
|
521
|
+
: (hardStop.blocked
|
|
522
|
+
? `budget-blocked: ${hardStop.reason}`
|
|
523
|
+
: (launch.launchFailed ? `launch-failed: ${launch.reason}` : describeResult(result))),
|
|
325
524
|
exitCode: result.exitCode,
|
|
326
525
|
artifactPath: result.artifactPath,
|
|
327
526
|
artifactHash: result.artifactHash,
|
|
@@ -343,7 +542,7 @@ export async function runVerificationLoop({
|
|
|
343
542
|
outputBytes: result.outputBytes,
|
|
344
543
|
artifactPath: result.artifactPath,
|
|
345
544
|
artifactHash: result.artifactHash,
|
|
346
|
-
status: result.timedOut ? 'timed-out' : (gatePassed ? 'passed' : 'failed'),
|
|
545
|
+
status: result.timedOut ? 'timed-out' : (gatePassed ? 'passed' : (launch.launchFailed ? 'blocked' : 'failed')),
|
|
347
546
|
retryClass: classification.retryClass,
|
|
348
547
|
retryReason: classification.reason,
|
|
349
548
|
evidence,
|
|
@@ -351,6 +550,21 @@ export async function runVerificationLoop({
|
|
|
351
550
|
|
|
352
551
|
lastResult = result;
|
|
353
552
|
|
|
553
|
+
// A command that never launched is not a red, and is not retried: a missing or
|
|
554
|
+
// unrunnable interpreter does not appear on a second attempt. It must never reach
|
|
555
|
+
// the retry machinery or the red-before-green check, so it stops here as blocked.
|
|
556
|
+
if (launch.launchFailed) {
|
|
557
|
+
return finalize({
|
|
558
|
+
status: 'blocked',
|
|
559
|
+
attempts,
|
|
560
|
+
tracker,
|
|
561
|
+
inputTreeHash,
|
|
562
|
+
criteriaHash,
|
|
563
|
+
stopReason: 'launch-failed',
|
|
564
|
+
diagnostic: `command never launched: ${command} — ${launch.reason}`,
|
|
565
|
+
});
|
|
566
|
+
}
|
|
567
|
+
|
|
354
568
|
if (hardStop.exhausted || hardStop.blocked) {
|
|
355
569
|
return finalize({
|
|
356
570
|
status: 'failed',
|