cadet-agent 0.20.2 → 0.24.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +46 -6
- package/package.json +35 -35
- package/src/cli.mjs +373 -33
- package/src/harness/archive.mjs +242 -0
- package/src/harness/budget.mjs +298 -0
- package/src/harness/context.mjs +229 -0
- package/src/harness/hook.mjs +147 -0
- package/src/harness/index.mjs +60 -0
- package/src/harness/ledger.mjs +309 -0
- package/src/harness/policy.mjs +359 -0
- package/src/harness/redaction.mjs +133 -0
- package/src/harness/routing.mjs +153 -0
- package/src/harness/state.mjs +646 -0
- package/src/harness/util.mjs +149 -0
- package/src/harness/verification.mjs +490 -0
- package/src/install.mjs +92 -154
|
@@ -0,0 +1,149 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Shared harness primitives: UUIDv4, SHA-256 over UTF-8 bytes, tree hashing,
|
|
3
|
+
* deterministic JSON serialization, and UTC timestamps.
|
|
4
|
+
*
|
|
5
|
+
* Contract: docs/core/HarnessContract.md §2 (identifiers, hashes).
|
|
6
|
+
*/
|
|
7
|
+
|
|
8
|
+
import { createHash, randomUUID } from 'node:crypto';
|
|
9
|
+
import { readFileSync, existsSync } from 'node:fs';
|
|
10
|
+
import { spawnSync } from 'node:child_process';
|
|
11
|
+
|
|
12
|
+
/** UUIDv4 identifier. */
|
|
13
|
+
export function newId() {
|
|
14
|
+
return randomUUID();
|
|
15
|
+
}
|
|
16
|
+
|
|
17
|
+
export function isUuid(value) {
|
|
18
|
+
return typeof value === 'string'
|
|
19
|
+
&& /^[0-9a-f]{8}-[0-9a-f]{4}-4[0-9a-f]{3}-[89ab][0-9a-f]{3}-[0-9a-f]{12}$/.test(value);
|
|
20
|
+
}
|
|
21
|
+
|
|
22
|
+
/** SHA-256 hex digest over UTF-8 bytes. */
|
|
23
|
+
export function sha256(value) {
|
|
24
|
+
return createHash('sha256').update(value, 'utf-8').digest('hex');
|
|
25
|
+
}
|
|
26
|
+
|
|
27
|
+
/** SHA-256 hex digest over raw bytes (Buffers are hashed as-is). */
|
|
28
|
+
export function sha256Bytes(buf) {
|
|
29
|
+
return createHash('sha256').update(buf).digest('hex');
|
|
30
|
+
}
|
|
31
|
+
|
|
32
|
+
/** Hash of a file's exact bytes, or null when the file is missing. */
|
|
33
|
+
export function hashFile(path) {
|
|
34
|
+
if (!existsSync(path)) return null;
|
|
35
|
+
try {
|
|
36
|
+
return sha256Bytes(readFileSync(path));
|
|
37
|
+
} catch {
|
|
38
|
+
return null;
|
|
39
|
+
}
|
|
40
|
+
}
|
|
41
|
+
|
|
42
|
+
/**
|
|
43
|
+
* Deterministic hash of a set of `(relativePath, fileHash)` pairs.
|
|
44
|
+
* Pairs are sorted by path so the hash is order-independent and stable across
|
|
45
|
+
* platforms. Files without a resolvable hash are recorded as `missing`.
|
|
46
|
+
*/
|
|
47
|
+
export function hashTree(pairs) {
|
|
48
|
+
const normalized = [...pairs]
|
|
49
|
+
.map(({ path, hash }) => ({
|
|
50
|
+
path: String(path).replace(/\\/g, '/'),
|
|
51
|
+
hash: hash || 'missing',
|
|
52
|
+
}))
|
|
53
|
+
.sort((a, b) => (a.path < b.path ? -1 : a.path > b.path ? 1 : 0));
|
|
54
|
+
return sha256(JSON.stringify(normalized));
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
/**
|
|
58
|
+
* Hash an acceptance-criteria document or list. `criteria` may be a string or an
|
|
59
|
+
* array of strings; a stable serialization is used either way.
|
|
60
|
+
*/
|
|
61
|
+
export function hashCriteria(criteria) {
|
|
62
|
+
if (criteria === null || criteria === undefined) return sha256('[]');
|
|
63
|
+
const arr = Array.isArray(criteria) ? criteria.map(String) : [String(criteria)];
|
|
64
|
+
return sha256(JSON.stringify(arr));
|
|
65
|
+
}
|
|
66
|
+
|
|
67
|
+
/** ISO-8601 UTC timestamp for a Date or "now". */
|
|
68
|
+
export function timestamp(at = new Date()) {
|
|
69
|
+
return (at instanceof Date ? at : new Date(at)).toISOString();
|
|
70
|
+
}
|
|
71
|
+
|
|
72
|
+
/** Current time provider; injectable for deterministic tests. */
|
|
73
|
+
export function nowMs() {
|
|
74
|
+
return Date.now();
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
/**
|
|
78
|
+
* Canonical JSON with sorted keys. Used for stable hashes and for comparing
|
|
79
|
+
* evidence records without key-order noise.
|
|
80
|
+
*/
|
|
81
|
+
export function canonicalJson(value) {
|
|
82
|
+
return JSON.stringify(sortKeys(value));
|
|
83
|
+
}
|
|
84
|
+
|
|
85
|
+
function sortKeys(value) {
|
|
86
|
+
if (Array.isArray(value)) return value.map(sortKeys);
|
|
87
|
+
if (value && typeof value === 'object') {
|
|
88
|
+
const out = {};
|
|
89
|
+
for (const key of Object.keys(value).sort()) out[key] = sortKeys(value[key]);
|
|
90
|
+
return out;
|
|
91
|
+
}
|
|
92
|
+
return value;
|
|
93
|
+
}
|
|
94
|
+
|
|
95
|
+
export { canonicalJson as stableStringify };
|
|
96
|
+
|
|
97
|
+
/**
|
|
98
|
+
* List the files changed in the working tree relative to HEAD, using git.
|
|
99
|
+
* Returns forward-slash relative paths. Returns an empty array when git is
|
|
100
|
+
* unavailable or the directory is not a repository — callers must not assume
|
|
101
|
+
* freshness coverage in that case; use `gitChangedFiles` when the distinction
|
|
102
|
+
* between "no changes" and "no git" matters.
|
|
103
|
+
*/
|
|
104
|
+
export function changedFiles(cwd, { runner = defaultGitRunner } = {}) {
|
|
105
|
+
return gitChangedFiles(cwd, { runner }).files;
|
|
106
|
+
}
|
|
107
|
+
|
|
108
|
+
/**
|
|
109
|
+
* List changed files and report whether git was actually queryable.
|
|
110
|
+
* Returns `{ available, files, reason }`. `available: false` means freshness
|
|
111
|
+
* coverage could not be established and callers must fail safe.
|
|
112
|
+
*/
|
|
113
|
+
export function gitChangedFiles(cwd, { runner = defaultGitRunner } = {}) {
|
|
114
|
+
let res;
|
|
115
|
+
try {
|
|
116
|
+
res = runner('git', ['-C', cwd, 'status', '--porcelain', '--untracked-files=all']);
|
|
117
|
+
} catch (err) {
|
|
118
|
+
return { available: false, files: [], reason: `git invocation failed: ${err.message}` };
|
|
119
|
+
}
|
|
120
|
+
if (!res) {
|
|
121
|
+
return { available: false, files: [], reason: 'git is not available' };
|
|
122
|
+
}
|
|
123
|
+
if (res.error || res.status === null) {
|
|
124
|
+
return { available: false, files: [], reason: 'git is not installed or could not be executed' };
|
|
125
|
+
}
|
|
126
|
+
if (res.status !== 0) {
|
|
127
|
+
// Not a repository, or git refused the query.
|
|
128
|
+
return { available: false, files: [], reason: String(res.stderr || '').trim() || `git exited ${res.status}` };
|
|
129
|
+
}
|
|
130
|
+
const files = new Set();
|
|
131
|
+
for (const line of String(res.stdout || '').split(/\r?\n/)) {
|
|
132
|
+
if (!line.trim()) continue;
|
|
133
|
+
// Porcelain v1: XY<space>path (rename: "old -> new").
|
|
134
|
+
let path = line.slice(3).trim();
|
|
135
|
+
if (path.includes(' -> ')) path = path.split(' -> ').pop().trim();
|
|
136
|
+
path = path.replace(/^"|"$/g, '');
|
|
137
|
+
if (path) files.add(path.replace(/\\/g, '/'));
|
|
138
|
+
}
|
|
139
|
+
return { available: true, files: [...files].sort(), reason: null };
|
|
140
|
+
}
|
|
141
|
+
|
|
142
|
+
function defaultGitRunner(cmd, args) {
|
|
143
|
+
try {
|
|
144
|
+
return spawnSync(cmd, args, { encoding: 'utf-8', windowsHide: true });
|
|
145
|
+
} catch {
|
|
146
|
+
return null;
|
|
147
|
+
}
|
|
148
|
+
}
|
|
149
|
+
|
|
@@ -0,0 +1,490 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Cadet-Agent verification runner.
|
|
3
|
+
*
|
|
4
|
+
* Deterministic, bounded verification with a single retry classifier. Skills may
|
|
5
|
+
* supply policy (commands, flaky signatures) but may not invent classifications.
|
|
6
|
+
*
|
|
7
|
+
* Contract: docs/core/HarnessContract.md §4 (retry classifier), §5 (commands), §6 (loop).
|
|
8
|
+
*/
|
|
9
|
+
|
|
10
|
+
import { spawn } from 'node:child_process';
|
|
11
|
+
import { mkdirSync, writeFileSync } from 'node:fs';
|
|
12
|
+
import { join } from 'node:path';
|
|
13
|
+
import { createEvidence, computeInputTreeHash } from './state.mjs';
|
|
14
|
+
import { hashCriteria, sha256Bytes, timestamp, newId } from './util.mjs';
|
|
15
|
+
import { BudgetTracker, budgetExhaustedResult, evaluateHardStop } from './budget.mjs';
|
|
16
|
+
import { redactString } from './redaction.mjs';
|
|
17
|
+
|
|
18
|
+
export const RETRY_CLASSES = Object.freeze(['deterministic', 'transient', 'repair', 'unknown']);
|
|
19
|
+
export const RESULT_STATUSES = Object.freeze(['passed', 'failed', 'flaky', 'blocked', 'timed-out']);
|
|
20
|
+
|
|
21
|
+
/** Backoff schedule for transient retries (ms). */
|
|
22
|
+
export const TRANSIENT_BACKOFF_MS = Object.freeze([250, 1000, 4000]);
|
|
23
|
+
|
|
24
|
+
/** Default flaky signatures (configurable via policy.estimation? no — via verification policy). */
|
|
25
|
+
export const DEFAULT_FLAKY_SIGNATURES = Object.freeze([
|
|
26
|
+
'econnreset',
|
|
27
|
+
'etimedout',
|
|
28
|
+
'socket hang up',
|
|
29
|
+
'connection refused',
|
|
30
|
+
'temporarily unavailable',
|
|
31
|
+
'service unavailable',
|
|
32
|
+
'process launch failed',
|
|
33
|
+
'eaddrnotavail',
|
|
34
|
+
]);
|
|
35
|
+
|
|
36
|
+
const DETERMINISTIC_SIGNATURES = Object.freeze([
|
|
37
|
+
'assertionerror',
|
|
38
|
+
'expected',
|
|
39
|
+
'syntaxerror',
|
|
40
|
+
'compile error',
|
|
41
|
+
'compilationerror',
|
|
42
|
+
'cs0',
|
|
43
|
+
'analyzer',
|
|
44
|
+
'unt',
|
|
45
|
+
'invalid input',
|
|
46
|
+
'usage:',
|
|
47
|
+
'unknown option',
|
|
48
|
+
'no such file',
|
|
49
|
+
'cannot find module',
|
|
50
|
+
'unhandled rejection',
|
|
51
|
+
]);
|
|
52
|
+
|
|
53
|
+
const TRANSIENT_SIGNATURES = DEFAULT_FLAKY_SIGNATURES;
|
|
54
|
+
|
|
55
|
+
/**
|
|
56
|
+
* Classify a command result. The classifier is the single source of truth for
|
|
57
|
+
* retry behavior — nothing else may decide to retry.
|
|
58
|
+
*
|
|
59
|
+
* Returns `{ retryClass, reason, retryable }`.
|
|
60
|
+
*/
|
|
61
|
+
export function classifyResult({ exitCode = null, timedOut = false, stdout = '', stderr = '', errorMessage = '', flakySignatures = DEFAULT_FLAKY_SIGNATURES } = {}) {
|
|
62
|
+
const haystack = `${errorMessage}\n${stderr}\n${stdout}`.toLowerCase();
|
|
63
|
+
|
|
64
|
+
if (timedOut) {
|
|
65
|
+
// A timeout is deterministic unless its signature is explicitly configured flaky.
|
|
66
|
+
if (flakySignatures.some((s) => haystack.includes(s))) {
|
|
67
|
+
return { retryClass: 'transient', reason: 'timeout with a configured flaky signature', retryable: true };
|
|
68
|
+
}
|
|
69
|
+
return { retryClass: 'deterministic', reason: 'reproducible timeout does not retry automatically', retryable: false };
|
|
70
|
+
}
|
|
71
|
+
|
|
72
|
+
if (TRANSIENT_SIGNATURES.some((s) => haystack.includes(s))) {
|
|
73
|
+
return { retryClass: 'transient', reason: `matched transient signature`, retryable: true };
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
if (exitCode === 0) {
|
|
77
|
+
return { retryClass: 'deterministic', reason: 'success', retryable: false };
|
|
78
|
+
}
|
|
79
|
+
|
|
80
|
+
if (DETERMINISTIC_SIGNATURES.some((s) => haystack.includes(s))) {
|
|
81
|
+
return { retryClass: 'deterministic', reason: 'matched a deterministic failure signature', retryable: false };
|
|
82
|
+
}
|
|
83
|
+
|
|
84
|
+
if (exitCode !== null && exitCode !== 0) {
|
|
85
|
+
return { retryClass: 'unknown', reason: `unrecognized failure (exit ${exitCode}); escalate`, retryable: false };
|
|
86
|
+
}
|
|
87
|
+
|
|
88
|
+
return { retryClass: 'unknown', reason: 'unrecognized result; escalate', retryable: false };
|
|
89
|
+
}
|
|
90
|
+
|
|
91
|
+
/** A `repair` retry is only valid when it references the failed evidence and changed files. */
|
|
92
|
+
export function classifyRepair({ failedEvidenceId, changedFiles = [] } = {}) {
|
|
93
|
+
if (!failedEvidenceId) {
|
|
94
|
+
return { retryClass: 'unknown', reason: 'repair retry requires a failed evidence reference', retryable: false };
|
|
95
|
+
}
|
|
96
|
+
if (!Array.isArray(changedFiles) || changedFiles.length === 0) {
|
|
97
|
+
return { retryClass: 'unknown', reason: 'repair retry requires changed files', retryable: false };
|
|
98
|
+
}
|
|
99
|
+
return { retryClass: 'repair', reason: 'code/config repair followed by a rerun', retryable: true, failedEvidenceId, changedFiles };
|
|
100
|
+
}
|
|
101
|
+
|
|
102
|
+
/**
|
|
103
|
+
* Run a command and capture bounded output. Never buffers unbounded output:
|
|
104
|
+
* output beyond `maxInlineBytes` is written to an artifact.
|
|
105
|
+
*
|
|
106
|
+
* Returns `{ exitCode, signal, timedOut, stdout, stderr, durationMs, outputBytes, artifactPath, artifactHash, preview }`.
|
|
107
|
+
*/
|
|
108
|
+
export function runCommand(command, {
|
|
109
|
+
cwd = process.cwd(),
|
|
110
|
+
env = {},
|
|
111
|
+
timeoutMs = 30 * 60 * 1000,
|
|
112
|
+
maxInlineBytes = 64 * 1024,
|
|
113
|
+
previewBytes = 4 * 1024,
|
|
114
|
+
artifactDir = null,
|
|
115
|
+
shell = true,
|
|
116
|
+
spawnImpl = spawn,
|
|
117
|
+
} = {}) {
|
|
118
|
+
return new Promise((resolve) => {
|
|
119
|
+
const startedAt = Date.now();
|
|
120
|
+
let child;
|
|
121
|
+
try {
|
|
122
|
+
child = spawnImpl(command, { cwd, env: { ...process.env, ...env }, shell, windowsHide: true });
|
|
123
|
+
} catch (err) {
|
|
124
|
+
resolve({
|
|
125
|
+
exitCode: null, signal: null, timedOut: false, stdout: '', stderr: '',
|
|
126
|
+
durationMs: Date.now() - startedAt, outputBytes: 0, artifactPath: null,
|
|
127
|
+
artifactHash: null, preview: '', errorMessage: err.message,
|
|
128
|
+
});
|
|
129
|
+
return;
|
|
130
|
+
}
|
|
131
|
+
|
|
132
|
+
const stdoutChunks = [];
|
|
133
|
+
const stderrChunks = [];
|
|
134
|
+
let stdoutBytes = 0;
|
|
135
|
+
let stderrBytes = 0;
|
|
136
|
+
let timedOut = false;
|
|
137
|
+
let finished = false;
|
|
138
|
+
|
|
139
|
+
const timer = setTimeout(() => {
|
|
140
|
+
timedOut = true;
|
|
141
|
+
try { child.kill('SIGKILL'); } catch { /* already gone */ }
|
|
142
|
+
}, timeoutMs);
|
|
143
|
+
|
|
144
|
+
child.stdout?.on('data', (chunk) => {
|
|
145
|
+
stdoutBytes += chunk.length;
|
|
146
|
+
if (stdoutBytes <= maxInlineBytes) stdoutChunks.push(chunk);
|
|
147
|
+
});
|
|
148
|
+
child.stderr?.on('data', (chunk) => {
|
|
149
|
+
stderrBytes += chunk.length;
|
|
150
|
+
if (stderrBytes <= maxInlineBytes) stderrChunks.push(chunk);
|
|
151
|
+
});
|
|
152
|
+
|
|
153
|
+
const finish = (exitCode, signal, errorMessage = '') => {
|
|
154
|
+
if (finished) return;
|
|
155
|
+
finished = true;
|
|
156
|
+
clearTimeout(timer);
|
|
157
|
+
const durationMs = Date.now() - startedAt;
|
|
158
|
+
const stdout = Buffer.concat(stdoutChunks).toString('utf-8');
|
|
159
|
+
const stderr = Buffer.concat(stderrChunks).toString('utf-8');
|
|
160
|
+
const outputBytes = stdoutBytes + stderrBytes;
|
|
161
|
+
|
|
162
|
+
let artifactPath = null;
|
|
163
|
+
let artifactHash = null;
|
|
164
|
+
if (outputBytes > maxInlineBytes && artifactDir) {
|
|
165
|
+
try {
|
|
166
|
+
mkdirSync(artifactDir, { recursive: true });
|
|
167
|
+
const full = join(artifactDir, `output-${newId()}.log`);
|
|
168
|
+
// Redact before persisting: the artifact must never contain secrets,
|
|
169
|
+
// even though the in-memory copy is kept raw for classification.
|
|
170
|
+
const body = redactString(`${stdout}\n${stderr}`);
|
|
171
|
+
writeFileSync(full, body, 'utf-8');
|
|
172
|
+
artifactPath = full;
|
|
173
|
+
// Hash covers the exact persisted (redacted) bytes.
|
|
174
|
+
artifactHash = sha256Bytes(Buffer.from(body, 'utf-8'));
|
|
175
|
+
} catch {
|
|
176
|
+
artifactPath = null;
|
|
177
|
+
}
|
|
178
|
+
}
|
|
179
|
+
const preview = redactString((stdout + stderr).slice(0, previewBytes));
|
|
180
|
+
resolve({
|
|
181
|
+
exitCode, signal, timedOut, stdout, stderr, durationMs, outputBytes,
|
|
182
|
+
artifactPath, artifactHash, preview, errorMessage,
|
|
183
|
+
});
|
|
184
|
+
};
|
|
185
|
+
|
|
186
|
+
child.on('error', (err) => finish(null, null, err.message));
|
|
187
|
+
child.on('close', (code, signal) => finish(code, signal));
|
|
188
|
+
});
|
|
189
|
+
}
|
|
190
|
+
|
|
191
|
+
/**
|
|
192
|
+
* Build a verification command descriptor for a gate.
|
|
193
|
+
* Project-specific commands (analyzer/compile/test) come from `.cadet/harness.json`.
|
|
194
|
+
*/
|
|
195
|
+
export function commandForGate(gate, { projectPath = '.', policy = null, unityAvailable = false } = {}) {
|
|
196
|
+
switch (gate) {
|
|
197
|
+
case 'testsPassed':
|
|
198
|
+
return policy?.testCommand
|
|
199
|
+
? { command: policy.testCommand, tool: 'test', automated: true }
|
|
200
|
+
: { command: 'npm test', tool: 'test', automated: true };
|
|
201
|
+
case 'compileCheckConfirmed':
|
|
202
|
+
if (policy?.compileCommand) {
|
|
203
|
+
return { command: policy.compileCommand, tool: 'unity-build', automated: true };
|
|
204
|
+
}
|
|
205
|
+
if (unityAvailable) {
|
|
206
|
+
return {
|
|
207
|
+
command: `unity build ${projectPath} --target StandaloneWindows64 -o "${join(projectPath, 'Temp', 'cadet-build')}" --format json`,
|
|
208
|
+
tool: 'unity-build',
|
|
209
|
+
automated: true,
|
|
210
|
+
};
|
|
211
|
+
}
|
|
212
|
+
return { command: null, tool: 'manual-confirmation', automated: false, reason: 'Unity CLI unavailable' };
|
|
213
|
+
case 'unityAnalyzerClean':
|
|
214
|
+
if (policy?.analyzerCommand) {
|
|
215
|
+
return {
|
|
216
|
+
command: `unity run ${projectPath} --command ${policy.analyzerCommand} --format json`,
|
|
217
|
+
tool: 'unity-analyzer',
|
|
218
|
+
automated: true,
|
|
219
|
+
};
|
|
220
|
+
}
|
|
221
|
+
return { command: null, tool: 'unity-analyzer', automated: false, reason: 'analyzer command not declared in .cadet/harness.json' };
|
|
222
|
+
default:
|
|
223
|
+
return { command: null, tool: 'agent-owned', automated: false, reason: `${gate} is agent-owned` };
|
|
224
|
+
}
|
|
225
|
+
}
|
|
226
|
+
|
|
227
|
+
/** Detect zero `UNT*` diagnostics in a Unity analyzer JSON envelope. */
|
|
228
|
+
export function analyzerClean(stdout) {
|
|
229
|
+
try {
|
|
230
|
+
const parsed = JSON.parse(stdout);
|
|
231
|
+
const text = JSON.stringify(parsed);
|
|
232
|
+
return !/UNT\d+/.test(text);
|
|
233
|
+
} catch {
|
|
234
|
+
// Non-JSON: treat any UNT token in the raw text as a finding.
|
|
235
|
+
return !/UNT\d+/.test(stdout);
|
|
236
|
+
}
|
|
237
|
+
}
|
|
238
|
+
|
|
239
|
+
/**
|
|
240
|
+
* The verification loop contract (contract §6). Runs a command once, classifies,
|
|
241
|
+
* retries only when the class is retryable and budget remains, and records every
|
|
242
|
+
* attempt. No attempt ever overwrites a previous one.
|
|
243
|
+
*
|
|
244
|
+
* `runner` is injectable for tests: `async (attempt) => { exitCode, ... }`.
|
|
245
|
+
*/
|
|
246
|
+
export async function runVerificationLoop({
|
|
247
|
+
gate,
|
|
248
|
+
command,
|
|
249
|
+
workItemId,
|
|
250
|
+
phase,
|
|
251
|
+
acceptanceCriterionId = null,
|
|
252
|
+
relevantFiles = [],
|
|
253
|
+
criteria = [],
|
|
254
|
+
rootDir = process.cwd(),
|
|
255
|
+
policy,
|
|
256
|
+
budgets,
|
|
257
|
+
runCommandImpl = runCommand,
|
|
258
|
+
sleepImpl = (ms) => new Promise((r) => setTimeout(r, ms)),
|
|
259
|
+
maxAttemptsOverride = null,
|
|
260
|
+
tool = 'test',
|
|
261
|
+
maxInlineBytes,
|
|
262
|
+
previewBytes,
|
|
263
|
+
artifactDir = null,
|
|
264
|
+
flakySignatures = DEFAULT_FLAKY_SIGNATURES,
|
|
265
|
+
priorEvidence = [],
|
|
266
|
+
requireRedFirst = null,
|
|
267
|
+
now = () => new Date(),
|
|
268
|
+
} = {}) {
|
|
269
|
+
const tracker = budgets || new BudgetTracker(policy);
|
|
270
|
+
const attempts = [];
|
|
271
|
+
const inputTreeHash = computeInputTreeHash(rootDir, relevantFiles);
|
|
272
|
+
const criteriaHash = hashCriteria(criteria);
|
|
273
|
+
const perStepLimit = maxAttemptsOverride ?? (policy?.budgets?.maxRetriesPerStep?.hard ?? 2) + 1;
|
|
274
|
+
|
|
275
|
+
// Red-before-green: a testable gate must not be satisfied by green evidence
|
|
276
|
+
// unless a prior failed (red) record exists for the same work item and gate.
|
|
277
|
+
// `requireRedFirst` may be set explicitly; otherwise it applies to
|
|
278
|
+
// `testsPassed` by default.
|
|
279
|
+
const redFirstRequired = requireRedFirst === null ? gate === 'testsPassed' : requireRedFirst === true;
|
|
280
|
+
const priorRed = Array.isArray(priorEvidence)
|
|
281
|
+
&& priorEvidence.some((e) => e && e.gate === gate && e.workItemId === workItemId && e.status === 'failed');
|
|
282
|
+
|
|
283
|
+
let lastResult = null;
|
|
284
|
+
let stopped = false;
|
|
285
|
+
|
|
286
|
+
for (let attempt = 1; attempt <= perStepLimit; attempt++) {
|
|
287
|
+
const startedAt = now();
|
|
288
|
+
const result = await runCommandImpl(command, { cwd: rootDir, timeoutMs: policy?.budgets?.maxWallClockMs?.hard, maxInlineBytes, previewBytes, artifactDir });
|
|
289
|
+
tracker.add('toolCalls', 1);
|
|
290
|
+
tracker.add('wallClockMs', result.durationMs || 0);
|
|
291
|
+
// Command output counts against the output-token budget (estimated from the
|
|
292
|
+
// UTF-8 byte length), so the output budget is enforced rather than advisory.
|
|
293
|
+
const bytesPerToken = policy?.estimation?.bytesPerToken || 3;
|
|
294
|
+
tracker.add('outputTokens', Math.ceil((result.outputBytes || 0) / bytesPerToken));
|
|
295
|
+
|
|
296
|
+
// Hard budgets are enforceable, not advisory: if this attempt pushed a hard
|
|
297
|
+
// limit (tool calls, wall-clock, output tokens, cost), stop immediately and
|
|
298
|
+
// never report a passing gate.
|
|
299
|
+
const hardStop = evaluateHardStop(tracker, { policy });
|
|
300
|
+
|
|
301
|
+
const classification = classifyResult({ ...result, flakySignatures });
|
|
302
|
+
const passed = result.exitCode === 0 && !result.timedOut;
|
|
303
|
+
let gatePassed = passed;
|
|
304
|
+
if (gate === 'unityAnalyzerClean' && passed) {
|
|
305
|
+
gatePassed = analyzerClean(result.stdout);
|
|
306
|
+
}
|
|
307
|
+
// A gate can never be satisfied when a hard budget was exceeded, or when a
|
|
308
|
+
// configured budget (e.g. cost) could not be measured.
|
|
309
|
+
if (hardStop.exhausted || hardStop.blocked) gatePassed = false;
|
|
310
|
+
|
|
311
|
+
const evidence = createEvidence({
|
|
312
|
+
evidenceId: newId(),
|
|
313
|
+
workItemId,
|
|
314
|
+
acceptanceCriterionId,
|
|
315
|
+
phase,
|
|
316
|
+
gate,
|
|
317
|
+
status: hardStop.exhausted || hardStop.blocked
|
|
318
|
+
? 'blocked'
|
|
319
|
+
: (gatePassed ? 'passed' : (result.timedOut ? 'blocked' : 'failed')),
|
|
320
|
+
command,
|
|
321
|
+
result: hardStop.exhausted
|
|
322
|
+
? `budget-exhausted: ${hardStop.reason}`
|
|
323
|
+
: (hardStop.blocked ? `budget-blocked: ${hardStop.reason}` : describeResult(result)),
|
|
324
|
+
exitCode: result.exitCode,
|
|
325
|
+
artifactPath: result.artifactPath,
|
|
326
|
+
artifactHash: result.artifactHash,
|
|
327
|
+
inputTreeHash,
|
|
328
|
+
criteriaHash,
|
|
329
|
+
relevantFiles,
|
|
330
|
+
createdAt: startedAt,
|
|
331
|
+
source: 'automated',
|
|
332
|
+
});
|
|
333
|
+
|
|
334
|
+
attempts.push({
|
|
335
|
+
attempt,
|
|
336
|
+
spanId: newId(),
|
|
337
|
+
tool,
|
|
338
|
+
command,
|
|
339
|
+
exitCode: result.exitCode,
|
|
340
|
+
durationMs: result.durationMs,
|
|
341
|
+
outputBytes: result.outputBytes,
|
|
342
|
+
artifactPath: result.artifactPath,
|
|
343
|
+
artifactHash: result.artifactHash,
|
|
344
|
+
status: result.timedOut ? 'timed-out' : (gatePassed ? 'passed' : 'failed'),
|
|
345
|
+
retryClass: classification.retryClass,
|
|
346
|
+
retryReason: classification.reason,
|
|
347
|
+
evidence,
|
|
348
|
+
});
|
|
349
|
+
|
|
350
|
+
lastResult = result;
|
|
351
|
+
|
|
352
|
+
if (hardStop.exhausted || hardStop.blocked) {
|
|
353
|
+
return finalize({
|
|
354
|
+
status: 'failed',
|
|
355
|
+
attempts,
|
|
356
|
+
tracker,
|
|
357
|
+
inputTreeHash,
|
|
358
|
+
criteriaHash,
|
|
359
|
+
stopReason: hardStop.exhausted ? 'budget-exhausted' : 'budget-blocked',
|
|
360
|
+
diagnostic: hardStop.reason,
|
|
361
|
+
});
|
|
362
|
+
}
|
|
363
|
+
|
|
364
|
+
if (gatePassed) {
|
|
365
|
+
// Enforce red-before-green for testable gates. A red record may come from
|
|
366
|
+
// prior state or from an earlier failed attempt in this same loop.
|
|
367
|
+
const inLoopRed = attempts.slice(0, -1).some((a) => a.status === 'failed');
|
|
368
|
+
if (redFirstRequired && !priorRed && !inLoopRed) {
|
|
369
|
+
return finalize({
|
|
370
|
+
status: 'failed',
|
|
371
|
+
attempts,
|
|
372
|
+
tracker,
|
|
373
|
+
inputTreeHash,
|
|
374
|
+
criteriaHash,
|
|
375
|
+
stopReason: 'red-required',
|
|
376
|
+
diagnostic: 'a failing (red) record is required before a green testsPassed result; run the test against the unimplemented behavior first',
|
|
377
|
+
});
|
|
378
|
+
}
|
|
379
|
+
return finalize({ status: 'passed', attempts, tracker, inputTreeHash, criteriaHash });
|
|
380
|
+
}
|
|
381
|
+
if (result.timedOut) {
|
|
382
|
+
if (classification.retryable && tracker.checkStepRetries(attempt - 1).status !== 'exhausted') {
|
|
383
|
+
// fall through to retry
|
|
384
|
+
} else {
|
|
385
|
+
return finalize({ status: 'timed-out', attempts, tracker, inputTreeHash, criteriaHash });
|
|
386
|
+
}
|
|
387
|
+
}
|
|
388
|
+
if (!classification.retryable) {
|
|
389
|
+
return finalize({
|
|
390
|
+
status: 'failed',
|
|
391
|
+
attempts,
|
|
392
|
+
tracker,
|
|
393
|
+
inputTreeHash,
|
|
394
|
+
criteriaHash,
|
|
395
|
+
stopReason: classification.retryClass,
|
|
396
|
+
diagnostic: classification.reason,
|
|
397
|
+
});
|
|
398
|
+
}
|
|
399
|
+
|
|
400
|
+
// Retryable — check both the per-step and total retry budgets.
|
|
401
|
+
// `attempt` counts executions; retries performed so far = attempt - 1.
|
|
402
|
+
const stepCheck = tracker.checkStepRetries(attempt - 1);
|
|
403
|
+
if (stepCheck.status === 'exhausted') {
|
|
404
|
+
return finalize({ status: 'failed', attempts, tracker, inputTreeHash, criteriaHash, stopReason: 'retry-exhausted', diagnostic: 'retry budget exhausted' });
|
|
405
|
+
}
|
|
406
|
+
const runCheck = tracker.check('retries');
|
|
407
|
+
if (runCheck.status === 'exhausted') {
|
|
408
|
+
return finalize({ status: 'failed', attempts, tracker, inputTreeHash, criteriaHash, stopReason: 'retry-exhausted', diagnostic: 'total retry budget exhausted' });
|
|
409
|
+
}
|
|
410
|
+
const timeCheck = tracker.check('wallClockMs');
|
|
411
|
+
if (timeCheck.status === 'exhausted') {
|
|
412
|
+
return finalize({ status: 'failed', attempts, tracker, inputTreeHash, criteriaHash, stopReason: 'budget-exhausted', diagnostic: 'wall-clock budget exhausted' });
|
|
413
|
+
}
|
|
414
|
+
|
|
415
|
+
tracker.add('retries', 1);
|
|
416
|
+
const backoff = TRANSIENT_BACKOFF_MS[Math.min(attempt - 1, TRANSIENT_BACKOFF_MS.length - 1)];
|
|
417
|
+
if (attempt < perStepLimit) {
|
|
418
|
+
await sleepImpl(backoff);
|
|
419
|
+
}
|
|
420
|
+
stopped = false;
|
|
421
|
+
}
|
|
422
|
+
|
|
423
|
+
return finalize({
|
|
424
|
+
status: 'failed',
|
|
425
|
+
attempts,
|
|
426
|
+
tracker,
|
|
427
|
+
inputTreeHash,
|
|
428
|
+
criteriaHash,
|
|
429
|
+
stopReason: 'retry-exhausted',
|
|
430
|
+
diagnostic: 'retry limit reached',
|
|
431
|
+
});
|
|
432
|
+
}
|
|
433
|
+
|
|
434
|
+
function describeResult(result) {
|
|
435
|
+
if (result.timedOut) return 'timed-out';
|
|
436
|
+
if (result.exitCode === 0) return 'exit 0';
|
|
437
|
+
return `exit ${result.exitCode}`;
|
|
438
|
+
}
|
|
439
|
+
|
|
440
|
+
function finalize({ status, attempts, tracker, inputTreeHash, criteriaHash, stopReason = null, diagnostic = null }) {
|
|
441
|
+
const flattened = ['passed', 'flaky'].includes(status) ? status : status;
|
|
442
|
+
return {
|
|
443
|
+
status,
|
|
444
|
+
ok: status === 'passed',
|
|
445
|
+
gateSatisfied: status === 'passed',
|
|
446
|
+
stopReason,
|
|
447
|
+
diagnostic,
|
|
448
|
+
attempts,
|
|
449
|
+
evidence: attempts.map((a) => a.evidence),
|
|
450
|
+
finalEvidence: attempts.length ? attempts[attempts.length - 1].evidence : null,
|
|
451
|
+
inputTreeHash,
|
|
452
|
+
criteriaHash,
|
|
453
|
+
budget: tracker.result(),
|
|
454
|
+
};
|
|
455
|
+
}
|
|
456
|
+
|
|
457
|
+
/**
|
|
458
|
+
* Build a manual-confirmation evidence record when automation is unavailable.
|
|
459
|
+
* Recorded as a user-owned decision, never imitated as automated evidence.
|
|
460
|
+
*/
|
|
461
|
+
export function manualConfirmation({
|
|
462
|
+
gate, workItemId, phase, projectPath, editorVersion, scope, acceptanceCriterionId = null,
|
|
463
|
+
relevantFiles = [], criteria = [], rootDir = process.cwd(), approvedBy = 'user', at = new Date(),
|
|
464
|
+
} = {}) {
|
|
465
|
+
const inputTreeHash = computeInputTreeHash(rootDir, relevantFiles);
|
|
466
|
+
const evidence = createEvidence({
|
|
467
|
+
evidenceId: newId(),
|
|
468
|
+
workItemId,
|
|
469
|
+
acceptanceCriterionId,
|
|
470
|
+
phase,
|
|
471
|
+
gate,
|
|
472
|
+
status: 'manual-confirmation',
|
|
473
|
+
command: null,
|
|
474
|
+
result: `manual confirmation: project=${projectPath} editor=${editorVersion} scope=${scope}`,
|
|
475
|
+
exitCode: null,
|
|
476
|
+
inputTreeHash,
|
|
477
|
+
criteriaHash: hashCriteria(criteria),
|
|
478
|
+
relevantFiles,
|
|
479
|
+
createdAt: at,
|
|
480
|
+
source: 'manual-confirmation',
|
|
481
|
+
});
|
|
482
|
+
return { evidence, approvedBy, projectPath, editorVersion, scope, recordedAt: timestamp(at) };
|
|
483
|
+
}
|
|
484
|
+
|
|
485
|
+
/** Convenience: is the verification result an exhaustion that must not read as success? */
|
|
486
|
+
export function isBudgetExhaustion(result) {
|
|
487
|
+
return result?.stopReason === 'budget-exhausted';
|
|
488
|
+
}
|
|
489
|
+
|
|
490
|
+
export { budgetExhaustedResult };
|