cadet-agent 0.30.0 → 0.32.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +37 -37
- package/src/cli.mjs +336 -7
- package/src/harness/index.mjs +10 -3
- package/src/harness/policy.mjs +560 -359
- package/src/harness/state.mjs +919 -661
- package/src/harness/verification.mjs +523 -490
- package/src/harness/verify-acs.mjs +291 -0
|
@@ -1,490 +1,523 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* Cadet-Agent verification runner.
|
|
3
|
-
*
|
|
4
|
-
* Deterministic, bounded verification with a single retry classifier. Skills may
|
|
5
|
-
* supply policy (commands, flaky signatures) but may not invent classifications.
|
|
6
|
-
*
|
|
7
|
-
* Contract: docs/core/HarnessContract.md §4 (retry classifier), §5 (commands), §6 (loop).
|
|
8
|
-
*/
|
|
9
|
-
|
|
10
|
-
import { spawn } from 'node:child_process';
|
|
11
|
-
import { mkdirSync, writeFileSync } from 'node:fs';
|
|
12
|
-
import { join } from 'node:path';
|
|
13
|
-
import { createEvidence, computeInputTreeHash } from './state.mjs';
|
|
14
|
-
import { hashCriteria, sha256Bytes, timestamp, newId } from './util.mjs';
|
|
15
|
-
import { BudgetTracker, budgetExhaustedResult, evaluateHardStop } from './budget.mjs';
|
|
16
|
-
import { redactString } from './redaction.mjs';
|
|
17
|
-
|
|
18
|
-
export const RETRY_CLASSES = Object.freeze(['deterministic', 'transient', 'repair', 'unknown']);
|
|
19
|
-
export const RESULT_STATUSES = Object.freeze(['passed', 'failed', 'flaky', 'blocked', 'timed-out']);
|
|
20
|
-
|
|
21
|
-
/** Backoff schedule for transient retries (ms). */
|
|
22
|
-
export const TRANSIENT_BACKOFF_MS = Object.freeze([250, 1000, 4000]);
|
|
23
|
-
|
|
24
|
-
/** Default flaky signatures (configurable via policy.estimation? no — via verification policy). */
|
|
25
|
-
export const DEFAULT_FLAKY_SIGNATURES = Object.freeze([
|
|
26
|
-
'econnreset',
|
|
27
|
-
'etimedout',
|
|
28
|
-
'socket hang up',
|
|
29
|
-
'connection refused',
|
|
30
|
-
'temporarily unavailable',
|
|
31
|
-
'service unavailable',
|
|
32
|
-
'process launch failed',
|
|
33
|
-
'eaddrnotavail',
|
|
34
|
-
]);
|
|
35
|
-
|
|
36
|
-
const DETERMINISTIC_SIGNATURES = Object.freeze([
|
|
37
|
-
'assertionerror',
|
|
38
|
-
'expected',
|
|
39
|
-
'syntaxerror',
|
|
40
|
-
'compile error',
|
|
41
|
-
'compilationerror',
|
|
42
|
-
'cs0',
|
|
43
|
-
'analyzer',
|
|
44
|
-
'unt',
|
|
45
|
-
'invalid input',
|
|
46
|
-
'usage:',
|
|
47
|
-
'unknown option',
|
|
48
|
-
'no such file',
|
|
49
|
-
'cannot find module',
|
|
50
|
-
'unhandled rejection',
|
|
51
|
-
]);
|
|
52
|
-
|
|
53
|
-
const TRANSIENT_SIGNATURES = DEFAULT_FLAKY_SIGNATURES;
|
|
54
|
-
|
|
55
|
-
/**
|
|
56
|
-
* Classify a command result. The classifier is the single source of truth for
|
|
57
|
-
* retry behavior — nothing else may decide to retry.
|
|
58
|
-
*
|
|
59
|
-
* Returns `{ retryClass, reason, retryable }`.
|
|
60
|
-
*/
|
|
61
|
-
export function classifyResult({ exitCode = null, timedOut = false, stdout = '', stderr = '', errorMessage = '', flakySignatures = DEFAULT_FLAKY_SIGNATURES } = {}) {
|
|
62
|
-
const haystack = `${errorMessage}\n${stderr}\n${stdout}`.toLowerCase();
|
|
63
|
-
|
|
64
|
-
if (timedOut) {
|
|
65
|
-
// A timeout is deterministic unless its signature is explicitly configured flaky.
|
|
66
|
-
if (flakySignatures.some((s) => haystack.includes(s))) {
|
|
67
|
-
return { retryClass: 'transient', reason: 'timeout with a configured flaky signature', retryable: true };
|
|
68
|
-
}
|
|
69
|
-
return { retryClass: 'deterministic', reason: 'reproducible timeout does not retry automatically', retryable: false };
|
|
70
|
-
}
|
|
71
|
-
|
|
72
|
-
if (TRANSIENT_SIGNATURES.some((s) => haystack.includes(s))) {
|
|
73
|
-
return { retryClass: 'transient', reason: `matched transient signature`, retryable: true };
|
|
74
|
-
}
|
|
75
|
-
|
|
76
|
-
if (exitCode === 0) {
|
|
77
|
-
return { retryClass: 'deterministic', reason: 'success', retryable: false };
|
|
78
|
-
}
|
|
79
|
-
|
|
80
|
-
if (DETERMINISTIC_SIGNATURES.some((s) => haystack.includes(s))) {
|
|
81
|
-
return { retryClass: 'deterministic', reason: 'matched a deterministic failure signature', retryable: false };
|
|
82
|
-
}
|
|
83
|
-
|
|
84
|
-
if (exitCode !== null && exitCode !== 0) {
|
|
85
|
-
return { retryClass: 'unknown', reason: `unrecognized failure (exit ${exitCode}); escalate`, retryable: false };
|
|
86
|
-
}
|
|
87
|
-
|
|
88
|
-
return { retryClass: 'unknown', reason: 'unrecognized result; escalate', retryable: false };
|
|
89
|
-
}
|
|
90
|
-
|
|
91
|
-
/** A `repair` retry is only valid when it references the failed evidence and changed files. */
|
|
92
|
-
export function classifyRepair({ failedEvidenceId, changedFiles = [] } = {}) {
|
|
93
|
-
if (!failedEvidenceId) {
|
|
94
|
-
return { retryClass: 'unknown', reason: 'repair retry requires a failed evidence reference', retryable: false };
|
|
95
|
-
}
|
|
96
|
-
if (!Array.isArray(changedFiles) || changedFiles.length === 0) {
|
|
97
|
-
return { retryClass: 'unknown', reason: 'repair retry requires changed files', retryable: false };
|
|
98
|
-
}
|
|
99
|
-
return { retryClass: 'repair', reason: 'code/config repair followed by a rerun', retryable: true, failedEvidenceId, changedFiles };
|
|
100
|
-
}
|
|
101
|
-
|
|
102
|
-
/**
|
|
103
|
-
* Run a command and capture bounded output. Never buffers unbounded output:
|
|
104
|
-
* output beyond `maxInlineBytes` is written to an artifact.
|
|
105
|
-
*
|
|
106
|
-
* Returns `{ exitCode, signal, timedOut, stdout, stderr, durationMs, outputBytes, artifactPath, artifactHash, preview }`.
|
|
107
|
-
*/
|
|
108
|
-
export function runCommand(command, {
|
|
109
|
-
cwd = process.cwd(),
|
|
110
|
-
env = {},
|
|
111
|
-
timeoutMs = 30 * 60 * 1000,
|
|
112
|
-
maxInlineBytes = 64 * 1024,
|
|
113
|
-
previewBytes = 4 * 1024,
|
|
114
|
-
artifactDir = null,
|
|
115
|
-
shell = true,
|
|
116
|
-
spawnImpl = spawn,
|
|
117
|
-
} = {}) {
|
|
118
|
-
return new Promise((resolve) => {
|
|
119
|
-
const startedAt = Date.now();
|
|
120
|
-
let child;
|
|
121
|
-
try {
|
|
122
|
-
child = spawnImpl(command, { cwd, env: { ...process.env, ...env }, shell, windowsHide: true });
|
|
123
|
-
} catch (err) {
|
|
124
|
-
resolve({
|
|
125
|
-
exitCode: null, signal: null, timedOut: false, stdout: '', stderr: '',
|
|
126
|
-
durationMs: Date.now() - startedAt, outputBytes: 0, artifactPath: null,
|
|
127
|
-
artifactHash: null, preview: '', errorMessage: err.message,
|
|
128
|
-
});
|
|
129
|
-
return;
|
|
130
|
-
}
|
|
131
|
-
|
|
132
|
-
const stdoutChunks = [];
|
|
133
|
-
const stderrChunks = [];
|
|
134
|
-
let stdoutBytes = 0;
|
|
135
|
-
let stderrBytes = 0;
|
|
136
|
-
let timedOut = false;
|
|
137
|
-
let finished = false;
|
|
138
|
-
|
|
139
|
-
const timer = setTimeout(() => {
|
|
140
|
-
timedOut = true;
|
|
141
|
-
try { child.kill('SIGKILL'); } catch { /* already gone */ }
|
|
142
|
-
}, timeoutMs);
|
|
143
|
-
|
|
144
|
-
child.stdout?.on('data', (chunk) => {
|
|
145
|
-
stdoutBytes += chunk.length;
|
|
146
|
-
if (stdoutBytes <= maxInlineBytes) stdoutChunks.push(chunk);
|
|
147
|
-
});
|
|
148
|
-
child.stderr?.on('data', (chunk) => {
|
|
149
|
-
stderrBytes += chunk.length;
|
|
150
|
-
if (stderrBytes <= maxInlineBytes) stderrChunks.push(chunk);
|
|
151
|
-
});
|
|
152
|
-
|
|
153
|
-
const finish = (exitCode, signal, errorMessage = '') => {
|
|
154
|
-
if (finished) return;
|
|
155
|
-
finished = true;
|
|
156
|
-
clearTimeout(timer);
|
|
157
|
-
const durationMs = Date.now() - startedAt;
|
|
158
|
-
const stdout = Buffer.concat(stdoutChunks).toString('utf-8');
|
|
159
|
-
const stderr = Buffer.concat(stderrChunks).toString('utf-8');
|
|
160
|
-
const outputBytes = stdoutBytes + stderrBytes;
|
|
161
|
-
|
|
162
|
-
let artifactPath = null;
|
|
163
|
-
let artifactHash = null;
|
|
164
|
-
if (outputBytes > maxInlineBytes && artifactDir) {
|
|
165
|
-
try {
|
|
166
|
-
mkdirSync(artifactDir, { recursive: true });
|
|
167
|
-
const full = join(artifactDir, `output-${newId()}.log`);
|
|
168
|
-
// Redact before persisting: the artifact must never contain secrets,
|
|
169
|
-
// even though the in-memory copy is kept raw for classification.
|
|
170
|
-
const body = redactString(`${stdout}\n${stderr}`);
|
|
171
|
-
writeFileSync(full, body, 'utf-8');
|
|
172
|
-
artifactPath = full;
|
|
173
|
-
// Hash covers the exact persisted (redacted) bytes.
|
|
174
|
-
artifactHash = sha256Bytes(Buffer.from(body, 'utf-8'));
|
|
175
|
-
} catch {
|
|
176
|
-
artifactPath = null;
|
|
177
|
-
}
|
|
178
|
-
}
|
|
179
|
-
const preview = redactString((stdout + stderr).slice(0, previewBytes));
|
|
180
|
-
resolve({
|
|
181
|
-
exitCode, signal, timedOut, stdout, stderr, durationMs, outputBytes,
|
|
182
|
-
artifactPath, artifactHash, preview, errorMessage,
|
|
183
|
-
});
|
|
184
|
-
};
|
|
185
|
-
|
|
186
|
-
child.on('error', (err) => finish(null, null, err.message));
|
|
187
|
-
child.on('close', (code, signal) => finish(code, signal));
|
|
188
|
-
});
|
|
189
|
-
}
|
|
190
|
-
|
|
191
|
-
/**
|
|
192
|
-
* Build a verification command descriptor for a gate.
|
|
193
|
-
* Project-specific commands (analyzer/compile/test) come from `.cadet/harness.json`.
|
|
194
|
-
*/
|
|
195
|
-
export function commandForGate(gate, { projectPath = '.', policy = null, unityAvailable = false } = {}) {
|
|
196
|
-
switch (gate) {
|
|
197
|
-
case 'testsPassed':
|
|
198
|
-
return policy?.testCommand
|
|
199
|
-
? { command: policy.testCommand, tool: 'test', automated: true }
|
|
200
|
-
: { command: 'npm test', tool: 'test', automated: true };
|
|
201
|
-
case 'compileCheckConfirmed':
|
|
202
|
-
if (policy?.compileCommand) {
|
|
203
|
-
return { command: policy.compileCommand, tool: 'unity-build', automated: true };
|
|
204
|
-
}
|
|
205
|
-
if (unityAvailable) {
|
|
206
|
-
return {
|
|
207
|
-
command: `unity build ${projectPath} --target StandaloneWindows64 -o "${join(projectPath, 'Temp', 'cadet-build')}" --format json`,
|
|
208
|
-
tool: 'unity-build',
|
|
209
|
-
automated: true,
|
|
210
|
-
};
|
|
211
|
-
}
|
|
212
|
-
return { command: null, tool: 'manual-confirmation', automated: false, reason: 'Unity CLI unavailable' };
|
|
213
|
-
case 'unityAnalyzerClean':
|
|
214
|
-
if (policy?.analyzerCommand) {
|
|
215
|
-
return {
|
|
216
|
-
command: `unity run ${projectPath} --command ${policy.analyzerCommand} --format json`,
|
|
217
|
-
tool: 'unity-analyzer',
|
|
218
|
-
automated: true,
|
|
219
|
-
};
|
|
220
|
-
}
|
|
221
|
-
return { command: null, tool: 'unity-analyzer', automated: false, reason: 'analyzer command not declared in .cadet/harness.json' };
|
|
222
|
-
default:
|
|
223
|
-
return { command: null, tool: 'agent-owned', automated: false, reason: `${gate} is agent-owned` };
|
|
224
|
-
}
|
|
225
|
-
}
|
|
226
|
-
|
|
227
|
-
/** Detect zero `UNT*` diagnostics in a Unity analyzer JSON envelope. */
|
|
228
|
-
export function analyzerClean(stdout) {
|
|
229
|
-
try {
|
|
230
|
-
const parsed = JSON.parse(stdout);
|
|
231
|
-
const text = JSON.stringify(parsed);
|
|
232
|
-
return !/UNT\d+/.test(text);
|
|
233
|
-
} catch {
|
|
234
|
-
// Non-JSON: treat any UNT token in the raw text as a finding.
|
|
235
|
-
return !/UNT\d+/.test(stdout);
|
|
236
|
-
}
|
|
237
|
-
}
|
|
238
|
-
|
|
239
|
-
/**
|
|
240
|
-
* The verification loop contract (contract §6). Runs a command once, classifies,
|
|
241
|
-
* retries only when the class is retryable and budget remains, and records every
|
|
242
|
-
* attempt. No attempt ever overwrites a previous one.
|
|
243
|
-
*
|
|
244
|
-
* `runner` is injectable for tests: `async (attempt) => { exitCode, ... }`.
|
|
245
|
-
*/
|
|
246
|
-
export async function runVerificationLoop({
|
|
247
|
-
gate,
|
|
248
|
-
command,
|
|
249
|
-
workItemId,
|
|
250
|
-
phase,
|
|
251
|
-
acceptanceCriterionId = null,
|
|
252
|
-
relevantFiles = [],
|
|
253
|
-
criteria = [],
|
|
254
|
-
rootDir = process.cwd(),
|
|
255
|
-
policy,
|
|
256
|
-
budgets,
|
|
257
|
-
runCommandImpl = runCommand,
|
|
258
|
-
sleepImpl = (ms) => new Promise((r) => setTimeout(r, ms)),
|
|
259
|
-
maxAttemptsOverride = null,
|
|
260
|
-
tool = 'test',
|
|
261
|
-
maxInlineBytes,
|
|
262
|
-
previewBytes,
|
|
263
|
-
artifactDir = null,
|
|
264
|
-
flakySignatures = DEFAULT_FLAKY_SIGNATURES,
|
|
265
|
-
priorEvidence = [],
|
|
266
|
-
requireRedFirst = null,
|
|
267
|
-
now = () => new Date(),
|
|
268
|
-
} = {}) {
|
|
269
|
-
const tracker = budgets || new BudgetTracker(policy);
|
|
270
|
-
const attempts = [];
|
|
271
|
-
const inputTreeHash = computeInputTreeHash(rootDir, relevantFiles);
|
|
272
|
-
const criteriaHash = hashCriteria(criteria);
|
|
273
|
-
const perStepLimit = maxAttemptsOverride ?? (policy?.budgets?.maxRetriesPerStep?.hard ?? 2) + 1;
|
|
274
|
-
|
|
275
|
-
// Red-before-green: a testable gate must not be satisfied by green evidence
|
|
276
|
-
// unless a prior failed (red) record exists for the same work item and gate.
|
|
277
|
-
// `requireRedFirst` may be set explicitly; otherwise it applies to
|
|
278
|
-
// `testsPassed` by default.
|
|
279
|
-
const redFirstRequired = requireRedFirst === null ? gate === 'testsPassed' : requireRedFirst === true;
|
|
280
|
-
const priorRed = Array.isArray(priorEvidence)
|
|
281
|
-
&& priorEvidence.some((e) => e && e.gate === gate && e.workItemId === workItemId && e.status === 'failed');
|
|
282
|
-
|
|
283
|
-
let lastResult = null;
|
|
284
|
-
let stopped = false;
|
|
285
|
-
|
|
286
|
-
for (let attempt = 1; attempt <= perStepLimit; attempt++) {
|
|
287
|
-
const startedAt = now();
|
|
288
|
-
const result = await runCommandImpl(command, { cwd: rootDir, timeoutMs: policy?.budgets?.maxWallClockMs?.hard, maxInlineBytes, previewBytes, artifactDir });
|
|
289
|
-
tracker.add('toolCalls', 1);
|
|
290
|
-
tracker.add('wallClockMs', result.durationMs || 0);
|
|
291
|
-
// Command output counts against the output-token budget (estimated from the
|
|
292
|
-
// UTF-8 byte length), so the output budget is enforced rather than advisory.
|
|
293
|
-
const bytesPerToken = policy?.estimation?.bytesPerToken || 3;
|
|
294
|
-
tracker.add('outputTokens', Math.ceil((result.outputBytes || 0) / bytesPerToken));
|
|
295
|
-
|
|
296
|
-
// Hard budgets are enforceable, not advisory: if this attempt pushed a hard
|
|
297
|
-
// limit (tool calls, wall-clock, output tokens, cost), stop immediately and
|
|
298
|
-
// never report a passing gate.
|
|
299
|
-
const hardStop = evaluateHardStop(tracker, { policy });
|
|
300
|
-
|
|
301
|
-
const classification = classifyResult({ ...result, flakySignatures });
|
|
302
|
-
const passed = result.exitCode === 0 && !result.timedOut;
|
|
303
|
-
let gatePassed = passed;
|
|
304
|
-
if (gate === 'unityAnalyzerClean' && passed) {
|
|
305
|
-
gatePassed = analyzerClean(result.stdout);
|
|
306
|
-
}
|
|
307
|
-
// A gate can never be satisfied when a hard budget was exceeded, or when a
|
|
308
|
-
// configured budget (e.g. cost) could not be measured.
|
|
309
|
-
if (hardStop.exhausted || hardStop.blocked) gatePassed = false;
|
|
310
|
-
|
|
311
|
-
const evidence = createEvidence({
|
|
312
|
-
evidenceId: newId(),
|
|
313
|
-
workItemId,
|
|
314
|
-
acceptanceCriterionId,
|
|
315
|
-
phase,
|
|
316
|
-
gate,
|
|
317
|
-
status: hardStop.exhausted || hardStop.blocked
|
|
318
|
-
? 'blocked'
|
|
319
|
-
: (gatePassed ? 'passed' : (result.timedOut ? 'blocked' : 'failed')),
|
|
320
|
-
command,
|
|
321
|
-
result: hardStop.exhausted
|
|
322
|
-
? `budget-exhausted: ${hardStop.reason}`
|
|
323
|
-
: (hardStop.blocked ? `budget-blocked: ${hardStop.reason}` : describeResult(result)),
|
|
324
|
-
exitCode: result.exitCode,
|
|
325
|
-
artifactPath: result.artifactPath,
|
|
326
|
-
artifactHash: result.artifactHash,
|
|
327
|
-
inputTreeHash,
|
|
328
|
-
criteriaHash,
|
|
329
|
-
relevantFiles,
|
|
330
|
-
createdAt: startedAt,
|
|
331
|
-
source: 'automated',
|
|
332
|
-
});
|
|
333
|
-
|
|
334
|
-
attempts.push({
|
|
335
|
-
attempt,
|
|
336
|
-
spanId: newId(),
|
|
337
|
-
tool,
|
|
338
|
-
command,
|
|
339
|
-
exitCode: result.exitCode,
|
|
340
|
-
durationMs: result.durationMs,
|
|
341
|
-
outputBytes: result.outputBytes,
|
|
342
|
-
artifactPath: result.artifactPath,
|
|
343
|
-
artifactHash: result.artifactHash,
|
|
344
|
-
status: result.timedOut ? 'timed-out' : (gatePassed ? 'passed' : 'failed'),
|
|
345
|
-
retryClass: classification.retryClass,
|
|
346
|
-
retryReason: classification.reason,
|
|
347
|
-
evidence,
|
|
348
|
-
});
|
|
349
|
-
|
|
350
|
-
lastResult = result;
|
|
351
|
-
|
|
352
|
-
if (hardStop.exhausted || hardStop.blocked) {
|
|
353
|
-
return finalize({
|
|
354
|
-
status: 'failed',
|
|
355
|
-
attempts,
|
|
356
|
-
tracker,
|
|
357
|
-
inputTreeHash,
|
|
358
|
-
criteriaHash,
|
|
359
|
-
stopReason: hardStop.exhausted ? 'budget-exhausted' : 'budget-blocked',
|
|
360
|
-
diagnostic: hardStop.reason,
|
|
361
|
-
});
|
|
362
|
-
}
|
|
363
|
-
|
|
364
|
-
if (gatePassed) {
|
|
365
|
-
// Enforce red-before-green for testable gates. A red record may come from
|
|
366
|
-
// prior state or from an earlier failed attempt in this same loop.
|
|
367
|
-
const inLoopRed = attempts.slice(0, -1).some((a) => a.status === 'failed');
|
|
368
|
-
if (redFirstRequired && !priorRed && !inLoopRed) {
|
|
369
|
-
return finalize({
|
|
370
|
-
status: 'failed',
|
|
371
|
-
attempts,
|
|
372
|
-
tracker,
|
|
373
|
-
inputTreeHash,
|
|
374
|
-
criteriaHash,
|
|
375
|
-
stopReason: 'red-required',
|
|
376
|
-
diagnostic: 'a failing (red) record is required before a green testsPassed result; run the test against the unimplemented behavior first',
|
|
377
|
-
});
|
|
378
|
-
}
|
|
379
|
-
return finalize({ status: 'passed', attempts, tracker, inputTreeHash, criteriaHash });
|
|
380
|
-
}
|
|
381
|
-
if (result.timedOut) {
|
|
382
|
-
if (classification.retryable && tracker.checkStepRetries(attempt - 1).status !== 'exhausted') {
|
|
383
|
-
// fall through to retry
|
|
384
|
-
} else {
|
|
385
|
-
return finalize({ status: 'timed-out', attempts, tracker, inputTreeHash, criteriaHash });
|
|
386
|
-
}
|
|
387
|
-
}
|
|
388
|
-
if (!classification.retryable) {
|
|
389
|
-
return finalize({
|
|
390
|
-
status: 'failed',
|
|
391
|
-
attempts,
|
|
392
|
-
tracker,
|
|
393
|
-
inputTreeHash,
|
|
394
|
-
criteriaHash,
|
|
395
|
-
stopReason: classification.retryClass,
|
|
396
|
-
diagnostic: classification.reason,
|
|
397
|
-
});
|
|
398
|
-
}
|
|
399
|
-
|
|
400
|
-
// Retryable — check both the per-step and total retry budgets.
|
|
401
|
-
// `attempt` counts executions; retries performed so far = attempt - 1.
|
|
402
|
-
const stepCheck = tracker.checkStepRetries(attempt - 1);
|
|
403
|
-
if (stepCheck.status === 'exhausted') {
|
|
404
|
-
return finalize({ status: 'failed', attempts, tracker, inputTreeHash, criteriaHash, stopReason: 'retry-exhausted', diagnostic: 'retry budget exhausted' });
|
|
405
|
-
}
|
|
406
|
-
const runCheck = tracker.check('retries');
|
|
407
|
-
if (runCheck.status === 'exhausted') {
|
|
408
|
-
return finalize({ status: 'failed', attempts, tracker, inputTreeHash, criteriaHash, stopReason: 'retry-exhausted', diagnostic: 'total retry budget exhausted' });
|
|
409
|
-
}
|
|
410
|
-
const timeCheck = tracker.check('wallClockMs');
|
|
411
|
-
if (timeCheck.status === 'exhausted') {
|
|
412
|
-
return finalize({ status: 'failed', attempts, tracker, inputTreeHash, criteriaHash, stopReason: 'budget-exhausted', diagnostic: 'wall-clock budget exhausted' });
|
|
413
|
-
}
|
|
414
|
-
|
|
415
|
-
tracker.add('retries', 1);
|
|
416
|
-
const backoff = TRANSIENT_BACKOFF_MS[Math.min(attempt - 1, TRANSIENT_BACKOFF_MS.length - 1)];
|
|
417
|
-
if (attempt < perStepLimit) {
|
|
418
|
-
await sleepImpl(backoff);
|
|
419
|
-
}
|
|
420
|
-
stopped = false;
|
|
421
|
-
}
|
|
422
|
-
|
|
423
|
-
return finalize({
|
|
424
|
-
status: 'failed',
|
|
425
|
-
attempts,
|
|
426
|
-
tracker,
|
|
427
|
-
inputTreeHash,
|
|
428
|
-
criteriaHash,
|
|
429
|
-
stopReason: 'retry-exhausted',
|
|
430
|
-
diagnostic: 'retry limit reached',
|
|
431
|
-
});
|
|
432
|
-
}
|
|
433
|
-
|
|
434
|
-
function describeResult(result) {
|
|
435
|
-
if (result.timedOut) return 'timed-out';
|
|
436
|
-
if (result.exitCode === 0) return 'exit 0';
|
|
437
|
-
return `exit ${result.exitCode}`;
|
|
438
|
-
}
|
|
439
|
-
|
|
440
|
-
function finalize({ status, attempts, tracker, inputTreeHash, criteriaHash, stopReason = null, diagnostic = null }) {
|
|
441
|
-
const flattened = ['passed', 'flaky'].includes(status) ? status : status;
|
|
442
|
-
return {
|
|
443
|
-
status,
|
|
444
|
-
ok: status === 'passed',
|
|
445
|
-
gateSatisfied: status === 'passed',
|
|
446
|
-
stopReason,
|
|
447
|
-
diagnostic,
|
|
448
|
-
attempts,
|
|
449
|
-
evidence: attempts.map((a) => a.evidence),
|
|
450
|
-
finalEvidence: attempts.length ? attempts[attempts.length - 1].evidence : null,
|
|
451
|
-
inputTreeHash,
|
|
452
|
-
criteriaHash,
|
|
453
|
-
budget: tracker.result(),
|
|
454
|
-
};
|
|
455
|
-
}
|
|
456
|
-
|
|
457
|
-
/**
|
|
458
|
-
* Build a manual-confirmation evidence record when automation is unavailable.
|
|
459
|
-
* Recorded as a user-owned decision, never imitated as automated evidence.
|
|
460
|
-
*/
|
|
461
|
-
export function manualConfirmation({
|
|
462
|
-
gate, workItemId, phase, projectPath, editorVersion, scope, acceptanceCriterionId = null,
|
|
463
|
-
relevantFiles = [], criteria = [], rootDir = process.cwd(), approvedBy = 'user', at = new Date(),
|
|
464
|
-
|
|
465
|
-
|
|
466
|
-
const
|
|
467
|
-
|
|
468
|
-
|
|
469
|
-
|
|
470
|
-
|
|
471
|
-
|
|
472
|
-
|
|
473
|
-
|
|
474
|
-
|
|
475
|
-
|
|
476
|
-
|
|
477
|
-
|
|
478
|
-
|
|
479
|
-
|
|
480
|
-
|
|
481
|
-
|
|
482
|
-
|
|
483
|
-
|
|
484
|
-
|
|
485
|
-
|
|
486
|
-
|
|
487
|
-
|
|
488
|
-
}
|
|
489
|
-
|
|
490
|
-
|
|
1
|
+
/**
|
|
2
|
+
* Cadet-Agent verification runner.
|
|
3
|
+
*
|
|
4
|
+
* Deterministic, bounded verification with a single retry classifier. Skills may
|
|
5
|
+
* supply policy (commands, flaky signatures) but may not invent classifications.
|
|
6
|
+
*
|
|
7
|
+
* Contract: docs/core/HarnessContract.md §4 (retry classifier), §5 (commands), §6 (loop).
|
|
8
|
+
*/
|
|
9
|
+
|
|
10
|
+
import { spawn } from 'node:child_process';
|
|
11
|
+
import { mkdirSync, writeFileSync } from 'node:fs';
|
|
12
|
+
import { join } from 'node:path';
|
|
13
|
+
import { createEvidence, computeInputTreeHash } from './state.mjs';
|
|
14
|
+
import { hashCriteria, sha256Bytes, timestamp, newId } from './util.mjs';
|
|
15
|
+
import { BudgetTracker, budgetExhaustedResult, evaluateHardStop } from './budget.mjs';
|
|
16
|
+
import { redactString } from './redaction.mjs';
|
|
17
|
+
|
|
18
|
+
export const RETRY_CLASSES = Object.freeze(['deterministic', 'transient', 'repair', 'unknown']);
|
|
19
|
+
export const RESULT_STATUSES = Object.freeze(['passed', 'failed', 'flaky', 'blocked', 'timed-out']);
|
|
20
|
+
|
|
21
|
+
/** Backoff schedule for transient retries (ms). */
|
|
22
|
+
export const TRANSIENT_BACKOFF_MS = Object.freeze([250, 1000, 4000]);
|
|
23
|
+
|
|
24
|
+
/** Default flaky signatures (configurable via policy.estimation? no — via verification policy). */
|
|
25
|
+
export const DEFAULT_FLAKY_SIGNATURES = Object.freeze([
|
|
26
|
+
'econnreset',
|
|
27
|
+
'etimedout',
|
|
28
|
+
'socket hang up',
|
|
29
|
+
'connection refused',
|
|
30
|
+
'temporarily unavailable',
|
|
31
|
+
'service unavailable',
|
|
32
|
+
'process launch failed',
|
|
33
|
+
'eaddrnotavail',
|
|
34
|
+
]);
|
|
35
|
+
|
|
36
|
+
const DETERMINISTIC_SIGNATURES = Object.freeze([
|
|
37
|
+
'assertionerror',
|
|
38
|
+
'expected',
|
|
39
|
+
'syntaxerror',
|
|
40
|
+
'compile error',
|
|
41
|
+
'compilationerror',
|
|
42
|
+
'cs0',
|
|
43
|
+
'analyzer',
|
|
44
|
+
'unt',
|
|
45
|
+
'invalid input',
|
|
46
|
+
'usage:',
|
|
47
|
+
'unknown option',
|
|
48
|
+
'no such file',
|
|
49
|
+
'cannot find module',
|
|
50
|
+
'unhandled rejection',
|
|
51
|
+
]);
|
|
52
|
+
|
|
53
|
+
const TRANSIENT_SIGNATURES = DEFAULT_FLAKY_SIGNATURES;
|
|
54
|
+
|
|
55
|
+
/**
|
|
56
|
+
* Classify a command result. The classifier is the single source of truth for
|
|
57
|
+
* retry behavior — nothing else may decide to retry.
|
|
58
|
+
*
|
|
59
|
+
* Returns `{ retryClass, reason, retryable }`.
|
|
60
|
+
*/
|
|
61
|
+
export function classifyResult({ exitCode = null, timedOut = false, stdout = '', stderr = '', errorMessage = '', flakySignatures = DEFAULT_FLAKY_SIGNATURES } = {}) {
|
|
62
|
+
const haystack = `${errorMessage}\n${stderr}\n${stdout}`.toLowerCase();
|
|
63
|
+
|
|
64
|
+
if (timedOut) {
|
|
65
|
+
// A timeout is deterministic unless its signature is explicitly configured flaky.
|
|
66
|
+
if (flakySignatures.some((s) => haystack.includes(s))) {
|
|
67
|
+
return { retryClass: 'transient', reason: 'timeout with a configured flaky signature', retryable: true };
|
|
68
|
+
}
|
|
69
|
+
return { retryClass: 'deterministic', reason: 'reproducible timeout does not retry automatically', retryable: false };
|
|
70
|
+
}
|
|
71
|
+
|
|
72
|
+
if (TRANSIENT_SIGNATURES.some((s) => haystack.includes(s))) {
|
|
73
|
+
return { retryClass: 'transient', reason: `matched transient signature`, retryable: true };
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
if (exitCode === 0) {
|
|
77
|
+
return { retryClass: 'deterministic', reason: 'success', retryable: false };
|
|
78
|
+
}
|
|
79
|
+
|
|
80
|
+
if (DETERMINISTIC_SIGNATURES.some((s) => haystack.includes(s))) {
|
|
81
|
+
return { retryClass: 'deterministic', reason: 'matched a deterministic failure signature', retryable: false };
|
|
82
|
+
}
|
|
83
|
+
|
|
84
|
+
if (exitCode !== null && exitCode !== 0) {
|
|
85
|
+
return { retryClass: 'unknown', reason: `unrecognized failure (exit ${exitCode}); escalate`, retryable: false };
|
|
86
|
+
}
|
|
87
|
+
|
|
88
|
+
return { retryClass: 'unknown', reason: 'unrecognized result; escalate', retryable: false };
|
|
89
|
+
}
|
|
90
|
+
|
|
91
|
+
/** A `repair` retry is only valid when it references the failed evidence and changed files. */
|
|
92
|
+
export function classifyRepair({ failedEvidenceId, changedFiles = [] } = {}) {
|
|
93
|
+
if (!failedEvidenceId) {
|
|
94
|
+
return { retryClass: 'unknown', reason: 'repair retry requires a failed evidence reference', retryable: false };
|
|
95
|
+
}
|
|
96
|
+
if (!Array.isArray(changedFiles) || changedFiles.length === 0) {
|
|
97
|
+
return { retryClass: 'unknown', reason: 'repair retry requires changed files', retryable: false };
|
|
98
|
+
}
|
|
99
|
+
return { retryClass: 'repair', reason: 'code/config repair followed by a rerun', retryable: true, failedEvidenceId, changedFiles };
|
|
100
|
+
}
|
|
101
|
+
|
|
102
|
+
/**
|
|
103
|
+
* Run a command and capture bounded output. Never buffers unbounded output:
|
|
104
|
+
* output beyond `maxInlineBytes` is written to an artifact.
|
|
105
|
+
*
|
|
106
|
+
* Returns `{ exitCode, signal, timedOut, stdout, stderr, durationMs, outputBytes, artifactPath, artifactHash, preview }`.
|
|
107
|
+
*/
|
|
108
|
+
export function runCommand(command, {
|
|
109
|
+
cwd = process.cwd(),
|
|
110
|
+
env = {},
|
|
111
|
+
timeoutMs = 30 * 60 * 1000,
|
|
112
|
+
maxInlineBytes = 64 * 1024,
|
|
113
|
+
previewBytes = 4 * 1024,
|
|
114
|
+
artifactDir = null,
|
|
115
|
+
shell = true,
|
|
116
|
+
spawnImpl = spawn,
|
|
117
|
+
} = {}) {
|
|
118
|
+
return new Promise((resolve) => {
|
|
119
|
+
const startedAt = Date.now();
|
|
120
|
+
let child;
|
|
121
|
+
try {
|
|
122
|
+
child = spawnImpl(command, { cwd, env: { ...process.env, ...env }, shell, windowsHide: true });
|
|
123
|
+
} catch (err) {
|
|
124
|
+
resolve({
|
|
125
|
+
exitCode: null, signal: null, timedOut: false, stdout: '', stderr: '',
|
|
126
|
+
durationMs: Date.now() - startedAt, outputBytes: 0, artifactPath: null,
|
|
127
|
+
artifactHash: null, preview: '', errorMessage: err.message,
|
|
128
|
+
});
|
|
129
|
+
return;
|
|
130
|
+
}
|
|
131
|
+
|
|
132
|
+
const stdoutChunks = [];
|
|
133
|
+
const stderrChunks = [];
|
|
134
|
+
let stdoutBytes = 0;
|
|
135
|
+
let stderrBytes = 0;
|
|
136
|
+
let timedOut = false;
|
|
137
|
+
let finished = false;
|
|
138
|
+
|
|
139
|
+
const timer = setTimeout(() => {
|
|
140
|
+
timedOut = true;
|
|
141
|
+
try { child.kill('SIGKILL'); } catch { /* already gone */ }
|
|
142
|
+
}, timeoutMs);
|
|
143
|
+
|
|
144
|
+
child.stdout?.on('data', (chunk) => {
|
|
145
|
+
stdoutBytes += chunk.length;
|
|
146
|
+
if (stdoutBytes <= maxInlineBytes) stdoutChunks.push(chunk);
|
|
147
|
+
});
|
|
148
|
+
child.stderr?.on('data', (chunk) => {
|
|
149
|
+
stderrBytes += chunk.length;
|
|
150
|
+
if (stderrBytes <= maxInlineBytes) stderrChunks.push(chunk);
|
|
151
|
+
});
|
|
152
|
+
|
|
153
|
+
const finish = (exitCode, signal, errorMessage = '') => {
|
|
154
|
+
if (finished) return;
|
|
155
|
+
finished = true;
|
|
156
|
+
clearTimeout(timer);
|
|
157
|
+
const durationMs = Date.now() - startedAt;
|
|
158
|
+
const stdout = Buffer.concat(stdoutChunks).toString('utf-8');
|
|
159
|
+
const stderr = Buffer.concat(stderrChunks).toString('utf-8');
|
|
160
|
+
const outputBytes = stdoutBytes + stderrBytes;
|
|
161
|
+
|
|
162
|
+
let artifactPath = null;
|
|
163
|
+
let artifactHash = null;
|
|
164
|
+
if (outputBytes > maxInlineBytes && artifactDir) {
|
|
165
|
+
try {
|
|
166
|
+
mkdirSync(artifactDir, { recursive: true });
|
|
167
|
+
const full = join(artifactDir, `output-${newId()}.log`);
|
|
168
|
+
// Redact before persisting: the artifact must never contain secrets,
|
|
169
|
+
// even though the in-memory copy is kept raw for classification.
|
|
170
|
+
const body = redactString(`${stdout}\n${stderr}`);
|
|
171
|
+
writeFileSync(full, body, 'utf-8');
|
|
172
|
+
artifactPath = full;
|
|
173
|
+
// Hash covers the exact persisted (redacted) bytes.
|
|
174
|
+
artifactHash = sha256Bytes(Buffer.from(body, 'utf-8'));
|
|
175
|
+
} catch {
|
|
176
|
+
artifactPath = null;
|
|
177
|
+
}
|
|
178
|
+
}
|
|
179
|
+
const preview = redactString((stdout + stderr).slice(0, previewBytes));
|
|
180
|
+
resolve({
|
|
181
|
+
exitCode, signal, timedOut, stdout, stderr, durationMs, outputBytes,
|
|
182
|
+
artifactPath, artifactHash, preview, errorMessage,
|
|
183
|
+
});
|
|
184
|
+
};
|
|
185
|
+
|
|
186
|
+
child.on('error', (err) => finish(null, null, err.message));
|
|
187
|
+
child.on('close', (code, signal) => finish(code, signal));
|
|
188
|
+
});
|
|
189
|
+
}
|
|
190
|
+
|
|
191
|
+
/**
|
|
192
|
+
* Build a verification command descriptor for a gate.
|
|
193
|
+
* Project-specific commands (analyzer/compile/test) come from `.cadet/harness.json`.
|
|
194
|
+
*/
|
|
195
|
+
export function commandForGate(gate, { projectPath = '.', policy = null, unityAvailable = false } = {}) {
|
|
196
|
+
switch (gate) {
|
|
197
|
+
case 'testsPassed':
|
|
198
|
+
return policy?.testCommand
|
|
199
|
+
? { command: policy.testCommand, tool: 'test', automated: true }
|
|
200
|
+
: { command: 'npm test', tool: 'test', automated: true };
|
|
201
|
+
case 'compileCheckConfirmed':
|
|
202
|
+
if (policy?.compileCommand) {
|
|
203
|
+
return { command: policy.compileCommand, tool: 'unity-build', automated: true };
|
|
204
|
+
}
|
|
205
|
+
if (unityAvailable) {
|
|
206
|
+
return {
|
|
207
|
+
command: `unity build ${projectPath} --target StandaloneWindows64 -o "${join(projectPath, 'Temp', 'cadet-build')}" --format json`,
|
|
208
|
+
tool: 'unity-build',
|
|
209
|
+
automated: true,
|
|
210
|
+
};
|
|
211
|
+
}
|
|
212
|
+
return { command: null, tool: 'manual-confirmation', automated: false, reason: 'Unity CLI unavailable' };
|
|
213
|
+
case 'unityAnalyzerClean':
|
|
214
|
+
if (policy?.analyzerCommand) {
|
|
215
|
+
return {
|
|
216
|
+
command: `unity run ${projectPath} --command ${policy.analyzerCommand} --format json`,
|
|
217
|
+
tool: 'unity-analyzer',
|
|
218
|
+
automated: true,
|
|
219
|
+
};
|
|
220
|
+
}
|
|
221
|
+
return { command: null, tool: 'unity-analyzer', automated: false, reason: 'analyzer command not declared in .cadet/harness.json' };
|
|
222
|
+
default:
|
|
223
|
+
return { command: null, tool: 'agent-owned', automated: false, reason: `${gate} is agent-owned` };
|
|
224
|
+
}
|
|
225
|
+
}
|
|
226
|
+
|
|
227
|
+
/** Detect zero `UNT*` diagnostics in a Unity analyzer JSON envelope. */
|
|
228
|
+
export function analyzerClean(stdout) {
|
|
229
|
+
try {
|
|
230
|
+
const parsed = JSON.parse(stdout);
|
|
231
|
+
const text = JSON.stringify(parsed);
|
|
232
|
+
return !/UNT\d+/.test(text);
|
|
233
|
+
} catch {
|
|
234
|
+
// Non-JSON: treat any UNT token in the raw text as a finding.
|
|
235
|
+
return !/UNT\d+/.test(stdout);
|
|
236
|
+
}
|
|
237
|
+
}
|
|
238
|
+
|
|
239
|
+
/**
|
|
240
|
+
* The verification loop contract (contract §6). Runs a command once, classifies,
|
|
241
|
+
* retries only when the class is retryable and budget remains, and records every
|
|
242
|
+
* attempt. No attempt ever overwrites a previous one.
|
|
243
|
+
*
|
|
244
|
+
* `runner` is injectable for tests: `async (attempt) => { exitCode, ... }`.
|
|
245
|
+
*/
|
|
246
|
+
export async function runVerificationLoop({
|
|
247
|
+
gate,
|
|
248
|
+
command,
|
|
249
|
+
workItemId,
|
|
250
|
+
phase,
|
|
251
|
+
acceptanceCriterionId = null,
|
|
252
|
+
relevantFiles = [],
|
|
253
|
+
criteria = [],
|
|
254
|
+
rootDir = process.cwd(),
|
|
255
|
+
policy,
|
|
256
|
+
budgets,
|
|
257
|
+
runCommandImpl = runCommand,
|
|
258
|
+
sleepImpl = (ms) => new Promise((r) => setTimeout(r, ms)),
|
|
259
|
+
maxAttemptsOverride = null,
|
|
260
|
+
tool = 'test',
|
|
261
|
+
maxInlineBytes,
|
|
262
|
+
previewBytes,
|
|
263
|
+
artifactDir = null,
|
|
264
|
+
flakySignatures = DEFAULT_FLAKY_SIGNATURES,
|
|
265
|
+
priorEvidence = [],
|
|
266
|
+
requireRedFirst = null,
|
|
267
|
+
now = () => new Date(),
|
|
268
|
+
} = {}) {
|
|
269
|
+
const tracker = budgets || new BudgetTracker(policy);
|
|
270
|
+
const attempts = [];
|
|
271
|
+
const inputTreeHash = computeInputTreeHash(rootDir, relevantFiles);
|
|
272
|
+
const criteriaHash = hashCriteria(criteria);
|
|
273
|
+
const perStepLimit = maxAttemptsOverride ?? (policy?.budgets?.maxRetriesPerStep?.hard ?? 2) + 1;
|
|
274
|
+
|
|
275
|
+
// Red-before-green: a testable gate must not be satisfied by green evidence
|
|
276
|
+
// unless a prior failed (red) record exists for the same work item and gate.
|
|
277
|
+
// `requireRedFirst` may be set explicitly; otherwise it applies to
|
|
278
|
+
// `testsPassed` by default.
|
|
279
|
+
const redFirstRequired = requireRedFirst === null ? gate === 'testsPassed' : requireRedFirst === true;
|
|
280
|
+
const priorRed = Array.isArray(priorEvidence)
|
|
281
|
+
&& priorEvidence.some((e) => e && e.gate === gate && e.workItemId === workItemId && e.status === 'failed');
|
|
282
|
+
|
|
283
|
+
let lastResult = null;
|
|
284
|
+
let stopped = false;
|
|
285
|
+
|
|
286
|
+
for (let attempt = 1; attempt <= perStepLimit; attempt++) {
|
|
287
|
+
const startedAt = now();
|
|
288
|
+
const result = await runCommandImpl(command, { cwd: rootDir, timeoutMs: policy?.budgets?.maxWallClockMs?.hard, maxInlineBytes, previewBytes, artifactDir });
|
|
289
|
+
tracker.add('toolCalls', 1);
|
|
290
|
+
tracker.add('wallClockMs', result.durationMs || 0);
|
|
291
|
+
// Command output counts against the output-token budget (estimated from the
|
|
292
|
+
// UTF-8 byte length), so the output budget is enforced rather than advisory.
|
|
293
|
+
const bytesPerToken = policy?.estimation?.bytesPerToken || 3;
|
|
294
|
+
tracker.add('outputTokens', Math.ceil((result.outputBytes || 0) / bytesPerToken));
|
|
295
|
+
|
|
296
|
+
// Hard budgets are enforceable, not advisory: if this attempt pushed a hard
|
|
297
|
+
// limit (tool calls, wall-clock, output tokens, cost), stop immediately and
|
|
298
|
+
// never report a passing gate.
|
|
299
|
+
const hardStop = evaluateHardStop(tracker, { policy });
|
|
300
|
+
|
|
301
|
+
const classification = classifyResult({ ...result, flakySignatures });
|
|
302
|
+
const passed = result.exitCode === 0 && !result.timedOut;
|
|
303
|
+
let gatePassed = passed;
|
|
304
|
+
if (gate === 'unityAnalyzerClean' && passed) {
|
|
305
|
+
gatePassed = analyzerClean(result.stdout);
|
|
306
|
+
}
|
|
307
|
+
// A gate can never be satisfied when a hard budget was exceeded, or when a
|
|
308
|
+
// configured budget (e.g. cost) could not be measured.
|
|
309
|
+
if (hardStop.exhausted || hardStop.blocked) gatePassed = false;
|
|
310
|
+
|
|
311
|
+
const evidence = createEvidence({
|
|
312
|
+
evidenceId: newId(),
|
|
313
|
+
workItemId,
|
|
314
|
+
acceptanceCriterionId,
|
|
315
|
+
phase,
|
|
316
|
+
gate,
|
|
317
|
+
status: hardStop.exhausted || hardStop.blocked
|
|
318
|
+
? 'blocked'
|
|
319
|
+
: (gatePassed ? 'passed' : (result.timedOut ? 'blocked' : 'failed')),
|
|
320
|
+
command,
|
|
321
|
+
result: hardStop.exhausted
|
|
322
|
+
? `budget-exhausted: ${hardStop.reason}`
|
|
323
|
+
: (hardStop.blocked ? `budget-blocked: ${hardStop.reason}` : describeResult(result)),
|
|
324
|
+
exitCode: result.exitCode,
|
|
325
|
+
artifactPath: result.artifactPath,
|
|
326
|
+
artifactHash: result.artifactHash,
|
|
327
|
+
inputTreeHash,
|
|
328
|
+
criteriaHash,
|
|
329
|
+
relevantFiles,
|
|
330
|
+
createdAt: startedAt,
|
|
331
|
+
source: 'automated',
|
|
332
|
+
});
|
|
333
|
+
|
|
334
|
+
attempts.push({
|
|
335
|
+
attempt,
|
|
336
|
+
spanId: newId(),
|
|
337
|
+
tool,
|
|
338
|
+
command,
|
|
339
|
+
exitCode: result.exitCode,
|
|
340
|
+
durationMs: result.durationMs,
|
|
341
|
+
outputBytes: result.outputBytes,
|
|
342
|
+
artifactPath: result.artifactPath,
|
|
343
|
+
artifactHash: result.artifactHash,
|
|
344
|
+
status: result.timedOut ? 'timed-out' : (gatePassed ? 'passed' : 'failed'),
|
|
345
|
+
retryClass: classification.retryClass,
|
|
346
|
+
retryReason: classification.reason,
|
|
347
|
+
evidence,
|
|
348
|
+
});
|
|
349
|
+
|
|
350
|
+
lastResult = result;
|
|
351
|
+
|
|
352
|
+
if (hardStop.exhausted || hardStop.blocked) {
|
|
353
|
+
return finalize({
|
|
354
|
+
status: 'failed',
|
|
355
|
+
attempts,
|
|
356
|
+
tracker,
|
|
357
|
+
inputTreeHash,
|
|
358
|
+
criteriaHash,
|
|
359
|
+
stopReason: hardStop.exhausted ? 'budget-exhausted' : 'budget-blocked',
|
|
360
|
+
diagnostic: hardStop.reason,
|
|
361
|
+
});
|
|
362
|
+
}
|
|
363
|
+
|
|
364
|
+
if (gatePassed) {
|
|
365
|
+
// Enforce red-before-green for testable gates. A red record may come from
|
|
366
|
+
// prior state or from an earlier failed attempt in this same loop.
|
|
367
|
+
const inLoopRed = attempts.slice(0, -1).some((a) => a.status === 'failed');
|
|
368
|
+
if (redFirstRequired && !priorRed && !inLoopRed) {
|
|
369
|
+
return finalize({
|
|
370
|
+
status: 'failed',
|
|
371
|
+
attempts,
|
|
372
|
+
tracker,
|
|
373
|
+
inputTreeHash,
|
|
374
|
+
criteriaHash,
|
|
375
|
+
stopReason: 'red-required',
|
|
376
|
+
diagnostic: 'a failing (red) record is required before a green testsPassed result; run the test against the unimplemented behavior first',
|
|
377
|
+
});
|
|
378
|
+
}
|
|
379
|
+
return finalize({ status: 'passed', attempts, tracker, inputTreeHash, criteriaHash });
|
|
380
|
+
}
|
|
381
|
+
if (result.timedOut) {
|
|
382
|
+
if (classification.retryable && tracker.checkStepRetries(attempt - 1).status !== 'exhausted') {
|
|
383
|
+
// fall through to retry
|
|
384
|
+
} else {
|
|
385
|
+
return finalize({ status: 'timed-out', attempts, tracker, inputTreeHash, criteriaHash });
|
|
386
|
+
}
|
|
387
|
+
}
|
|
388
|
+
if (!classification.retryable) {
|
|
389
|
+
return finalize({
|
|
390
|
+
status: 'failed',
|
|
391
|
+
attempts,
|
|
392
|
+
tracker,
|
|
393
|
+
inputTreeHash,
|
|
394
|
+
criteriaHash,
|
|
395
|
+
stopReason: classification.retryClass,
|
|
396
|
+
diagnostic: classification.reason,
|
|
397
|
+
});
|
|
398
|
+
}
|
|
399
|
+
|
|
400
|
+
// Retryable — check both the per-step and total retry budgets.
|
|
401
|
+
// `attempt` counts executions; retries performed so far = attempt - 1.
|
|
402
|
+
const stepCheck = tracker.checkStepRetries(attempt - 1);
|
|
403
|
+
if (stepCheck.status === 'exhausted') {
|
|
404
|
+
return finalize({ status: 'failed', attempts, tracker, inputTreeHash, criteriaHash, stopReason: 'retry-exhausted', diagnostic: 'retry budget exhausted' });
|
|
405
|
+
}
|
|
406
|
+
const runCheck = tracker.check('retries');
|
|
407
|
+
if (runCheck.status === 'exhausted') {
|
|
408
|
+
return finalize({ status: 'failed', attempts, tracker, inputTreeHash, criteriaHash, stopReason: 'retry-exhausted', diagnostic: 'total retry budget exhausted' });
|
|
409
|
+
}
|
|
410
|
+
const timeCheck = tracker.check('wallClockMs');
|
|
411
|
+
if (timeCheck.status === 'exhausted') {
|
|
412
|
+
return finalize({ status: 'failed', attempts, tracker, inputTreeHash, criteriaHash, stopReason: 'budget-exhausted', diagnostic: 'wall-clock budget exhausted' });
|
|
413
|
+
}
|
|
414
|
+
|
|
415
|
+
tracker.add('retries', 1);
|
|
416
|
+
const backoff = TRANSIENT_BACKOFF_MS[Math.min(attempt - 1, TRANSIENT_BACKOFF_MS.length - 1)];
|
|
417
|
+
if (attempt < perStepLimit) {
|
|
418
|
+
await sleepImpl(backoff);
|
|
419
|
+
}
|
|
420
|
+
stopped = false;
|
|
421
|
+
}
|
|
422
|
+
|
|
423
|
+
return finalize({
|
|
424
|
+
status: 'failed',
|
|
425
|
+
attempts,
|
|
426
|
+
tracker,
|
|
427
|
+
inputTreeHash,
|
|
428
|
+
criteriaHash,
|
|
429
|
+
stopReason: 'retry-exhausted',
|
|
430
|
+
diagnostic: 'retry limit reached',
|
|
431
|
+
});
|
|
432
|
+
}
|
|
433
|
+
|
|
434
|
+
function describeResult(result) {
|
|
435
|
+
if (result.timedOut) return 'timed-out';
|
|
436
|
+
if (result.exitCode === 0) return 'exit 0';
|
|
437
|
+
return `exit ${result.exitCode}`;
|
|
438
|
+
}
|
|
439
|
+
|
|
440
|
+
function finalize({ status, attempts, tracker, inputTreeHash, criteriaHash, stopReason = null, diagnostic = null }) {
|
|
441
|
+
const flattened = ['passed', 'flaky'].includes(status) ? status : status;
|
|
442
|
+
return {
|
|
443
|
+
status,
|
|
444
|
+
ok: status === 'passed',
|
|
445
|
+
gateSatisfied: status === 'passed',
|
|
446
|
+
stopReason,
|
|
447
|
+
diagnostic,
|
|
448
|
+
attempts,
|
|
449
|
+
evidence: attempts.map((a) => a.evidence),
|
|
450
|
+
finalEvidence: attempts.length ? attempts[attempts.length - 1].evidence : null,
|
|
451
|
+
inputTreeHash,
|
|
452
|
+
criteriaHash,
|
|
453
|
+
budget: tracker.result(),
|
|
454
|
+
};
|
|
455
|
+
}
|
|
456
|
+
|
|
457
|
+
/**
|
|
458
|
+
* Build a manual-confirmation evidence record when automation is unavailable.
|
|
459
|
+
* Recorded as a user-owned decision, never imitated as automated evidence.
|
|
460
|
+
*/
|
|
461
|
+
export function manualConfirmation({
|
|
462
|
+
gate, workItemId, phase, projectPath, editorVersion, scope, acceptanceCriterionId = null,
|
|
463
|
+
relevantFiles = [], criteria = [], rootDir = process.cwd(), approvedBy = 'user', at = new Date(),
|
|
464
|
+
reason = null, expiresAt = null, environment = null, expiresInMs = null,
|
|
465
|
+
} = {}) {
|
|
466
|
+
const inputTreeHash = computeInputTreeHash(rootDir, relevantFiles);
|
|
467
|
+
// v3 quality fields. `scope` is declared both as the free-text `result` line
|
|
468
|
+
// (v2 shape, kept for audit) and as a machine-checkable array when supplied.
|
|
469
|
+
//
|
|
470
|
+
// SECURITY: `reason`, `result`, `scope` and the environment values are free
|
|
471
|
+
// human prose and are persisted into state.json, which is committed. The run
|
|
472
|
+
// ledger is redacted; state must be too, or a pasted token ends up in git.
|
|
473
|
+
// Redaction has no bypass (contract §8).
|
|
474
|
+
const scopeList = (Array.isArray(scope) ? scope : (scope ? [scope] : [])).map(redactString);
|
|
475
|
+
const safeReason = reason === null || reason === undefined ? null : redactString(String(reason));
|
|
476
|
+
const env = Object.fromEntries(
|
|
477
|
+
Object.entries(environment || (projectPath || editorVersion
|
|
478
|
+
? { projectPath: projectPath || null, editorVersion: editorVersion || null }
|
|
479
|
+
: {})).map(([k, v]) => [k, typeof v === 'string' ? redactString(v) : v]),
|
|
480
|
+
);
|
|
481
|
+
const hasEnv = Object.keys(env).length > 0;
|
|
482
|
+
const expiry = expiresAt || (expiresInMs ? new Date(at.getTime() + expiresInMs) : null);
|
|
483
|
+
const resultText = [
|
|
484
|
+
'manual confirmation:',
|
|
485
|
+
projectPath ? `project=${redactString(String(projectPath))}` : null,
|
|
486
|
+
editorVersion ? `editor=${redactString(String(editorVersion))}` : null,
|
|
487
|
+
scopeList.length ? `scope=${scopeList.join('; ')}` : null,
|
|
488
|
+
safeReason ? `reason=${safeReason}` : null,
|
|
489
|
+
].filter(Boolean).join(' ');
|
|
490
|
+
|
|
491
|
+
const evidence = {
|
|
492
|
+
...createEvidence({
|
|
493
|
+
evidenceId: newId(),
|
|
494
|
+
workItemId,
|
|
495
|
+
acceptanceCriterionId,
|
|
496
|
+
phase,
|
|
497
|
+
gate,
|
|
498
|
+
status: 'manual-confirmation',
|
|
499
|
+
command: null,
|
|
500
|
+
result: resultText,
|
|
501
|
+
exitCode: null,
|
|
502
|
+
inputTreeHash,
|
|
503
|
+
criteriaHash: hashCriteria(criteria),
|
|
504
|
+
relevantFiles,
|
|
505
|
+
createdAt: at,
|
|
506
|
+
expiresAt: expiry,
|
|
507
|
+
source: 'manual-confirmation',
|
|
508
|
+
}),
|
|
509
|
+
// Present only when supplied, so a v2-shaped record is unchanged when the
|
|
510
|
+
// caller does not ask for the v3 fields. Values are already redacted above.
|
|
511
|
+
...(safeReason !== null ? { reason: safeReason } : {}),
|
|
512
|
+
...(hasEnv ? { environment: env } : {}),
|
|
513
|
+
...(scopeList.length ? { scope: scopeList } : {}),
|
|
514
|
+
};
|
|
515
|
+
return { evidence, approvedBy, projectPath, editorVersion, scope: scopeList, reason: safeReason, environment: env, recordedAt: timestamp(at) };
|
|
516
|
+
}
|
|
517
|
+
|
|
518
|
+
/** Convenience: is the verification result an exhaustion that must not read as success? */
|
|
519
|
+
export function isBudgetExhaustion(result) {
|
|
520
|
+
return result?.stopReason === 'budget-exhausted';
|
|
521
|
+
}
|
|
522
|
+
|
|
523
|
+
export { budgetExhaustedResult };
|