cadet-agent 0.20.2 → 0.24.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +46 -6
- package/package.json +35 -35
- package/src/cli.mjs +373 -33
- package/src/harness/archive.mjs +242 -0
- package/src/harness/budget.mjs +298 -0
- package/src/harness/context.mjs +229 -0
- package/src/harness/hook.mjs +147 -0
- package/src/harness/index.mjs +60 -0
- package/src/harness/ledger.mjs +309 -0
- package/src/harness/policy.mjs +359 -0
- package/src/harness/redaction.mjs +133 -0
- package/src/harness/routing.mjs +153 -0
- package/src/harness/state.mjs +646 -0
- package/src/harness/util.mjs +149 -0
- package/src/harness/verification.mjs +490 -0
- package/src/install.mjs +92 -154
|
@@ -0,0 +1,309 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Cadet-Agent execution ledger.
|
|
3
|
+
*
|
|
4
|
+
* Creates and persists sanitized run records: spans, evidence, decisions,
|
|
5
|
+
* budget accounting, and capability labels. All structured values pass through
|
|
6
|
+
* redaction before persistence, and output beyond the inline limit is written to
|
|
7
|
+
* a bounded artifact with a diagnostic preview.
|
|
8
|
+
*
|
|
9
|
+
* Contract: docs/core/HarnessContract.md §4 (accounting), §8 (privacy), §9 (retention).
|
|
10
|
+
*/
|
|
11
|
+
|
|
12
|
+
import {
|
|
13
|
+
mkdirSync, writeFileSync, readFileSync, existsSync, readdirSync, unlinkSync,
|
|
14
|
+
} from 'node:fs';
|
|
15
|
+
import { join } from 'node:path';
|
|
16
|
+
import { newId, timestamp, sha256Bytes } from './util.mjs';
|
|
17
|
+
import { redact, redactString, REDACTED } from './redaction.mjs';
|
|
18
|
+
import { BudgetTracker, normalizeUsage, estimateCost, budgetReport } from './budget.mjs';
|
|
19
|
+
import { writeJsonAtomic } from './state.mjs';
|
|
20
|
+
|
|
21
|
+
export const RUN_SCHEMA_VERSION = 2;
|
|
22
|
+
|
|
23
|
+
class LedgerError extends Error {
|
|
24
|
+
constructor(message) {
|
|
25
|
+
super(message);
|
|
26
|
+
this.name = 'LedgerError';
|
|
27
|
+
}
|
|
28
|
+
}
|
|
29
|
+
|
|
30
|
+
export function runsDir(targetDir) {
|
|
31
|
+
return join(targetDir, '.cadet', 'runs');
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
/** A single run: created empty, appended to, then finalized. */
|
|
35
|
+
export class RunLedger {
|
|
36
|
+
constructor({ targetDir = process.cwd(), policy, runId = null, workItemId = null, phase = null, budgets = null, capabilities = null } = {}) {
|
|
37
|
+
if (!policy) throw new LedgerError('RunLedger requires a resolved policy');
|
|
38
|
+
this.targetDir = targetDir;
|
|
39
|
+
this.policy = policy;
|
|
40
|
+
this.runId = runId || newId();
|
|
41
|
+
this.workItemId = workItemId;
|
|
42
|
+
this.phase = phase;
|
|
43
|
+
this.tracker = budgets || new BudgetTracker(policy);
|
|
44
|
+
this.capabilities = capabilities;
|
|
45
|
+
this.startedAt = timestamp();
|
|
46
|
+
this.finishedAt = null;
|
|
47
|
+
this.spans = [];
|
|
48
|
+
this.evidence = [];
|
|
49
|
+
this.decisions = [];
|
|
50
|
+
this.usage = null;
|
|
51
|
+
this.status = 'running';
|
|
52
|
+
this.retention = {
|
|
53
|
+
rawPromptRetained: false,
|
|
54
|
+
keepOnFailure: policy.retention?.keepOnFailure ?? true,
|
|
55
|
+
};
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
/** Append a sanitized span. Never stores secrets. */
|
|
59
|
+
addSpan(span) {
|
|
60
|
+
const safe = redact({ ...span, runId: this.runId, spanId: span.spanId || newId() });
|
|
61
|
+
this.spans.push(safe);
|
|
62
|
+
return safe;
|
|
63
|
+
}
|
|
64
|
+
|
|
65
|
+
/** Append an evidence record (already structured; redacted defensively). */
|
|
66
|
+
addEvidence(evidence) {
|
|
67
|
+
const safe = redact(evidence);
|
|
68
|
+
this.evidence.push(safe);
|
|
69
|
+
return safe;
|
|
70
|
+
}
|
|
71
|
+
|
|
72
|
+
/** Append a decision (transition, escalation, budget override, stop). */
|
|
73
|
+
addDecision(decision) {
|
|
74
|
+
const safe = redact({ decisionId: decision.decisionId || newId(), runId: this.runId, createdAt: decision.createdAt || timestamp(), ...decision });
|
|
75
|
+
this.decisions.push(safe);
|
|
76
|
+
return safe;
|
|
77
|
+
}
|
|
78
|
+
|
|
79
|
+
/** Record provider or estimated usage and update the cost budget. */
|
|
80
|
+
recordUsage(usage) {
|
|
81
|
+
const normalized = normalizeUsage(usage, this.policy);
|
|
82
|
+
const cost = estimateCost(normalized.inputTokens ?? 0, normalized.outputTokens ?? 0, this.policy);
|
|
83
|
+
this.usage = {
|
|
84
|
+
...normalized,
|
|
85
|
+
usd: cost.known ? cost.usd : null,
|
|
86
|
+
model: cost.model,
|
|
87
|
+
rateCardId: cost.known ? cost.rateCardId : null,
|
|
88
|
+
effectiveDate: cost.known ? cost.effectiveDate : null,
|
|
89
|
+
};
|
|
90
|
+
if (cost.known) {
|
|
91
|
+
this.tracker.set('estimatedCostUsd', cost.usd);
|
|
92
|
+
} else {
|
|
93
|
+
// Unknown cost must not silently satisfy the cost envelope: mark the
|
|
94
|
+
// counter unmeasurable so a configured cost budget cannot be confirmed.
|
|
95
|
+
this.tracker.markUnknown('estimatedCostUsd', 'no provider rate card available');
|
|
96
|
+
this.usage.costUnknown = true;
|
|
97
|
+
}
|
|
98
|
+
return this.usage;
|
|
99
|
+
}
|
|
100
|
+
|
|
101
|
+
/**
|
|
102
|
+
* Record bounded tool output. Output over `maxInlineBytes` is written to a
|
|
103
|
+
* redacted artifact; only path, hash, byte count, and a redacted preview are
|
|
104
|
+
* stored inline. Redaction is mandatory and runs before anything reaches disk.
|
|
105
|
+
*/
|
|
106
|
+
recordOutput({ name = 'output', output, status = 'ok', tool = null, args = {}, durationMs = null, exitCode = null, retryNumber = 0 } = {}) {
|
|
107
|
+
const rawText = typeof output === 'string' ? output : JSON.stringify(output ?? '');
|
|
108
|
+
// Redaction is mandatory: never add an option to bypass it.
|
|
109
|
+
const text = redactString(rawText);
|
|
110
|
+
const bytes = Buffer.byteLength(text, 'utf-8');
|
|
111
|
+
const maxInline = this.policy.output?.maxInlineBytes ?? 64 * 1024;
|
|
112
|
+
const previewBytes = this.policy.output?.previewBytes ?? 4 * 1024;
|
|
113
|
+
// Estimated output tokens (contract §4) so output volume is bounded too.
|
|
114
|
+
const per = this.policy.estimation?.bytesPerToken || 3;
|
|
115
|
+
this.tracker.add('outputTokens', Math.ceil(bytes / per));
|
|
116
|
+
|
|
117
|
+
let artifactPath = null;
|
|
118
|
+
let artifactHash = null;
|
|
119
|
+
if (bytes > maxInline) {
|
|
120
|
+
const dir = join(runsDir(this.targetDir), 'artifacts');
|
|
121
|
+
mkdirSync(dir, { recursive: true });
|
|
122
|
+
const full = join(dir, `${this.runId}-${name}-${newId().slice(0, 8)}.log`);
|
|
123
|
+
writeFileSync(full, text, 'utf-8');
|
|
124
|
+
artifactPath = full;
|
|
125
|
+
// Hash covers the exact persisted (redacted) bytes.
|
|
126
|
+
artifactHash = sha256Bytes(Buffer.from(text, 'utf-8'));
|
|
127
|
+
}
|
|
128
|
+
|
|
129
|
+
return this.addSpan({
|
|
130
|
+
kind: 'tool-output',
|
|
131
|
+
name,
|
|
132
|
+
tool,
|
|
133
|
+
args: redact(args),
|
|
134
|
+
result: status,
|
|
135
|
+
exitCode,
|
|
136
|
+
durationMs,
|
|
137
|
+
outputBytes: bytes,
|
|
138
|
+
artifactPath,
|
|
139
|
+
artifactHash,
|
|
140
|
+
preview: artifactPath ? text.slice(0, previewBytes) : null,
|
|
141
|
+
retryNumber,
|
|
142
|
+
status,
|
|
143
|
+
});
|
|
144
|
+
}
|
|
145
|
+
|
|
146
|
+
finalize({ status = null } = {}) {
|
|
147
|
+
this.finishedAt = timestamp();
|
|
148
|
+
const result = this.tracker.result();
|
|
149
|
+
// The budget result is authoritative for failure states: a caller may not
|
|
150
|
+
// label an exhausted, blocked, or unmeasurable-cost run as success.
|
|
151
|
+
let effective = status || (result.exhausted ? 'exhausted' : result.ok ? 'ok' : 'warning');
|
|
152
|
+
const unsafeToClaimSuccess = result.exhausted || result.costUnmeasurable === true;
|
|
153
|
+
if (unsafeToClaimSuccess && (effective === 'ok' || effective === 'warning' || effective === 'running')) {
|
|
154
|
+
effective = result.exhausted ? 'exhausted' : 'blocked';
|
|
155
|
+
}
|
|
156
|
+
this.status = effective;
|
|
157
|
+
return this.toRecord();
|
|
158
|
+
}
|
|
159
|
+
|
|
160
|
+
toRecord() {
|
|
161
|
+
return {
|
|
162
|
+
runId: this.runId,
|
|
163
|
+
schemaVersion: RUN_SCHEMA_VERSION,
|
|
164
|
+
workItemId: this.workItemId,
|
|
165
|
+
phase: this.phase,
|
|
166
|
+
startedAt: this.startedAt,
|
|
167
|
+
finishedAt: this.finishedAt,
|
|
168
|
+
status: this.status,
|
|
169
|
+
spans: this.spans,
|
|
170
|
+
evidence: this.evidence,
|
|
171
|
+
decisions: this.decisions,
|
|
172
|
+
budget: this.tracker.result(),
|
|
173
|
+
usage: this.usage,
|
|
174
|
+
capabilities: this.capabilities,
|
|
175
|
+
retention: this.retention,
|
|
176
|
+
};
|
|
177
|
+
}
|
|
178
|
+
|
|
179
|
+
/** Persist the ledger to `.cadet/runs/<runId>.json` (atomically). */
|
|
180
|
+
persist() {
|
|
181
|
+
const dir = runsDir(this.targetDir);
|
|
182
|
+
mkdirSync(dir, { recursive: true });
|
|
183
|
+
const path = join(dir, `${this.runId}.json`);
|
|
184
|
+
// Atomic write: a process interruption cannot truncate a run record.
|
|
185
|
+
writeJsonAtomic(path, this.toRecord());
|
|
186
|
+
return path;
|
|
187
|
+
}
|
|
188
|
+
}
|
|
189
|
+
|
|
190
|
+
/** Load a run record by id. */
|
|
191
|
+
export function loadRun(targetDir, runId) {
|
|
192
|
+
const path = join(runsDir(targetDir), `${runId}.json`);
|
|
193
|
+
if (!existsSync(path)) return null;
|
|
194
|
+
return JSON.parse(readFileSync(path, 'utf-8'));
|
|
195
|
+
}
|
|
196
|
+
|
|
197
|
+
/** List run records (id + status + timestamps), newest first. */
|
|
198
|
+
export function listRuns(targetDir) {
|
|
199
|
+
const dir = runsDir(targetDir);
|
|
200
|
+
if (!existsSync(dir)) return [];
|
|
201
|
+
return readdirSync(dir)
|
|
202
|
+
.filter((f) => f.endsWith('.json'))
|
|
203
|
+
.map((f) => {
|
|
204
|
+
try {
|
|
205
|
+
const rec = JSON.parse(readFileSync(join(dir, f), 'utf-8'));
|
|
206
|
+
return { runId: rec.runId, status: rec.status, startedAt: rec.startedAt, finishedAt: rec.finishedAt, file: f };
|
|
207
|
+
} catch {
|
|
208
|
+
return null;
|
|
209
|
+
}
|
|
210
|
+
})
|
|
211
|
+
.filter(Boolean)
|
|
212
|
+
.sort((a, b) => (a.startedAt < b.startedAt ? 1 : -1));
|
|
213
|
+
}
|
|
214
|
+
|
|
215
|
+
/**
|
|
216
|
+
* Apply the retention policy: delete successful run records unless
|
|
217
|
+
* `retainRunRecords` is set; keep failed runs when `keepOnFailure`. Never deletes
|
|
218
|
+
* unrelated files.
|
|
219
|
+
*/
|
|
220
|
+
export function cleanupRuns(targetDir, policy, { olderThanMs = null, now = new Date() } = {}) {
|
|
221
|
+
const runs = listRuns(targetDir);
|
|
222
|
+
const retainAll = policy?.retention?.retainRunRecords === true;
|
|
223
|
+
const keepOnFailure = policy?.retention?.keepOnFailure !== false;
|
|
224
|
+
const deleted = [];
|
|
225
|
+
const kept = [];
|
|
226
|
+
|
|
227
|
+
for (const run of runs) {
|
|
228
|
+
const age = now.getTime() - new Date(run.startedAt).getTime();
|
|
229
|
+
const oldEnough = olderThanMs === null || age >= olderThanMs;
|
|
230
|
+
const failed = ['failed', 'blocked', 'exhausted'].includes(run.status);
|
|
231
|
+
if (retainAll) { kept.push(run.runId); continue; }
|
|
232
|
+
if (failed && keepOnFailure) { kept.push(run.runId); continue; }
|
|
233
|
+
if (!oldEnough) { kept.push(run.runId); continue; }
|
|
234
|
+
try {
|
|
235
|
+
unlinkSync(join(runsDir(targetDir), run.file));
|
|
236
|
+
deleted.push(run.runId);
|
|
237
|
+
} catch {
|
|
238
|
+
kept.push(run.runId);
|
|
239
|
+
}
|
|
240
|
+
}
|
|
241
|
+
return { deleted, kept };
|
|
242
|
+
}
|
|
243
|
+
|
|
244
|
+
/**
|
|
245
|
+
* Build a secret-free run report: budget consumption and failures.
|
|
246
|
+
* Unknown telemetry is labeled, never replaced with invented values.
|
|
247
|
+
*/
|
|
248
|
+
export function buildReport(run) {
|
|
249
|
+
const checks = run.budget?.checks || [];
|
|
250
|
+
return {
|
|
251
|
+
runId: run.runId,
|
|
252
|
+
status: run.status,
|
|
253
|
+
workItemId: run.workItemId,
|
|
254
|
+
phase: run.phase,
|
|
255
|
+
startedAt: run.startedAt,
|
|
256
|
+
finishedAt: run.finishedAt,
|
|
257
|
+
budgets: checks.map((c) => ({
|
|
258
|
+
budget: c.budget,
|
|
259
|
+
used: c.used,
|
|
260
|
+
hard: c.hard,
|
|
261
|
+
remaining: c.remaining,
|
|
262
|
+
overage: c.hard !== null && c.used > c.hard,
|
|
263
|
+
status: c.status,
|
|
264
|
+
})),
|
|
265
|
+
toolCalls: checks.find((c) => c.counter === 'toolCalls')?.used ?? null,
|
|
266
|
+
retries: checks.find((c) => c.counter === 'retries')?.used ?? null,
|
|
267
|
+
usage: run.usage
|
|
268
|
+
? {
|
|
269
|
+
source: run.usage.source,
|
|
270
|
+
inputTokens: run.usage.inputTokens,
|
|
271
|
+
outputTokens: run.usage.outputTokens,
|
|
272
|
+
usd: run.usage.usd,
|
|
273
|
+
model: run.usage.model,
|
|
274
|
+
costUnknown: run.usage.costUnknown === true || run.usage.usd === null,
|
|
275
|
+
}
|
|
276
|
+
: { source: 'unknown', costUnknown: true },
|
|
277
|
+
telemetry: {
|
|
278
|
+
tokenTelemetry: run.capabilities?.tokenTelemetry?.provider ? 'provider' : 'unavailable-estimated-or-unknown',
|
|
279
|
+
costTelemetry: run.usage?.usd != null ? 'estimated' : 'unavailable',
|
|
280
|
+
mcp: run.capabilities?.mcp?.available ? 'available' : 'unavailable',
|
|
281
|
+
},
|
|
282
|
+
failures: (run.spans || []).filter((s) => ['failed', 'blocked', 'exhausted', 'timed-out'].includes(s.status || s.result)),
|
|
283
|
+
spanCount: (run.spans || []).length,
|
|
284
|
+
evidenceCount: (run.evidence || []).length,
|
|
285
|
+
decisionCount: (run.decisions || []).length,
|
|
286
|
+
};
|
|
287
|
+
}
|
|
288
|
+
|
|
289
|
+
/** Machine-readable, secret-free report summary for display. */
|
|
290
|
+
export function formatReport(run) {
|
|
291
|
+
const r = buildReport(run);
|
|
292
|
+
const lines = [];
|
|
293
|
+
lines.push(`Run ${r.runId} — ${r.status}`);
|
|
294
|
+
lines.push(` Work item: ${r.workItemId || 'unscoped'} Phase: ${r.phase || 'unknown'}`);
|
|
295
|
+
lines.push(' Budgets:');
|
|
296
|
+
for (const b of r.budgets) {
|
|
297
|
+
const cap = b.hard === null ? 'unbounded' : String(b.hard);
|
|
298
|
+
lines.push(` ${b.budget.padEnd(22)} used ${String(b.used).padStart(8)} / ${cap.padStart(8)} (${b.status})${b.overage ? ' OVER' : ''}`);
|
|
299
|
+
}
|
|
300
|
+
lines.push(` Usage: tokens=${r.usage.source}; usd=${r.usage.usd ?? 'unknown'}`);
|
|
301
|
+
lines.push(` Telemetry: ${r.telemetry.tokenTelemetry}; cost=${r.telemetry.costTelemetry}; mcp=${r.telemetry.mcp}`);
|
|
302
|
+
if (r.failures.length) {
|
|
303
|
+
lines.push(` Failures (${r.failures.length}):`);
|
|
304
|
+
for (const f of r.failures.slice(0, 20)) lines.push(` ${f.kind || 'span'} ${f.name || f.tool || ''} ${f.result || f.status}`);
|
|
305
|
+
}
|
|
306
|
+
return lines.join('\n');
|
|
307
|
+
}
|
|
308
|
+
|
|
309
|
+
export { LedgerError, REDACTED, budgetReport };
|
|
@@ -0,0 +1,359 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Cadet-Agent harness policy.
|
|
3
|
+
*
|
|
4
|
+
* Loads hard-coded conservative defaults, optionally overlays a repository-local
|
|
5
|
+
* `.cadet/harness.json`, validates the merged result, and exposes the resolved
|
|
6
|
+
* budget/limit policy used by every other harness module.
|
|
7
|
+
*
|
|
8
|
+
* Contract: docs/core/HarnessContract.md §3. Phase names and gate names are frozen
|
|
9
|
+
* compatibility invariants — this module may validate them but must not rename them.
|
|
10
|
+
*/
|
|
11
|
+
|
|
12
|
+
import { readFileSync, existsSync } from 'node:fs';
|
|
13
|
+
import { join } from 'node:path';
|
|
14
|
+
|
|
15
|
+
/** Frozen phase names (compatibility invariant C1). */
|
|
16
|
+
export const PHASES = Object.freeze([
|
|
17
|
+
'context-resolution',
|
|
18
|
+
'requirements',
|
|
19
|
+
'requirementsComplete',
|
|
20
|
+
'architecture',
|
|
21
|
+
'architectureComplete',
|
|
22
|
+
'spikes',
|
|
23
|
+
'story-breakdown',
|
|
24
|
+
'implementation',
|
|
25
|
+
'review',
|
|
26
|
+
'validation',
|
|
27
|
+
'closed',
|
|
28
|
+
]);
|
|
29
|
+
|
|
30
|
+
/** Frozen gate names (compatibility invariant C3). */
|
|
31
|
+
export const GATES = Object.freeze([
|
|
32
|
+
'codeReviewCompleted',
|
|
33
|
+
'testsPassed',
|
|
34
|
+
'storyTrackingUpdated',
|
|
35
|
+
'compileCheckConfirmed',
|
|
36
|
+
'unityAnalyzerClean',
|
|
37
|
+
'acceptanceCriteriaValidated',
|
|
38
|
+
'securityReviewPassed',
|
|
39
|
+
'designArtifactSyncConfirmed',
|
|
40
|
+
]);
|
|
41
|
+
|
|
42
|
+
/** Legal phase transitions (compatibility invariant C4). */
|
|
43
|
+
export const TRANSITIONS = Object.freeze({
|
|
44
|
+
implementation: { to: 'review', gates: ['testsPassed', 'compileCheckConfirmed', 'unityAnalyzerClean', 'storyTrackingUpdated'] },
|
|
45
|
+
review: { to: 'validation', gates: ['codeReviewCompleted', 'securityReviewPassed', 'acceptanceCriteriaValidated'] },
|
|
46
|
+
validation: { to: 'closed', gates: ['designArtifactSyncConfirmed'] },
|
|
47
|
+
});
|
|
48
|
+
|
|
49
|
+
/** Evidence statuses. */
|
|
50
|
+
export const EVIDENCE_STATUSES = Object.freeze(['passed', 'failed', 'blocked', 'manual-confirmation', 'superseded']);
|
|
51
|
+
|
|
52
|
+
/** Retry classes. */
|
|
53
|
+
export const RETRY_CLASSES = Object.freeze(['deterministic', 'transient', 'repair', 'unknown']);
|
|
54
|
+
|
|
55
|
+
/** Context tiers. */
|
|
56
|
+
export const CONTEXT_TIERS = Object.freeze(['tier0', 'tier1', 'tier2', 'tier3']);
|
|
57
|
+
|
|
58
|
+
const MIB = 1024 * 1024;
|
|
59
|
+
|
|
60
|
+
/**
|
|
61
|
+
* Conservative default budgets. Every value is validated at load time.
|
|
62
|
+
* `hard` is the hard-stop threshold; `warn` is the soft-warning threshold (fraction 0..1).
|
|
63
|
+
*/
|
|
64
|
+
export const DEFAULT_BUDGETS = Object.freeze({
|
|
65
|
+
maxContextTokens: { hard: 64000, warn: 0.80, unit: 'tokens' },
|
|
66
|
+
maxOutputTokens: { hard: 8000, warn: 0.80, unit: 'tokens' },
|
|
67
|
+
maxToolCalls: { hard: 80, warn: 0.75, unit: 'calls' },
|
|
68
|
+
maxRetriesPerStep: { hard: 2, warn: null, unit: 'retries' },
|
|
69
|
+
maxTotalRetries: { hard: 8, warn: 0.75, unit: 'retries' },
|
|
70
|
+
maxWallClockMs: { hard: 30 * 60 * 1000, warn: 0.80, unit: 'ms' },
|
|
71
|
+
maxEstimatedCostUsd: { hard: 2.00, warn: 0.80, unit: 'usd' },
|
|
72
|
+
maxDownloadedBytes: { hard: 25 * MIB, warn: 0.80, unit: 'bytes' },
|
|
73
|
+
maxDecompressedBytes: { hard: 100 * MIB, warn: 0.80, unit: 'bytes' },
|
|
74
|
+
maxArchiveFiles: { hard: 2000, warn: 0.80, unit: 'files' },
|
|
75
|
+
});
|
|
76
|
+
|
|
77
|
+
/**
|
|
78
|
+
* Hard safety ceilings. A repository override may raise a hard limit only when an
|
|
79
|
+
* explicit compatibility flag is set; it may never lower one below the default
|
|
80
|
+
* without the same explicit flag (contract §3, plan §5.3).
|
|
81
|
+
*/
|
|
82
|
+
export const HARD_CEILINGS = Object.freeze({
|
|
83
|
+
maxDownloadedBytes: 100 * MIB,
|
|
84
|
+
maxDecompressedBytes: 500 * MIB,
|
|
85
|
+
maxArchiveFiles: 10000,
|
|
86
|
+
maxWallClockMs: 4 * 60 * 60 * 1000,
|
|
87
|
+
});
|
|
88
|
+
|
|
89
|
+
/** Default archive-safety limits (Phase 6 consumes these). */
|
|
90
|
+
export const DEFAULT_ARCHIVE_LIMITS = Object.freeze({
|
|
91
|
+
maxCompressedBytes: 25 * MIB,
|
|
92
|
+
maxDecompressedBytes: 100 * MIB,
|
|
93
|
+
maxFiles: 2000,
|
|
94
|
+
maxFilenameBytes: 240,
|
|
95
|
+
maxCompressionRatio: 100,
|
|
96
|
+
});
|
|
97
|
+
|
|
98
|
+
/** Default tool-output and retention policy. */
|
|
99
|
+
export const DEFAULT_OUTPUT_POLICY = Object.freeze({
|
|
100
|
+
maxInlineBytes: 64 * 1024,
|
|
101
|
+
previewBytes: 4 * 1024,
|
|
102
|
+
});
|
|
103
|
+
|
|
104
|
+
export const DEFAULT_RETENTION = Object.freeze({
|
|
105
|
+
keepOnFailure: true,
|
|
106
|
+
retainRunRecords: false,
|
|
107
|
+
retainRawPrompt: false,
|
|
108
|
+
});
|
|
109
|
+
|
|
110
|
+
/** Default token/cost estimation policy. */
|
|
111
|
+
export const DEFAULT_ESTIMATION = Object.freeze({
|
|
112
|
+
bytesPerToken: 3,
|
|
113
|
+
// Known rate cards keyed by model id. Absence of a card => no USD is reported.
|
|
114
|
+
rateCards: Object.freeze({}),
|
|
115
|
+
});
|
|
116
|
+
|
|
117
|
+
/** Git-guard hook behavior. */
|
|
118
|
+
export const DEFAULT_HOOK_POLICY = Object.freeze({
|
|
119
|
+
mode: 'ask-on-recognized-write', // or 'fail-open' (opt-in, visible)
|
|
120
|
+
});
|
|
121
|
+
|
|
122
|
+
const BUDGET_KEYS = Object.keys(DEFAULT_BUDGETS);
|
|
123
|
+
|
|
124
|
+
class PolicyError extends Error {
|
|
125
|
+
constructor(message) {
|
|
126
|
+
super(message);
|
|
127
|
+
this.name = 'PolicyError';
|
|
128
|
+
}
|
|
129
|
+
}
|
|
130
|
+
|
|
131
|
+
function isPlainObject(v) {
|
|
132
|
+
return v !== null && typeof v === 'object' && !Array.isArray(v);
|
|
133
|
+
}
|
|
134
|
+
|
|
135
|
+
function isFiniteNumber(v) {
|
|
136
|
+
return typeof v === 'number' && Number.isFinite(v);
|
|
137
|
+
}
|
|
138
|
+
|
|
139
|
+
function isNonNegativeInt(v) {
|
|
140
|
+
return Number.isInteger(v) && v >= 0;
|
|
141
|
+
}
|
|
142
|
+
|
|
143
|
+
/**
|
|
144
|
+
* Validate a single budget override. Returns the normalized entry or throws PolicyError.
|
|
145
|
+
*/
|
|
146
|
+
function validateBudgetOverride(key, value, defaults) {
|
|
147
|
+
const def = defaults[key];
|
|
148
|
+
if (!def) {
|
|
149
|
+
throw new PolicyError(`Unknown budget key "${key}".`);
|
|
150
|
+
}
|
|
151
|
+
if (isPlainObject(value)) {
|
|
152
|
+
const out = { ...def };
|
|
153
|
+
for (const field of Object.keys(value)) {
|
|
154
|
+
if (field !== 'hard' && field !== 'warn') {
|
|
155
|
+
throw new PolicyError(`Budget "${key}" has unknown field "${field}".`);
|
|
156
|
+
}
|
|
157
|
+
const fv = value[field];
|
|
158
|
+
if (field === 'hard') {
|
|
159
|
+
if (!isFiniteNumber(fv) || fv < 0) {
|
|
160
|
+
throw new PolicyError(`Budget "${key}.hard" must be a non-negative finite number.`);
|
|
161
|
+
}
|
|
162
|
+
out.hard = fv;
|
|
163
|
+
} else {
|
|
164
|
+
if (fv !== null && (!isFiniteNumber(fv) || fv < 0 || fv > 1)) {
|
|
165
|
+
throw new PolicyError(`Budget "${key}.warn" must be null or a fraction between 0 and 1.`);
|
|
166
|
+
}
|
|
167
|
+
out.warn = fv;
|
|
168
|
+
}
|
|
169
|
+
}
|
|
170
|
+
return out;
|
|
171
|
+
}
|
|
172
|
+
if (!isFiniteNumber(value) || value < 0) {
|
|
173
|
+
throw new PolicyError(`Budget "${key}" must be a non-negative finite number or an object.`);
|
|
174
|
+
}
|
|
175
|
+
return { ...def, hard: value };
|
|
176
|
+
}
|
|
177
|
+
|
|
178
|
+
/**
|
|
179
|
+
* A project may never lower a hard safety ceiling, and may only exceed it with an
|
|
180
|
+
* explicit compatibility flag.
|
|
181
|
+
*/
|
|
182
|
+
function enforceCeilings(resolved, { allowCeilingOverride = false } = {}) {
|
|
183
|
+
for (const [key, ceiling] of Object.entries(HARD_CEILINGS)) {
|
|
184
|
+
const value = resolved.budgets[key]?.hard;
|
|
185
|
+
if (value === undefined) continue;
|
|
186
|
+
if (value > ceiling && !allowCeilingOverride) {
|
|
187
|
+
throw new PolicyError(
|
|
188
|
+
`Budget "${key}" (${value}) exceeds the hard safety ceiling (${ceiling}). ` +
|
|
189
|
+
`Set allowBudgetCeilingOverride in .cadet/harness.json to opt in explicitly.`
|
|
190
|
+
);
|
|
191
|
+
}
|
|
192
|
+
}
|
|
193
|
+
}
|
|
194
|
+
|
|
195
|
+
/**
|
|
196
|
+
* Parse and validate a repository harness policy document.
|
|
197
|
+
* Unknown top-level keys are rejected so misconfiguration fails loudly.
|
|
198
|
+
*/
|
|
199
|
+
export function validatePolicy(raw, defaults = DEFAULT_BUDGETS) {
|
|
200
|
+
if (!isPlainObject(raw)) {
|
|
201
|
+
throw new PolicyError('Harness policy must be a JSON object.');
|
|
202
|
+
}
|
|
203
|
+
const allowed = new Set([
|
|
204
|
+
'$schema',
|
|
205
|
+
'budgets', 'archive', 'output', 'retention', 'estimation', 'hook',
|
|
206
|
+
'allowBudgetCeilingOverride', 'scopes', 'model', 'analyzerCommand',
|
|
207
|
+
'compileCommand', 'testCommand', 'allowEmptyFreshness',
|
|
208
|
+
]);
|
|
209
|
+
for (const key of Object.keys(raw)) {
|
|
210
|
+
if (!allowed.has(key)) {
|
|
211
|
+
throw new PolicyError(`Unknown harness policy key "${key}".`);
|
|
212
|
+
}
|
|
213
|
+
}
|
|
214
|
+
|
|
215
|
+
const budgets = {};
|
|
216
|
+
for (const [key, def] of Object.entries(defaults)) {
|
|
217
|
+
budgets[key] = { ...def };
|
|
218
|
+
}
|
|
219
|
+
if (raw.budgets !== undefined) {
|
|
220
|
+
if (!isPlainObject(raw.budgets)) {
|
|
221
|
+
throw new PolicyError('"budgets" must be an object.');
|
|
222
|
+
}
|
|
223
|
+
for (const [key, value] of Object.entries(raw.budgets)) {
|
|
224
|
+
budgets[key] = validateBudgetOverride(key, value, defaults);
|
|
225
|
+
}
|
|
226
|
+
}
|
|
227
|
+
|
|
228
|
+
const archive = { ...DEFAULT_ARCHIVE_LIMITS, ...(raw.archive || {}) };
|
|
229
|
+
for (const [key, value] of Object.entries(raw.archive || {})) {
|
|
230
|
+
if (!(key in DEFAULT_ARCHIVE_LIMITS)) {
|
|
231
|
+
throw new PolicyError(`Unknown archive limit "${key}".`);
|
|
232
|
+
}
|
|
233
|
+
if (!isNonNegativeInt(value)) {
|
|
234
|
+
throw new PolicyError(`Archive limit "${key}" must be a non-negative integer.`);
|
|
235
|
+
}
|
|
236
|
+
}
|
|
237
|
+
|
|
238
|
+
const output = { ...DEFAULT_OUTPUT_POLICY, ...(raw.output || {}) };
|
|
239
|
+
for (const [key, value] of Object.entries(raw.output || {})) {
|
|
240
|
+
if (!(key in DEFAULT_OUTPUT_POLICY)) {
|
|
241
|
+
throw new PolicyError(`Unknown output policy key "${key}".`);
|
|
242
|
+
}
|
|
243
|
+
if (!isNonNegativeInt(value)) {
|
|
244
|
+
throw new PolicyError(`Output policy "${key}" must be a non-negative integer.`);
|
|
245
|
+
}
|
|
246
|
+
}
|
|
247
|
+
|
|
248
|
+
const retention = { ...DEFAULT_RETENTION, ...(raw.retention || {}) };
|
|
249
|
+
for (const [key, value] of Object.entries(raw.retention || {})) {
|
|
250
|
+
if (!(key in DEFAULT_RETENTION)) {
|
|
251
|
+
throw new PolicyError(`Unknown retention key "${key}".`);
|
|
252
|
+
}
|
|
253
|
+
if (typeof value !== 'boolean') {
|
|
254
|
+
throw new PolicyError(`Retention "${key}" must be a boolean.`);
|
|
255
|
+
}
|
|
256
|
+
}
|
|
257
|
+
|
|
258
|
+
const estimation = {
|
|
259
|
+
...DEFAULT_ESTIMATION,
|
|
260
|
+
...(raw.estimation || {}),
|
|
261
|
+
rateCards: { ...(raw.estimation?.rateCards || {}) },
|
|
262
|
+
};
|
|
263
|
+
if (!isNonNegativeInt(estimation.bytesPerToken) || estimation.bytesPerToken < 1) {
|
|
264
|
+
throw new PolicyError('"estimation.bytesPerToken" must be a positive integer.');
|
|
265
|
+
}
|
|
266
|
+
|
|
267
|
+
const hook = { ...DEFAULT_HOOK_POLICY, ...(raw.hook || {}) };
|
|
268
|
+
if (!['ask-on-recognized-write', 'fail-open'].includes(hook.mode)) {
|
|
269
|
+
throw new PolicyError('"hook.mode" must be "ask-on-recognized-write" or "fail-open".');
|
|
270
|
+
}
|
|
271
|
+
|
|
272
|
+
const allowCeilingOverride = raw.allowBudgetCeilingOverride === true;
|
|
273
|
+
|
|
274
|
+
const resolved = {
|
|
275
|
+
budgets,
|
|
276
|
+
archive,
|
|
277
|
+
output,
|
|
278
|
+
retention,
|
|
279
|
+
estimation,
|
|
280
|
+
hook,
|
|
281
|
+
allowBudgetCeilingOverride: allowCeilingOverride,
|
|
282
|
+
allowEmptyFreshness: raw.allowEmptyFreshness === true,
|
|
283
|
+
scopes: raw.scopes || { perRun: {}, perStory: {} },
|
|
284
|
+
model: raw.model || null,
|
|
285
|
+
analyzerCommand: raw.analyzerCommand || null,
|
|
286
|
+
compileCommand: raw.compileCommand || null,
|
|
287
|
+
testCommand: raw.testCommand || null,
|
|
288
|
+
};
|
|
289
|
+
|
|
290
|
+
if (raw.scopes !== undefined) {
|
|
291
|
+
if (!isPlainObject(raw.scopes)) throw new PolicyError('"scopes" must be an object.');
|
|
292
|
+
for (const scope of ['perRun', 'perStory']) {
|
|
293
|
+
const s = raw.scopes[scope];
|
|
294
|
+
if (s === undefined) continue;
|
|
295
|
+
if (!isPlainObject(s)) throw new PolicyError(`"scopes.${scope}" must be an object.`);
|
|
296
|
+
for (const [key, value] of Object.entries(s)) {
|
|
297
|
+
const v = validateBudgetOverride(key, isPlainObject(value) ? value.hard ?? value : value, defaults);
|
|
298
|
+
if (value !== null && isPlainObject(value) && value.warn !== undefined) {
|
|
299
|
+
if (value.warn !== null && (!isFiniteNumber(value.warn) || value.warn < 0 || value.warn > 1)) {
|
|
300
|
+
throw new PolicyError(`"scopes.${scope}.${key}.warn" must be null or a fraction between 0 and 1.`);
|
|
301
|
+
}
|
|
302
|
+
}
|
|
303
|
+
s[key] = v.hard;
|
|
304
|
+
}
|
|
305
|
+
}
|
|
306
|
+
}
|
|
307
|
+
|
|
308
|
+
enforceCeilings(resolved, { allowCeilingOverride });
|
|
309
|
+
|
|
310
|
+
return resolved;
|
|
311
|
+
}
|
|
312
|
+
|
|
313
|
+
/** Returns the built-in default policy, fully validated. */
|
|
314
|
+
export function defaultPolicy() {
|
|
315
|
+
return validatePolicy({});
|
|
316
|
+
}
|
|
317
|
+
|
|
318
|
+
export function policyPath(targetDir) {
|
|
319
|
+
return join(targetDir, '.cadet', 'harness.json');
|
|
320
|
+
}
|
|
321
|
+
|
|
322
|
+
/**
|
|
323
|
+
* Load the resolved policy for a repository. Missing file => defaults.
|
|
324
|
+
* Malformed/unknown-typed policy throws PolicyError rather than silently degrading.
|
|
325
|
+
*/
|
|
326
|
+
export function loadPolicy(targetDir, { defaults = DEFAULT_BUDGETS } = {}) {
|
|
327
|
+
const path = policyPath(targetDir);
|
|
328
|
+
if (!existsSync(path)) {
|
|
329
|
+
return { ...validatePolicy({}, defaults), sourcePath: null };
|
|
330
|
+
}
|
|
331
|
+
let raw;
|
|
332
|
+
try {
|
|
333
|
+
raw = JSON.parse(readFileSync(path, 'utf-8'));
|
|
334
|
+
} catch (err) {
|
|
335
|
+
throw new PolicyError(`Failed to parse ${path}: ${err.message}`);
|
|
336
|
+
}
|
|
337
|
+
return { ...validatePolicy(raw, defaults), sourcePath: path };
|
|
338
|
+
}
|
|
339
|
+
|
|
340
|
+
/** Resolve the effective budget for a scope, applying per-run/per-story overrides. */
|
|
341
|
+
export function budgetForScope(policy, scope = 'perRun') {
|
|
342
|
+
const budgets = {};
|
|
343
|
+
for (const [key, def] of Object.entries(policy.budgets)) {
|
|
344
|
+
budgets[key] = { ...def };
|
|
345
|
+
}
|
|
346
|
+
const overrides = policy.scopes?.[scope] || {};
|
|
347
|
+
for (const [key, value] of Object.entries(overrides)) {
|
|
348
|
+
if (budgets[key]) budgets[key] = { ...budgets[key], hard: value };
|
|
349
|
+
}
|
|
350
|
+
return budgets;
|
|
351
|
+
}
|
|
352
|
+
|
|
353
|
+
/** Warning threshold for a budget, or null when the budget has no warning. */
|
|
354
|
+
export function warnThreshold(def) {
|
|
355
|
+
if (def.warn === null || def.warn === undefined) return null;
|
|
356
|
+
return def.hard * def.warn;
|
|
357
|
+
}
|
|
358
|
+
|
|
359
|
+
export { PolicyError, BUDGET_KEYS, MIB };
|