cadet-agent 0.20.2 → 0.24.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,229 @@
1
+ /**
2
+ * Cadet-Agent context management.
3
+ *
4
+ * Builds a context manifest with tier, reason, authority, content hash, and size
5
+ * estimate for every loaded item. Deduplicates repeat content by hash before
6
+ * budget accounting, marks stale items when files change, and enforces a bounded
7
+ * expansion budget.
8
+ *
9
+ * Contract: docs/core/HarnessContract.md §7 (context tiers).
10
+ */
11
+
12
+ import { existsSync, statSync } from 'node:fs';
13
+ import { join, normalize, sep } from 'node:path';
14
+ import { CONTEXT_TIERS } from './policy.mjs';
15
+ import { hashFile, sha256, timestamp } from './util.mjs';
16
+ import { BudgetTracker } from './budget.mjs';
17
+
18
+ export { CONTEXT_TIERS };
19
+
20
+ /** Default reason used when a tier requires an explicit justification. */
21
+ export const TIER_REASONS_REQUIRED = Object.freeze(['tier2', 'tier3']);
22
+
23
+ /** Default maximum number of new context items a single step may add. */
24
+ export const DEFAULT_MAX_EXPANSION_PER_STEP = 12;
25
+
26
+ class ContextError extends Error {
27
+ constructor(message) {
28
+ super(message);
29
+ this.name = 'ContextError';
30
+ }
31
+ }
32
+
33
+ function assertTier(tier) {
34
+ if (!CONTEXT_TIERS.includes(tier)) {
35
+ throw new ContextError(`unknown context tier "${tier}" (expected one of ${CONTEXT_TIERS.join(', ')})`);
36
+ }
37
+ }
38
+
39
+ function resolveWithin(rootDir, reference) {
40
+ const normalizedRoot = normalize(rootDir).replace(/[\\/]+$/, '');
41
+ const abs = normalize(join(rootDir, reference));
42
+ if (abs !== normalizedRoot && !abs.startsWith(normalizedRoot + sep)) {
43
+ throw new ContextError(`context reference escapes the workspace: ${reference}`);
44
+ }
45
+ return abs;
46
+ }
47
+
48
+ /**
49
+ * A context manifest for one run/step. Records load decisions; never stores file
50
+ * contents, only hashes and sizes.
51
+ */
52
+ export class ContextManifest {
53
+ constructor({ policy, budgets = null, rootDir = process.cwd(), maxExpansionPerStep = DEFAULT_MAX_EXPANSION_PER_STEP } = {}) {
54
+ this.policy = policy;
55
+ this.rootDir = rootDir;
56
+ this.tracker = budgets || new BudgetTracker(policy);
57
+ this.maxExpansionPerStep = maxExpansionPerStep;
58
+ this.items = [];
59
+ this.seenHashes = new Map(); // hash -> canonical item
60
+ this.bytesAddedThisStep = 0;
61
+ this.stepCount = new Map(); // step label -> items added
62
+ this.deduplicated = 0;
63
+ this.denied = [];
64
+ }
65
+
66
+ /**
67
+ * Register a context load. Returns `{ item, deduplicated, counted }`.
68
+ * Duplicate content (same hash) is not counted against the context budget again.
69
+ */
70
+ load({ reference, tier = 'tier1', reason = null, authority = 'repository', step = 'step', content = null, bytes = null } = {}) {
71
+ assertTier(tier);
72
+ if (!reference) throw new ContextError('context reference is required');
73
+ if (TIER_REASONS_REQUIRED.includes(tier) && !reason) {
74
+ throw new ContextError(`tier ${tier} requires a recorded reason`);
75
+ }
76
+
77
+ const abs = resolveWithin(this.rootDir, reference);
78
+ const exists = existsSync(abs);
79
+ const fileBytes = exists && statSync(abs).isFile() ? statSync(abs).size : (bytes ?? (content != null ? Buffer.byteLength(String(content), 'utf-8') : 0));
80
+ const hash = exists && statSync(abs).isFile()
81
+ ? hashFile(abs)
82
+ : sha256(content != null ? String(content) : reference);
83
+
84
+ // Deduplicate by content hash before counting against the budget.
85
+ if (this.seenHashes.has(hash)) {
86
+ this.deduplicated++;
87
+ const canonical = this.seenHashes.get(hash);
88
+ const item = {
89
+ reference,
90
+ tier,
91
+ reason: reason || `duplicate of ${canonical.reference}`,
92
+ authority,
93
+ hash,
94
+ bytes: fileBytes,
95
+ estimatedTokens: estimateTokensForBytes(fileBytes, this.policy),
96
+ loadedAt: timestamp(),
97
+ step,
98
+ deduplicated: true,
99
+ exists,
100
+ stale: false,
101
+ };
102
+ this.items.push(item);
103
+ return { item, deduplicated: true, counted: false };
104
+ }
105
+
106
+ const item = {
107
+ reference,
108
+ tier,
109
+ reason,
110
+ authority,
111
+ hash,
112
+ bytes: fileBytes,
113
+ estimatedTokens: estimateTokensForBytes(fileBytes, this.policy),
114
+ loadedAt: timestamp(),
115
+ step,
116
+ deduplicated: false,
117
+ exists,
118
+ stale: false,
119
+ };
120
+
121
+ // Expansion budget: bounded per step.
122
+ const stepItems = this.stepCount.get(step) || 0;
123
+ if (stepItems >= this.maxExpansionPerStep) {
124
+ this.denied.push({ reference, tier, reason: 'per-step context expansion limit reached' });
125
+ throw new ContextError(`context expansion limit reached for ${step} (${this.maxExpansionPerStep})`);
126
+ }
127
+
128
+ // The context-token budget is a hard limit: adding this item must not push
129
+ // the run past it. Reject before recording so the manifest stays accurate.
130
+ const prospectiveTokens = this.tracker.counters.contextTokens + item.estimatedTokens;
131
+ const budgetCheck = this.tracker.check('contextTokens');
132
+ if (budgetCheck.hard !== null && prospectiveTokens > budgetCheck.hard) {
133
+ this.denied.push({ reference, tier, reason: 'context token budget exhausted' });
134
+ throw new ContextError(
135
+ `context token budget exhausted: adding ${reference} would use ${prospectiveTokens} of ${budgetCheck.hard}`
136
+ );
137
+ }
138
+
139
+ this.tracker.add('contextTokens', item.estimatedTokens);
140
+ this.stepCount.set(step, stepItems + 1);
141
+ this.seenHashes.set(hash, item);
142
+ this.items.push(item);
143
+ return { item, deduplicated: false, counted: true };
144
+ }
145
+
146
+ /** Mark items stale whose file hash changed or whose work item no longer matches. */
147
+ refresh({ changedReferences = null } = {}) {
148
+ const stale = [];
149
+ for (const item of this.items) {
150
+ if (item.deduplicated) continue;
151
+ const abs = resolveWithin(this.rootDir, item.reference);
152
+ const currentHash = existsSync(abs) && statSync(abs).isFile() ? hashFile(abs) : null;
153
+ const changed = changedReferences
154
+ ? changedReferences.includes(item.reference)
155
+ : currentHash !== item.hash;
156
+ if (changed) {
157
+ item.stale = true;
158
+ stale.push({ reference: item.reference, reason: 'content hash changed or work item changed' });
159
+ }
160
+ }
161
+ return { stale, anyStale: stale.length > 0 };
162
+ }
163
+
164
+ /** Items that are currently usable as evidence context. */
165
+ freshItems() {
166
+ return this.items.filter((i) => !i.stale && !i.deduplicated && i.exists);
167
+ }
168
+
169
+ manifest() {
170
+ const uniqueBytes = this.items.filter((i) => !i.deduplicated).reduce((a, i) => a + i.bytes, 0);
171
+ return {
172
+ rootDir: this.rootDir,
173
+ items: this.items.map((i) => ({ ...i })),
174
+ deduplicated: this.deduplicated,
175
+ uniqueBytes,
176
+ totalBytes: this.items.reduce((a, i) => a + i.bytes, 0),
177
+ denied: [...this.denied],
178
+ budget: this.tracker.result(),
179
+ };
180
+ }
181
+
182
+ /** Human/JSON-safe summary showing why each item was loaded. */
183
+ report() {
184
+ return this.items.map((i) => ({
185
+ reference: i.reference,
186
+ tier: i.tier,
187
+ reason: i.reason,
188
+ authority: i.authority,
189
+ hash: i.hash,
190
+ bytes: i.bytes,
191
+ estimatedTokens: i.estimatedTokens,
192
+ deduplicated: i.deduplicated,
193
+ stale: i.stale,
194
+ }));
195
+ }
196
+ }
197
+
198
+ function estimateTokensForBytes(bytes, policy) {
199
+ const per = policy?.estimation?.bytesPerToken || 3;
200
+ return Math.ceil((bytes || 0) / per);
201
+ }
202
+
203
+ /**
204
+ * Tier 0 context that must always be loaded. Missing entries are reported rather
205
+ * than silently skipped.
206
+ */
207
+ export function tier0References(targetDir) {
208
+ return [
209
+ { reference: '.cadet/agent/core/cadet-agent.md', authority: 'framework' },
210
+ { reference: '.cadet/harness.json', authority: 'framework', optional: true },
211
+ { reference: '.cadet/state.json', authority: 'session', optional: true },
212
+ ].map((r) => ({ ...r, abs: join(targetDir, r.reference), present: existsSync(join(targetDir, r.reference)) }));
213
+ }
214
+
215
+ /** Build a manifest seeded with Tier 0 items. */
216
+ export function buildBaseManifest({ targetDir, policy, budgets = null } = {}) {
217
+ const manifest = new ContextManifest({ policy, budgets, rootDir: targetDir });
218
+ const missingTier0 = [];
219
+ for (const ref of tier0References(targetDir)) {
220
+ if (!ref.present) {
221
+ if (!ref.optional) missingTier0.push(ref.reference);
222
+ continue;
223
+ }
224
+ manifest.load({ reference: ref.reference, tier: 'tier0', reason: 'always-load tier 0 context', authority: ref.authority, step: 'tier0' });
225
+ }
226
+ return { manifest, missingTier0 };
227
+ }
228
+
229
+ export { ContextError };
@@ -0,0 +1,147 @@
1
+ /**
2
+ * Cadet-Agent git-guard hook logic.
3
+ *
4
+ * A single, testable implementation of the PreToolUse decision. The shell and
5
+ * PowerShell hook scripts mirror this logic for host runtimes that cannot import
6
+ * Node modules; tests exercise this module and the scripts against the same
7
+ * payload fixtures.
8
+ *
9
+ * Behavior (contract §9):
10
+ * - Recognized git write (commit/push/gh pr merge) → `ask`.
11
+ * - Malformed JSON, or a recognized shell tool with unrecognized input → `hook-error`.
12
+ * - Non-relevant tool → `pass` (no decision, exit 0).
13
+ * - `fail-open` compatibility mode is opt-in and logged.
14
+ */
15
+
16
+ export const RELEVANT_TOOLS = Object.freeze(['run_in_terminal', 'execute', 'bash', 'shell']);
17
+
18
+ /**
19
+ * Git write patterns.
20
+ *
21
+ * Tolerates intervening options and option values (e.g. `git -c user.name=x commit`):
22
+ * we locate a `git` token and require `commit`/`push` to appear as a standalone
23
+ * token after it, before any command separator.
24
+ */
25
+ export const WRITE_PATTERNS = Object.freeze([
26
+ { id: 'git commit', re: /\bgit\b[^;&|]*?\bcommit\b/i },
27
+ { id: 'git push', re: /\bgit\b[^;&|]*?\bpush\b/i },
28
+ { id: 'gh pr merge', re: /\bgh\b[^;&|]*?\bpr\b[^;&|]*?\bmerge\b/i },
29
+ ]);
30
+
31
+ /** Command obfuscation used to hide a write: separators, quoting, env prefixes, `$(...)`. */
32
+ const OBFUSCATION_NORMALIZERS = [
33
+ (s) => s.replace(/["'`]/g, ''),
34
+ (s) => s.replace(/\\(?=[a-z])/g, ''),
35
+ (s) => s.replace(/\$\{[^}]*\}/g, ' '),
36
+ (s) => s.replace(/\$\([^)]*\)/g, ' '),
37
+ (s) => s.replace(/\s+/g, ' '),
38
+ ];
39
+
40
+ function normalizeCommand(input) {
41
+ let out = String(input || '').toLowerCase();
42
+ for (const fn of OBFUSCATION_NORMALIZERS) out = fn(out);
43
+ return out;
44
+ }
45
+
46
+ /** Detect a git write operation in a command string, tolerant of light obfuscation. */
47
+ export function detectWrite(command) {
48
+ if (!command) return null;
49
+ const normalized = normalizeCommand(command);
50
+ for (const pattern of WRITE_PATTERNS) {
51
+ if (pattern.re.test(normalized)) return pattern.id;
52
+ }
53
+ return null;
54
+ }
55
+
56
+ function decision(permissionDecision, reason) {
57
+ return {
58
+ exitCode: 0,
59
+ output: permissionDecision
60
+ ? {
61
+ hookSpecificOutput: {
62
+ hookEventName: 'PreToolUse',
63
+ permissionDecision,
64
+ permissionDecisionReason: reason,
65
+ },
66
+ }
67
+ : null,
68
+ };
69
+ }
70
+
71
+ /**
72
+ * Evaluate a raw hook payload string.
73
+ * Returns `{ decision: 'ask'|'deny'|'pass', output, exitCode, diagnostics }`.
74
+ */
75
+ export function evaluateHook(rawPayload, { mode = 'ask-on-recognized-write', log = null } = {}) {
76
+ const diagnostics = [];
77
+ const failOpen = mode === 'fail-open';
78
+ const logError = (message) => {
79
+ diagnostics.push(message);
80
+ if (typeof log === 'function') log(message);
81
+ else if (log && typeof log.error === 'function') log.error(message);
82
+ };
83
+
84
+ if (rawPayload === null || rawPayload === undefined || String(rawPayload).trim() === '') {
85
+ // No payload: nothing to guard. Treat as pass (the host called us with no input).
86
+ return { decision: 'pass', ...decision(null, ''), diagnostics };
87
+ }
88
+
89
+ let payload;
90
+ try {
91
+ payload = JSON.parse(String(rawPayload));
92
+ } catch (err) {
93
+ const message = `git-guard received malformed JSON: ${err.message}`;
94
+ logError(message);
95
+ if (failOpen) {
96
+ return { decision: 'pass', ...decision(null, ''), diagnostics, compatibilityMode: 'fail-open' };
97
+ }
98
+ return {
99
+ decision: 'deny',
100
+ ...decision('deny', 'git-guard blocked the tool call because the hook payload was malformed JSON.'),
101
+ diagnostics,
102
+ code: 'hook-error',
103
+ };
104
+ }
105
+
106
+ const toolName = String(payload.toolName || payload.tool_name || payload.tool || '');
107
+ if (!RELEVANT_TOOLS.includes(toolName)) {
108
+ return { decision: 'pass', ...decision(null, ''), diagnostics };
109
+ }
110
+
111
+ const raw = payload.toolInput ?? payload.toolArgs ?? payload.command ?? '';
112
+ const commandText = typeof raw === 'string'
113
+ ? raw
114
+ : (raw && typeof raw === 'object' ? JSON.stringify(raw) : String(raw ?? ''));
115
+
116
+ if (!commandText.trim()) {
117
+ // A recognized shell tool with no readable command is an unrecognized input.
118
+ const message = `git-guard received unrecognized input for tool "${toolName}"`;
119
+ logError(message);
120
+ if (failOpen) {
121
+ return { decision: 'pass', ...decision(null, ''), diagnostics, compatibilityMode: 'fail-open' };
122
+ }
123
+ return {
124
+ decision: 'deny',
125
+ ...decision('deny', 'git-guard blocked the tool call because its input could not be interpreted.'),
126
+ diagnostics,
127
+ code: 'hook-error',
128
+ };
129
+ }
130
+
131
+ const detected = detectWrite(commandText);
132
+ if (!detected) {
133
+ return { decision: 'pass', ...decision(null, ''), diagnostics };
134
+ }
135
+
136
+ return {
137
+ decision: 'ask',
138
+ ...decision('ask', `${detected} requires explicit user approval per Cadet policy. Review the proposed changes before approving.`),
139
+ diagnostics,
140
+ detected,
141
+ };
142
+ }
143
+
144
+ /** Serialize a hook decision to the JSON the host expects on stdout. */
145
+ export function hookOutput(decision) {
146
+ return decision.output ? JSON.stringify(decision.output) : '';
147
+ }
@@ -0,0 +1,60 @@
1
+ /**
2
+ * Cadet-Agent harness — stable internal entry point.
3
+ *
4
+ * The CLI and tests import from here so they never depend on the file layout of
5
+ * the individual harness modules.
6
+ */
7
+
8
+ export {
9
+ PHASES, GATES, TRANSITIONS, EVIDENCE_STATUSES, RETRY_CLASSES, CONTEXT_TIERS,
10
+ DEFAULT_BUDGETS, HARD_CEILINGS, DEFAULT_ARCHIVE_LIMITS, DEFAULT_OUTPUT_POLICY,
11
+ DEFAULT_RETENTION, DEFAULT_ESTIMATION, DEFAULT_HOOK_POLICY,
12
+ validatePolicy, defaultPolicy, loadPolicy, budgetForScope, policyPath, PolicyError,
13
+ } from './policy.mjs';
14
+
15
+ export {
16
+ BudgetTracker, budgetExhaustedResult, budgetReport, evaluateHardStop,
17
+ estimateTokens, estimateCost, normalizeUsage, BUDGET_RESULTS,
18
+ } from './budget.mjs';
19
+
20
+ export {
21
+ newId, isUuid, sha256, sha256Bytes, hashFile, hashTree, hashCriteria, timestamp, canonicalJson, changedFiles, gitChangedFiles,
22
+ } from './util.mjs';
23
+
24
+ export {
25
+ STATE_VERSION, validateState, migrateStateV1toV2, migrateStateFile,
26
+ createEvidence, computeInputTreeHash, workItemIdOf, evidenceFreshness,
27
+ latestEvidenceForGate, activeExceptions, requiredGates, evaluateTransition,
28
+ applyTransition, resetGatesForNewWorkItem, statePathFor, readState, writeState, writeJsonAtomic, StateError,
29
+ } from './state.mjs';
30
+
31
+ export {
32
+ RETRY_CLASSES as VERIFICATION_RETRY_CLASSES, RESULT_STATUSES, TRANSIENT_BACKOFF_MS,
33
+ DEFAULT_FLAKY_SIGNATURES, classifyResult, classifyRepair, runCommand,
34
+ commandForGate, analyzerClean, runVerificationLoop, manualConfirmation, isBudgetExhaustion,
35
+ } from './verification.mjs';
36
+
37
+ export {
38
+ ContextManifest, buildBaseManifest, tier0References, TIER_REASONS_REQUIRED,
39
+ DEFAULT_MAX_EXPANSION_PER_STEP, ContextError,
40
+ } from './context.mjs';
41
+
42
+ export {
43
+ TOOL_KINDS, ROUTING_REASONS, detectCapabilities, routeTask, toolCallSpan, RoutingError,
44
+ } from './routing.mjs';
45
+
46
+ export { redact, redactString, containsSecret, REDACTED, REDACTION_CATEGORIES } from './redaction.mjs';
47
+
48
+ export {
49
+ RUN_SCHEMA_VERSION, RunLedger, loadRun, listRuns, cleanupRuns, buildReport, formatReport,
50
+ runsDir, LedgerError,
51
+ } from './ledger.mjs';
52
+
53
+ export {
54
+ extractArchive, readArchiveEntry, readEntries, assertContained, crc32, findEocd,
55
+ ArchiveError, DEFAULT_ARCHIVE_LIMITS as ARCHIVE_LIMITS,
56
+ } from './archive.mjs';
57
+
58
+ export {
59
+ RELEVANT_TOOLS, WRITE_PATTERNS, detectWrite, evaluateHook, hookOutput,
60
+ } from './hook.mjs';