cadet-agent 0.37.0 → 0.39.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +16 -2
- package/package.json +37 -37
- package/src/cli.mjs +135 -26
- package/src/harness/commands.mjs +192 -0
- package/src/harness/index.mjs +5 -0
- package/src/harness/state.mjs +10 -4
package/README.md
CHANGED
|
@@ -190,16 +190,30 @@ Gates are backed by **evidence**, not assertion. Each claimed gate must have a f
|
|
|
190
190
|
|
|
191
191
|
```bash
|
|
192
192
|
cadet-agent state validate # validate state against the schema
|
|
193
|
-
cadet-agent state migrate # atomically upgrade v1 → v2
|
|
193
|
+
cadet-agent state migrate # atomically upgrade v1 → v2 (no writes if it fails)
|
|
194
194
|
cadet-agent state transition --to review # enforce the matrix + evidence
|
|
195
195
|
cadet-agent harness verify --gate testsPassed --files src/a.cs # bounded, classified loop
|
|
196
196
|
cadet-agent harness report # budget consumption and failures (no secrets)
|
|
197
|
-
cadet-agent harness cleanup
|
|
197
|
+
cadet-agent harness cleanup --older-than-ms <n> # apply the retention policy (bound required)
|
|
198
198
|
cadet-agent harness capabilities # available CLI/Unity/MCP/hook/token/cost telemetry
|
|
199
199
|
```
|
|
200
200
|
|
|
201
201
|
Every command supports `--format human|json` and exits nonzero for invalid state, failed verification, budget exhaustion, stale evidence, or safety rejection.
|
|
202
202
|
|
|
203
|
+
Every command also **declares whether it writes**, and the declaration is enforced rather than
|
|
204
|
+
trusted. `--help` is read-only at any depth, `--dry-run` is honoured by every mutating command, and a
|
|
205
|
+
command declared read-only is tested to perform no writes. A destructive command that may run
|
|
206
|
+
unattended must require a content-bearing bound — `cleanup` requires `--older-than-ms` — so an agent
|
|
207
|
+
states *what* it acts on rather than merely *that* it approves. Run
|
|
208
|
+
`cadet-agent harness capabilities --format json` to read the registry.
|
|
209
|
+
|
|
210
|
+
Two refusals exist so an unattended agent cannot destroy evidence by accident:
|
|
211
|
+
|
|
212
|
+
```bash
|
|
213
|
+
cadet-agent harness cleanup # exits 1: deletes nothing without a bound
|
|
214
|
+
cadet-agent harness record --dry-run # reports what it would write, writes nothing
|
|
215
|
+
```
|
|
216
|
+
|
|
203
217
|
See `docs/guidance/HarnessTroubleshooting.md` for stale evidence, budget exhaustion, unavailable Unity CLI, and live MCP connection failures.
|
|
204
218
|
|
|
205
219
|
## Examples
|
package/package.json
CHANGED
|
@@ -1,37 +1,37 @@
|
|
|
1
|
-
{
|
|
2
|
-
"name": "cadet-agent",
|
|
3
|
-
"version": "0.
|
|
4
|
-
"description": "Cross-IDE agent framework for Unity/C# game-development — one-command install",
|
|
5
|
-
"type": "module",
|
|
6
|
-
"bin": {
|
|
7
|
-
"cadet-agent": "bin/cli.mjs"
|
|
8
|
-
},
|
|
9
|
-
"scripts": {
|
|
10
|
-
"test": "node --test test/*.test.mjs",
|
|
11
|
-
"lint": "lychee --offline --include-fragments \"**/*.md\"",
|
|
12
|
-
"verify": "npm test && npm run lint"
|
|
13
|
-
},
|
|
14
|
-
"files": [
|
|
15
|
-
"bin/",
|
|
16
|
-
"src/"
|
|
17
|
-
],
|
|
18
|
-
"keywords": [
|
|
19
|
-
"cadet",
|
|
20
|
-
"cadet-agent",
|
|
21
|
-
"unity",
|
|
22
|
-
"game-development",
|
|
23
|
-
"ai-agent",
|
|
24
|
-
"copilot",
|
|
25
|
-
"cursor",
|
|
26
|
-
"claude-code"
|
|
27
|
-
],
|
|
28
|
-
"license": "CC-BY-4.0",
|
|
29
|
-
"repository": {
|
|
30
|
-
"type": "git",
|
|
31
|
-
"url": "git+https://github.com/naishtech/cadet-agent.git"
|
|
32
|
-
},
|
|
33
|
-
"homepage": "https://github.com/naishtech/cadet-agent#readme",
|
|
34
|
-
"engines": {
|
|
35
|
-
"node": ">=18.0.0"
|
|
36
|
-
}
|
|
37
|
-
}
|
|
1
|
+
{
|
|
2
|
+
"name": "cadet-agent",
|
|
3
|
+
"version": "0.39.0",
|
|
4
|
+
"description": "Cross-IDE agent framework for Unity/C# game-development — one-command install",
|
|
5
|
+
"type": "module",
|
|
6
|
+
"bin": {
|
|
7
|
+
"cadet-agent": "bin/cli.mjs"
|
|
8
|
+
},
|
|
9
|
+
"scripts": {
|
|
10
|
+
"test": "node --test test/*.test.mjs",
|
|
11
|
+
"lint": "lychee --offline --include-fragments \"**/*.md\"",
|
|
12
|
+
"verify": "npm test && npm run lint"
|
|
13
|
+
},
|
|
14
|
+
"files": [
|
|
15
|
+
"bin/",
|
|
16
|
+
"src/"
|
|
17
|
+
],
|
|
18
|
+
"keywords": [
|
|
19
|
+
"cadet",
|
|
20
|
+
"cadet-agent",
|
|
21
|
+
"unity",
|
|
22
|
+
"game-development",
|
|
23
|
+
"ai-agent",
|
|
24
|
+
"copilot",
|
|
25
|
+
"cursor",
|
|
26
|
+
"claude-code"
|
|
27
|
+
],
|
|
28
|
+
"license": "CC-BY-4.0",
|
|
29
|
+
"repository": {
|
|
30
|
+
"type": "git",
|
|
31
|
+
"url": "git+https://github.com/naishtech/cadet-agent.git"
|
|
32
|
+
},
|
|
33
|
+
"homepage": "https://github.com/naishtech/cadet-agent#readme",
|
|
34
|
+
"engines": {
|
|
35
|
+
"node": ">=18.0.0"
|
|
36
|
+
}
|
|
37
|
+
}
|
package/src/cli.mjs
CHANGED
|
@@ -10,6 +10,7 @@ import {
|
|
|
10
10
|
parseTestInventory, parseStoryCriteria, compareCoverage, describeCoverageGaps,
|
|
11
11
|
createEvidence, newId, computeInputTreeHash, hashCriteria,
|
|
12
12
|
collectDeclaredTestNames, reconcileTestNames,
|
|
13
|
+
resolveCommand, describeCommand, describeAllCommands, checkUnattendedRequirements, COMMANDS,
|
|
13
14
|
} from './harness/index.mjs';
|
|
14
15
|
|
|
15
16
|
const __filename = fileURLToPath(import.meta.url);
|
|
@@ -68,50 +69,93 @@ function showHelp() {
|
|
|
68
69
|
--matrix TDD matrix markdown to check (harness matrix-check)
|
|
69
70
|
--inventory Newline-separated test names, when no report is available (harness matrix-check)
|
|
70
71
|
--agents-md keep|overwrite|merge for an existing AGENTS.md (init/sync)
|
|
72
|
+
--older-than-ms Age bound, in ms, for records cleanup may delete (harness cleanup; required)
|
|
73
|
+
--dry-run Report what a mutating command would do and write nothing (all mutating commands)
|
|
71
74
|
--yes, -y Never prompt; keep existing files (non-interactive installs)
|
|
72
|
-
--help, -h Show this help
|
|
75
|
+
--help, -h Show this help (valid at any depth; never writes)
|
|
73
76
|
--version, -v Show version number
|
|
74
77
|
`);
|
|
75
78
|
}
|
|
76
79
|
|
|
80
|
+
/**
|
|
81
|
+
* Does the invocation ask for help?
|
|
82
|
+
*
|
|
83
|
+
* Scanned against the raw argv rather than the parsed options on purpose. Once
|
|
84
|
+
* parsing begins, `--help` in a value position (`--target --help`) is consumed
|
|
85
|
+
* as another flag's argument and never seen again — so it must be detected
|
|
86
|
+
* before `parseArgs` runs. `--` ends flag scanning, so a literal `--help` after
|
|
87
|
+
* it is an operand and does not trigger help.
|
|
88
|
+
*/
|
|
89
|
+
function wantsHelp(argv) {
|
|
90
|
+
for (let i = 2; i < argv.length; i++) {
|
|
91
|
+
const a = argv[i];
|
|
92
|
+
if (a === '--') return false;
|
|
93
|
+
if (a === '--help' || a === '-h') return true;
|
|
94
|
+
}
|
|
95
|
+
return false;
|
|
96
|
+
}
|
|
97
|
+
|
|
77
98
|
function parseArgs(argv) {
|
|
78
99
|
const opts = { format: 'human', targetDir: process.cwd(), sourceUrl: null, rest: [] };
|
|
79
100
|
// argv[2] is the top-level command (`state`/`harness`/`init`/...); argv[3] begins
|
|
80
101
|
// the subcommand and its options.
|
|
81
|
-
|
|
102
|
+
//
|
|
103
|
+
// `value()` reads the argument a flag expects and rejects the case where the
|
|
104
|
+
// next token is itself a flag. Without this check `--target --format` bound
|
|
105
|
+
// the literal string "--format" as the target directory and then wrote a
|
|
106
|
+
// ledger into a directory named `--format/` — a stray write, from a typo, in
|
|
107
|
+
// an arbitrary place. A silently swallowed option is worse than a rejected
|
|
108
|
+
// one because the command still reports success.
|
|
109
|
+
//
|
|
110
|
+
// A negative number is allowed through: it is a plausible value (`--older-than-ms -1`)
|
|
111
|
+
// and cannot be mistaken for one of this CLI's flags, all of which are words.
|
|
112
|
+
//
|
|
113
|
+
// `i` is declared here, outside the loop, because `value()` must advance the
|
|
114
|
+
// shared cursor — a closure over a loop-scoped `i` would not exist yet at
|
|
115
|
+
// definition time.
|
|
116
|
+
let i = 3;
|
|
117
|
+
const value = (flag) => {
|
|
118
|
+
const next = argv[i + 1];
|
|
119
|
+
if (next === undefined || (next.startsWith('-') && !/^-\d/.test(next))) {
|
|
120
|
+
fail(opts, `Option ${flag} requires a value.`, () => 1, { ok: false, code: 'missing-option-value', option: flag });
|
|
121
|
+
}
|
|
122
|
+
i += 1;
|
|
123
|
+
return next;
|
|
124
|
+
};
|
|
125
|
+
for (i = 3; i < argv.length; i++) {
|
|
82
126
|
const a = argv[i];
|
|
83
127
|
switch (a) {
|
|
84
|
-
case '--target': case '-t': opts.targetDir =
|
|
85
|
-
case '--source': opts.sourceUrl =
|
|
86
|
-
case '--format': opts.format =
|
|
87
|
-
case '--to': opts.to =
|
|
88
|
-
case '--gate': opts.gate =
|
|
89
|
-
case '--command': opts.command =
|
|
90
|
-
case '--work-item': opts.workItemId =
|
|
91
|
-
case '--phase': opts.phase =
|
|
92
|
-
case '--run': opts.runId =
|
|
93
|
-
case '--type': opts.type =
|
|
94
|
-
case '--reason': opts.reason =
|
|
95
|
-
case '--expires-at': opts.expiresAt =
|
|
96
|
-
case '--environment': opts.environment =
|
|
97
|
-
case '--scope': opts.scope = (
|
|
98
|
-
case '--evidence-status': opts.evidenceStatus =
|
|
128
|
+
case '--target': case '-t': opts.targetDir = value(a); break;
|
|
129
|
+
case '--source': opts.sourceUrl = value(a); break;
|
|
130
|
+
case '--format': opts.format = value(a); break;
|
|
131
|
+
case '--to': opts.to = value(a); break;
|
|
132
|
+
case '--gate': opts.gate = value(a); break;
|
|
133
|
+
case '--command': opts.command = value(a); break;
|
|
134
|
+
case '--work-item': opts.workItemId = value(a); break;
|
|
135
|
+
case '--phase': opts.phase = value(a); break;
|
|
136
|
+
case '--run': opts.runId = value(a); break;
|
|
137
|
+
case '--type': opts.type = value(a); break;
|
|
138
|
+
case '--reason': opts.reason = value(a); break;
|
|
139
|
+
case '--expires-at': opts.expiresAt = value(a); break;
|
|
140
|
+
case '--environment': opts.environment = value(a); break;
|
|
141
|
+
case '--scope': opts.scope = value(a).split(',').map((s) => s.trim()).filter(Boolean); break;
|
|
142
|
+
case '--evidence-status': opts.evidenceStatus = value(a); break;
|
|
99
143
|
// Track that the flag was supplied even when its value is empty, so an
|
|
100
144
|
// empty `--files ""` is rejected rather than silently falling back to the
|
|
101
145
|
// working-tree scan (which could bind evidence to Cadet's own files).
|
|
102
|
-
case '--files': opts.filesGiven = true; opts.files = (
|
|
103
|
-
case '--story': opts.story =
|
|
104
|
-
case '--report': opts.report =
|
|
146
|
+
case '--files': opts.filesGiven = true; opts.files = value(a).split(',').map((s) => s.trim()).filter(Boolean); break;
|
|
147
|
+
case '--story': opts.story = value(a); break;
|
|
148
|
+
case '--report': opts.report = value(a); break;
|
|
105
149
|
// AR-1: the revision a gate record attests, so a gate-related fix claim
|
|
106
150
|
// can be traced to the commit that contains it.
|
|
107
|
-
case '--commit': opts.commitGiven = true; opts.commit =
|
|
108
|
-
case '--matrix': opts.matrix =
|
|
109
|
-
case '--inventory': opts.inventory =
|
|
151
|
+
case '--commit': opts.commitGiven = true; opts.commit = value(a); break;
|
|
152
|
+
case '--matrix': opts.matrix = value(a); break;
|
|
153
|
+
case '--inventory': opts.inventory = value(a); break;
|
|
110
154
|
case '--write-coverage': opts.writeCoverage = true; break;
|
|
111
155
|
case '--strict-orphans': opts.strictOrphans = true; break;
|
|
112
156
|
case '--dry-run': opts.dryRun = true; break;
|
|
113
|
-
case '--older-than-ms': opts.olderThanMs = Number(
|
|
114
|
-
case '--agents-md': opts.agentsMd =
|
|
157
|
+
case '--older-than-ms': opts.olderThanMs = Number(value(a)); break;
|
|
158
|
+
case '--agents-md': opts.agentsMd = value(a); break;
|
|
115
159
|
case '--yes': case '-y': opts.yes = true; break;
|
|
116
160
|
default: opts.rest.push(a);
|
|
117
161
|
}
|
|
@@ -270,7 +314,13 @@ async function cmdHarness(opts) {
|
|
|
270
314
|
|
|
271
315
|
if (sub === 'capabilities') {
|
|
272
316
|
const caps = detectCapabilities({ targetDir: opts.targetDir });
|
|
273
|
-
|
|
317
|
+
// The command registry is published here so an agent can *ask* which
|
|
318
|
+
// commands write instead of inferring it from a name or trusting a flag it
|
|
319
|
+
// must remember to pass. It is informational: the safety guarantee does not
|
|
320
|
+
// depend on the agent reading it, because the dispatcher enforces the
|
|
321
|
+
// registry regardless.
|
|
322
|
+
const commands = describeAllCommands();
|
|
323
|
+
if (opts.format === 'json') emit(opts, '', { ok: true, capabilities: caps, commands });
|
|
274
324
|
else {
|
|
275
325
|
console.log('Cadet-Agent capability report');
|
|
276
326
|
console.log(` CLI: ${caps.cli ? 'available' : 'unavailable'}`);
|
|
@@ -280,6 +330,11 @@ async function cmdHarness(opts) {
|
|
|
280
330
|
console.log(` Token telemetry:${caps.tokenTelemetry.provider ? ' provider' : ' estimate/unknown'}`);
|
|
281
331
|
console.log(` Cost telemetry: ${caps.costTelemetry.available ? 'available' : `unavailable (${caps.costTelemetry.reason})`}`);
|
|
282
332
|
console.log(` Note: ${caps.hook.note}`);
|
|
333
|
+
console.log(' Commands (mutating commands honour --dry-run; nothing writes without it being declared):');
|
|
334
|
+
for (const c of commands) {
|
|
335
|
+
const bound = c.requiresForUnattended.length ? ` [unattended requires ${c.requiresForUnattended.join(', ')}]` : '';
|
|
336
|
+
console.log(` ${c.mutates ? 'WRITES ' : 'read '} ${c.command}${bound}`);
|
|
337
|
+
}
|
|
283
338
|
}
|
|
284
339
|
return;
|
|
285
340
|
}
|
|
@@ -879,7 +934,61 @@ async function cmdHarness(opts) {
|
|
|
879
934
|
|
|
880
935
|
export async function run(argv) {
|
|
881
936
|
const command = argv[2];
|
|
937
|
+
|
|
938
|
+
// `--help`/`-h` is a global, read-only flag: it must be honoured at ANY depth
|
|
939
|
+
// (`harness record --help`, `state transition --help`) and must short-circuit
|
|
940
|
+
// before dispatch. Previously it was only recognised as `argv[2]`, so a nested
|
|
941
|
+
// help flag fell through into `parseArgs`'s `rest` array — and mutating
|
|
942
|
+
// subcommands acted on it. `harness record --help` appended a ledger,
|
|
943
|
+
// `harness cleanup --help` applied the retention policy and deleted run
|
|
944
|
+
// records, and `state migrate --help` wrote a `.v1.bak` backup. "Checking the
|
|
945
|
+
// help" is not a read-only operation if help is never actually checked.
|
|
946
|
+
if (wantsHelp(argv)) {
|
|
947
|
+
showHelp();
|
|
948
|
+
return;
|
|
949
|
+
}
|
|
950
|
+
|
|
882
951
|
const opts = parseArgs(argv);
|
|
952
|
+
const commandKey = resolveCommand(argv);
|
|
953
|
+
|
|
954
|
+
// Global `--dry-run`, driven by the registry rather than by each handler.
|
|
955
|
+
//
|
|
956
|
+
// This is the structural fix for the class of bug where a mutating command
|
|
957
|
+
// simply did not check the flag: `--dry-run` was parsed globally but honoured
|
|
958
|
+
// by one command, so `harness record --dry-run` wrote a ledger and
|
|
959
|
+
// `cleanup --dry-run` would have deleted records. Declaring which commands
|
|
960
|
+
// write, and honouring the flag for all of them here, means a new mutating
|
|
961
|
+
// command is covered the moment it is registered — no per-handler check to
|
|
962
|
+
// remember, and no way to forget one.
|
|
963
|
+
//
|
|
964
|
+
// A read-only command needs no interception: it is asserted not to write.
|
|
965
|
+
// A command may opt out when its dry-run is an *evaluation* rather than a
|
|
966
|
+
// no-op: `state transition --dry-run` reports the same verdict a real
|
|
967
|
+
// transition would, so its handler owns the flag. See `evaluatesOnDryRun`.
|
|
968
|
+
if (opts.dryRun === true && commandKey && COMMANDS[commandKey].mutates === true && COMMANDS[commandKey].evaluatesOnDryRun !== true) {
|
|
969
|
+
emit(
|
|
970
|
+
opts,
|
|
971
|
+
`✅ Dry run: \"${commandKey}\" would run and may write ${(COMMANDS[commandKey].writes || []).join(', ') || 'state'} — nothing was written.`,
|
|
972
|
+
{
|
|
973
|
+
ok: true,
|
|
974
|
+
dryRun: true,
|
|
975
|
+
applied: false,
|
|
976
|
+
command: commandKey,
|
|
977
|
+
mutates: true,
|
|
978
|
+
writes: COMMANDS[commandKey].writes || [],
|
|
979
|
+
},
|
|
980
|
+
);
|
|
981
|
+
return;
|
|
982
|
+
}
|
|
983
|
+
|
|
984
|
+
// A destructive command that may run unattended must carry an explicit,
|
|
985
|
+
// content-bearing bound on what it acts on. `--older-than-ms` states *what*
|
|
986
|
+
// to delete; a bare confirmation flag would only state *that* something was
|
|
987
|
+
// approved, which an agent can pass without knowing what it is approving.
|
|
988
|
+
if (commandKey) {
|
|
989
|
+
const guard = checkUnattendedRequirements(commandKey, opts);
|
|
990
|
+
if (!guard.ok) fail(opts, guard.reason, () => 1, { ok: false, command: commandKey, code: 'unattended-requirement-missing', missing: guard.missing });
|
|
991
|
+
}
|
|
883
992
|
|
|
884
993
|
// Validate the create-only policy flag early so a typo fails loudly.
|
|
885
994
|
const AGENTS_MD_MODES = ['keep', 'overwrite', 'merge'];
|
|
@@ -0,0 +1,192 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Command registry — the single source of truth for what the CLI's commands do.
|
|
3
|
+
*
|
|
4
|
+
* Why this exists: before it, dispatch was two hand-written `if (sub === …)`
|
|
5
|
+
* chains and nothing declared which commands write. Every safety property was a
|
|
6
|
+
* convention the *caller* had to remember:
|
|
7
|
+
*
|
|
8
|
+
* - `--dry-run` was parsed globally but honoured by exactly one command, so
|
|
9
|
+
* `harness record --dry-run` silently wrote a ledger. An opt-in flag only
|
|
10
|
+
* protects the caller who already knows to pass it.
|
|
11
|
+
* - Nothing asserted that a command documented as read-only performs no
|
|
12
|
+
* writes, so a write in `report`/`matrix-check` would go unnoticed.
|
|
13
|
+
* - `cleanup` deletes run records, and an unattended agent could invoke it
|
|
14
|
+
* with no bound on what it deletes.
|
|
15
|
+
*
|
|
16
|
+
* The registry inverts that: safety is a property of the command, enforced by
|
|
17
|
+
* the dispatcher and asserted by tests, not a rule the caller must recall. An
|
|
18
|
+
* agent that does not know what a command does still cannot write by accident.
|
|
19
|
+
*
|
|
20
|
+
* Contract: docs/core/HarnessContract.md C13.
|
|
21
|
+
*/
|
|
22
|
+
|
|
23
|
+
/** Every command the CLI exposes, keyed by its invocation path. */
|
|
24
|
+
export const COMMANDS = {
|
|
25
|
+
init: {
|
|
26
|
+
mutates: true,
|
|
27
|
+
summary: 'Install Cadet-Agent into the target directory.',
|
|
28
|
+
writes: ['AGENTS.md', '.cadet/**'],
|
|
29
|
+
unattended: true,
|
|
30
|
+
},
|
|
31
|
+
sync: {
|
|
32
|
+
mutates: true,
|
|
33
|
+
summary: 'Update the framework, preserving local policies and plans.',
|
|
34
|
+
writes: ['.cadet/**', 'AGENTS.md'],
|
|
35
|
+
unattended: true,
|
|
36
|
+
},
|
|
37
|
+
|
|
38
|
+
'state validate': {
|
|
39
|
+
mutates: false,
|
|
40
|
+
summary: 'Validate .cadet/state.json against the schema.',
|
|
41
|
+
},
|
|
42
|
+
'state migrate': {
|
|
43
|
+
mutates: true,
|
|
44
|
+
summary: 'Atomically migrate v1 state to the current version.',
|
|
45
|
+
writes: ['.cadet/state.json', '.cadet/state.json.v1.bak'],
|
|
46
|
+
unattended: false,
|
|
47
|
+
// A failed migration must leave the tree exactly as it found it: no backup,
|
|
48
|
+
// no partial write. The backup is an artifact of a *successful* migration,
|
|
49
|
+
// so producing one and then failing would be a write the caller did not get.
|
|
50
|
+
atomicFailure: true,
|
|
51
|
+
},
|
|
52
|
+
'state transition': {
|
|
53
|
+
mutates: true,
|
|
54
|
+
summary: 'Enforce the transition matrix and evidence; applies the transition.',
|
|
55
|
+
writes: ['.cadet/state.json'],
|
|
56
|
+
unattended: true,
|
|
57
|
+
// `--dry-run` here is not merely "don't write" — it reports the *same
|
|
58
|
+
// verdict* a real transition would (`allowed`, plus every missing or stale
|
|
59
|
+
// gate), which is the entire point of the flag. The handler therefore keeps
|
|
60
|
+
// ownership of the dry-run path and the global interception stands down.
|
|
61
|
+
// Collapsing the two meanings would replace a useful verdict with a bare
|
|
62
|
+
// "nothing was written". Contract C13.
|
|
63
|
+
evaluatesOnDryRun: true,
|
|
64
|
+
},
|
|
65
|
+
|
|
66
|
+
'harness record': {
|
|
67
|
+
mutates: true,
|
|
68
|
+
summary: 'Append a sanitized span/evidence/decision event to the run ledger.',
|
|
69
|
+
writes: ['.cadet/runs/**'],
|
|
70
|
+
unattended: true,
|
|
71
|
+
},
|
|
72
|
+
'harness confirm': {
|
|
73
|
+
mutates: true,
|
|
74
|
+
summary: 'Record manual-confirmation evidence (ledger + state).',
|
|
75
|
+
writes: ['.cadet/runs/**', '.cadet/state.json'],
|
|
76
|
+
unattended: true,
|
|
77
|
+
},
|
|
78
|
+
'harness verify': {
|
|
79
|
+
mutates: true,
|
|
80
|
+
summary: 'Run a bounded, classified verification loop.',
|
|
81
|
+
writes: ['.cadet/runs/**', '.cadet/state.json'],
|
|
82
|
+
unattended: true,
|
|
83
|
+
},
|
|
84
|
+
'harness verify-acs': {
|
|
85
|
+
mutates: true,
|
|
86
|
+
summary: 'Verify declared AC↔test coverage; may write coverage + state.',
|
|
87
|
+
writes: ['.cadet/runs/**', '.cadet/state.json', '*.coverage.json'],
|
|
88
|
+
unattended: true,
|
|
89
|
+
},
|
|
90
|
+
'harness report': {
|
|
91
|
+
mutates: false,
|
|
92
|
+
summary: 'Summarize budget consumption and failures.',
|
|
93
|
+
},
|
|
94
|
+
'harness matrix-check': {
|
|
95
|
+
mutates: false,
|
|
96
|
+
summary: 'Reconcile a TDD matrix against a compiled test inventory.',
|
|
97
|
+
},
|
|
98
|
+
'harness capabilities': {
|
|
99
|
+
mutates: false,
|
|
100
|
+
summary: 'Report available CLI/Unity/MCP/hook/token/cost telemetry.',
|
|
101
|
+
},
|
|
102
|
+
'harness cleanup': {
|
|
103
|
+
mutates: true,
|
|
104
|
+
summary: 'Apply the retention policy to .cadet/runs/, deleting run records.',
|
|
105
|
+
writes: ['.cadet/runs/**'],
|
|
106
|
+
// Destructive and irreversible for the records it removes. An unattended
|
|
107
|
+
// agent may only run it with an explicit, content-bearing bound on what it
|
|
108
|
+
// deletes — see `requiresForUnattended`. A bare "yes" would be a boolean the
|
|
109
|
+
// agent can always supply without knowing what it is approving.
|
|
110
|
+
unattended: false,
|
|
111
|
+
requiresForUnattended: ['--older-than-ms'],
|
|
112
|
+
},
|
|
113
|
+
};
|
|
114
|
+
|
|
115
|
+
/** The set of commands that write to the filesystem. */
|
|
116
|
+
export function mutatingCommands() {
|
|
117
|
+
return Object.entries(COMMANDS).filter(([, c]) => c.mutates).map(([k]) => k);
|
|
118
|
+
}
|
|
119
|
+
|
|
120
|
+
/** The set of commands that are guaranteed not to write. */
|
|
121
|
+
export function readOnlyCommands() {
|
|
122
|
+
return Object.entries(COMMANDS).filter(([, c]) => !c.mutates).map(([k]) => k);
|
|
123
|
+
}
|
|
124
|
+
|
|
125
|
+
/**
|
|
126
|
+
* Resolve the command key for a parsed invocation.
|
|
127
|
+
*
|
|
128
|
+
* Top-level commands (`init`, `sync`) are keyed by name; `state`/`harness`
|
|
129
|
+
* subcommands are keyed by `<group> <sub>`. Returns null when the invocation
|
|
130
|
+
* does not name a known command — an unknown command is a usage error, and the
|
|
131
|
+
* caller reports it rather than guessing a safety posture.
|
|
132
|
+
*/
|
|
133
|
+
export function resolveCommand(argv) {
|
|
134
|
+
const top = argv[2];
|
|
135
|
+
if (!top) return null;
|
|
136
|
+
if (top === 'state' || top === 'harness') {
|
|
137
|
+
const sub = argv[3];
|
|
138
|
+
if (!sub || sub.startsWith('-')) return null;
|
|
139
|
+
const key = `${top} ${sub}`;
|
|
140
|
+
return Object.hasOwn(COMMANDS, key) ? key : null;
|
|
141
|
+
}
|
|
142
|
+
return Object.hasOwn(COMMANDS, top) ? top : null;
|
|
143
|
+
}
|
|
144
|
+
|
|
145
|
+
/** Describe a command for machine consumption. */
|
|
146
|
+
export function describeCommand(key) {
|
|
147
|
+
const c = COMMANDS[key];
|
|
148
|
+
if (!c) return null;
|
|
149
|
+
return {
|
|
150
|
+
command: key,
|
|
151
|
+
mutates: c.mutates === true,
|
|
152
|
+
summary: c.summary,
|
|
153
|
+
writes: c.mutates ? c.writes ?? [] : [],
|
|
154
|
+
unattended: c.unattended !== false,
|
|
155
|
+
requiresForUnattended: c.requiresForUnattended ?? [],
|
|
156
|
+
// True when `--dry-run` yields a verdict from the command's own evaluator
|
|
157
|
+
// rather than a generic no-op acknowledgement.
|
|
158
|
+
evaluatesOnDryRun: c.evaluatesOnDryRun === true,
|
|
159
|
+
};
|
|
160
|
+
}
|
|
161
|
+
|
|
162
|
+
/** The full registry as a machine-readable list, sorted by command path. */
|
|
163
|
+
export function describeAllCommands() {
|
|
164
|
+
return Object.keys(COMMANDS).sort().map(describeCommand);
|
|
165
|
+
}
|
|
166
|
+
|
|
167
|
+
/**
|
|
168
|
+
* Check that an invocation carries the flags a command requires when it may run
|
|
169
|
+
* unattended. Returns `{ ok: true }` or `{ ok: false, missing, reason }`.
|
|
170
|
+
*
|
|
171
|
+
* `--dry-run` satisfies nothing here on purpose: a dry run does not perform the
|
|
172
|
+
* action, so it is never a substitute for the bound the action requires.
|
|
173
|
+
*/
|
|
174
|
+
export function checkUnattendedRequirements(key, opts) {
|
|
175
|
+
const c = COMMANDS[key];
|
|
176
|
+
if (!c || c.mutates !== true) return { ok: true };
|
|
177
|
+
const required = c.requiresForUnattended ?? [];
|
|
178
|
+
if (required.length === 0) return { ok: true };
|
|
179
|
+
|
|
180
|
+
const missing = required.filter((flag) => {
|
|
181
|
+
if (flag === '--older-than-ms') {
|
|
182
|
+
return !Number.isFinite(opts.olderThanMs);
|
|
183
|
+
}
|
|
184
|
+
return true;
|
|
185
|
+
});
|
|
186
|
+
if (missing.length === 0) return { ok: true };
|
|
187
|
+
return {
|
|
188
|
+
ok: false,
|
|
189
|
+
missing,
|
|
190
|
+
reason: `"${key}" deletes records irreversibly, so it requires ${missing.join(', ')} when run unattended.`,
|
|
191
|
+
};
|
|
192
|
+
}
|
package/src/harness/index.mjs
CHANGED
|
@@ -73,3 +73,8 @@ export {
|
|
|
73
73
|
export {
|
|
74
74
|
collectDeclaredTestNames, reconcileTestNames, inventoryFromCSharpSources,
|
|
75
75
|
} from './matrix-check.mjs';
|
|
76
|
+
|
|
77
|
+
export {
|
|
78
|
+
COMMANDS, mutatingCommands, readOnlyCommands, resolveCommand,
|
|
79
|
+
describeCommand, describeAllCommands, checkUnattendedRequirements,
|
|
80
|
+
} from './commands.mjs';
|
package/src/harness/state.mjs
CHANGED
|
@@ -601,16 +601,22 @@ export function migrateStateFile(statePath, { backup = true } = {}) {
|
|
|
601
601
|
const tmpPath = join(tmpDir, 'state.json');
|
|
602
602
|
try {
|
|
603
603
|
writeFileSync(tmpPath, JSON.stringify(state, null, 2) + '\n', 'utf-8');
|
|
604
|
-
|
|
605
|
-
|
|
606
|
-
|
|
607
|
-
//
|
|
604
|
+
// Validate BEFORE writing anything to the real tree. The backup is an
|
|
605
|
+
// artifact of a *successful* migration, so producing one and then failing
|
|
606
|
+
// would leave a write the caller never got and did not ask for — a failed
|
|
607
|
+
// `migrate` must leave the directory exactly as it found it. Checking first
|
|
608
|
+
// also means a bad migration cannot overwrite a previous good backup.
|
|
609
|
+
//
|
|
608
610
|
// This is a structural-only check: a freshly migrated state has no on-disk
|
|
609
611
|
// evidence to bind, so freshness is intentionally not evaluated here.
|
|
610
612
|
const check = validateState(state, { structuralOnly: true });
|
|
611
613
|
if (!check.valid) {
|
|
612
614
|
throw new StateError(`migrated state failed validation: ${check.errors.map((e) => `${e.path}: ${e.message}`).join('; ')}`);
|
|
613
615
|
}
|
|
616
|
+
if (backup) {
|
|
617
|
+
copyFileSync(statePath, `${statePath}.v1.bak`);
|
|
618
|
+
}
|
|
619
|
+
// A failed rename leaves the original in place.
|
|
614
620
|
renameSync(tmpPath, statePath);
|
|
615
621
|
} finally {
|
|
616
622
|
rmSync(tmpDir, { recursive: true, force: true });
|