cadet-agent 0.36.0 → 0.39.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +16 -2
- package/package.json +37 -37
- package/src/cli.mjs +140 -26
- package/src/harness/commands.mjs +192 -0
- package/src/harness/index.mjs +5 -0
- package/src/harness/policy.mjs +6 -0
- package/src/harness/state.mjs +31 -7
package/README.md
CHANGED
|
@@ -190,16 +190,30 @@ Gates are backed by **evidence**, not assertion. Each claimed gate must have a f
|
|
|
190
190
|
|
|
191
191
|
```bash
|
|
192
192
|
cadet-agent state validate # validate state against the schema
|
|
193
|
-
cadet-agent state migrate # atomically upgrade v1 → v2
|
|
193
|
+
cadet-agent state migrate # atomically upgrade v1 → v2 (no writes if it fails)
|
|
194
194
|
cadet-agent state transition --to review # enforce the matrix + evidence
|
|
195
195
|
cadet-agent harness verify --gate testsPassed --files src/a.cs # bounded, classified loop
|
|
196
196
|
cadet-agent harness report # budget consumption and failures (no secrets)
|
|
197
|
-
cadet-agent harness cleanup
|
|
197
|
+
cadet-agent harness cleanup --older-than-ms <n> # apply the retention policy (bound required)
|
|
198
198
|
cadet-agent harness capabilities # available CLI/Unity/MCP/hook/token/cost telemetry
|
|
199
199
|
```
|
|
200
200
|
|
|
201
201
|
Every command supports `--format human|json` and exits nonzero for invalid state, failed verification, budget exhaustion, stale evidence, or safety rejection.
|
|
202
202
|
|
|
203
|
+
Every command also **declares whether it writes**, and the declaration is enforced rather than
|
|
204
|
+
trusted. `--help` is read-only at any depth, `--dry-run` is honoured by every mutating command, and a
|
|
205
|
+
command declared read-only is tested to perform no writes. A destructive command that may run
|
|
206
|
+
unattended must require a content-bearing bound — `cleanup` requires `--older-than-ms` — so an agent
|
|
207
|
+
states *what* it acts on rather than merely *that* it approves. Run
|
|
208
|
+
`cadet-agent harness capabilities --format json` to read the registry.
|
|
209
|
+
|
|
210
|
+
Two refusals exist so an unattended agent cannot destroy evidence by accident:
|
|
211
|
+
|
|
212
|
+
```bash
|
|
213
|
+
cadet-agent harness cleanup # exits 1: deletes nothing without a bound
|
|
214
|
+
cadet-agent harness record --dry-run # reports what it would write, writes nothing
|
|
215
|
+
```
|
|
216
|
+
|
|
203
217
|
See `docs/guidance/HarnessTroubleshooting.md` for stale evidence, budget exhaustion, unavailable Unity CLI, and live MCP connection failures.
|
|
204
218
|
|
|
205
219
|
## Examples
|
package/package.json
CHANGED
|
@@ -1,37 +1,37 @@
|
|
|
1
|
-
{
|
|
2
|
-
"name": "cadet-agent",
|
|
3
|
-
"version": "0.
|
|
4
|
-
"description": "Cross-IDE agent framework for Unity/C# game-development — one-command install",
|
|
5
|
-
"type": "module",
|
|
6
|
-
"bin": {
|
|
7
|
-
"cadet-agent": "bin/cli.mjs"
|
|
8
|
-
},
|
|
9
|
-
"scripts": {
|
|
10
|
-
"test": "node --test test/*.test.mjs",
|
|
11
|
-
"lint": "lychee --offline --include-fragments \"**/*.md\"",
|
|
12
|
-
"verify": "npm test && npm run lint"
|
|
13
|
-
},
|
|
14
|
-
"files": [
|
|
15
|
-
"bin/",
|
|
16
|
-
"src/"
|
|
17
|
-
],
|
|
18
|
-
"keywords": [
|
|
19
|
-
"cadet",
|
|
20
|
-
"cadet-agent",
|
|
21
|
-
"unity",
|
|
22
|
-
"game-development",
|
|
23
|
-
"ai-agent",
|
|
24
|
-
"copilot",
|
|
25
|
-
"cursor",
|
|
26
|
-
"claude-code"
|
|
27
|
-
],
|
|
28
|
-
"license": "CC-BY-4.0",
|
|
29
|
-
"repository": {
|
|
30
|
-
"type": "git",
|
|
31
|
-
"url": "git+https://github.com/naishtech/cadet-agent.git"
|
|
32
|
-
},
|
|
33
|
-
"homepage": "https://github.com/naishtech/cadet-agent#readme",
|
|
34
|
-
"engines": {
|
|
35
|
-
"node": ">=18.0.0"
|
|
36
|
-
}
|
|
37
|
-
}
|
|
1
|
+
{
|
|
2
|
+
"name": "cadet-agent",
|
|
3
|
+
"version": "0.39.0",
|
|
4
|
+
"description": "Cross-IDE agent framework for Unity/C# game-development — one-command install",
|
|
5
|
+
"type": "module",
|
|
6
|
+
"bin": {
|
|
7
|
+
"cadet-agent": "bin/cli.mjs"
|
|
8
|
+
},
|
|
9
|
+
"scripts": {
|
|
10
|
+
"test": "node --test test/*.test.mjs",
|
|
11
|
+
"lint": "lychee --offline --include-fragments \"**/*.md\"",
|
|
12
|
+
"verify": "npm test && npm run lint"
|
|
13
|
+
},
|
|
14
|
+
"files": [
|
|
15
|
+
"bin/",
|
|
16
|
+
"src/"
|
|
17
|
+
],
|
|
18
|
+
"keywords": [
|
|
19
|
+
"cadet",
|
|
20
|
+
"cadet-agent",
|
|
21
|
+
"unity",
|
|
22
|
+
"game-development",
|
|
23
|
+
"ai-agent",
|
|
24
|
+
"copilot",
|
|
25
|
+
"cursor",
|
|
26
|
+
"claude-code"
|
|
27
|
+
],
|
|
28
|
+
"license": "CC-BY-4.0",
|
|
29
|
+
"repository": {
|
|
30
|
+
"type": "git",
|
|
31
|
+
"url": "git+https://github.com/naishtech/cadet-agent.git"
|
|
32
|
+
},
|
|
33
|
+
"homepage": "https://github.com/naishtech/cadet-agent#readme",
|
|
34
|
+
"engines": {
|
|
35
|
+
"node": ">=18.0.0"
|
|
36
|
+
}
|
|
37
|
+
}
|
package/src/cli.mjs
CHANGED
|
@@ -10,6 +10,7 @@ import {
|
|
|
10
10
|
parseTestInventory, parseStoryCriteria, compareCoverage, describeCoverageGaps,
|
|
11
11
|
createEvidence, newId, computeInputTreeHash, hashCriteria,
|
|
12
12
|
collectDeclaredTestNames, reconcileTestNames,
|
|
13
|
+
resolveCommand, describeCommand, describeAllCommands, checkUnattendedRequirements, COMMANDS,
|
|
13
14
|
} from './harness/index.mjs';
|
|
14
15
|
|
|
15
16
|
const __filename = fileURLToPath(import.meta.url);
|
|
@@ -58,55 +59,103 @@ function showHelp() {
|
|
|
58
59
|
--gate Gate name (harness verify|confirm)
|
|
59
60
|
--command Command override (harness verify)
|
|
60
61
|
--files Comma-separated relevant files to bind evidence to (harness verify|confirm)
|
|
62
|
+
--commit Revision the gate attests, as a hex SHA (harness verify|confirm)
|
|
61
63
|
--reason Why automation was unavailable (harness confirm)
|
|
62
64
|
--expires-at ISO-8601 expiry bounding the confirmation (harness confirm)
|
|
63
65
|
--environment key=value,... describing what was verified (harness confirm)
|
|
64
66
|
--scope Comma-separated scope of the confirmation (harness confirm)
|
|
67
|
+
--story Story markdown declaring the acceptance criteria (harness verify-acs)
|
|
68
|
+
--report Test report to derive the inventory from (harness verify-acs|matrix-check)
|
|
69
|
+
--matrix TDD matrix markdown to check (harness matrix-check)
|
|
70
|
+
--inventory Newline-separated test names, when no report is available (harness matrix-check)
|
|
65
71
|
--agents-md keep|overwrite|merge for an existing AGENTS.md (init/sync)
|
|
72
|
+
--older-than-ms Age bound, in ms, for records cleanup may delete (harness cleanup; required)
|
|
73
|
+
--dry-run Report what a mutating command would do and write nothing (all mutating commands)
|
|
66
74
|
--yes, -y Never prompt; keep existing files (non-interactive installs)
|
|
67
|
-
--help, -h Show this help
|
|
75
|
+
--help, -h Show this help (valid at any depth; never writes)
|
|
68
76
|
--version, -v Show version number
|
|
69
77
|
`);
|
|
70
78
|
}
|
|
71
79
|
|
|
80
|
+
/**
|
|
81
|
+
* Does the invocation ask for help?
|
|
82
|
+
*
|
|
83
|
+
* Scanned against the raw argv rather than the parsed options on purpose. Once
|
|
84
|
+
* parsing begins, `--help` in a value position (`--target --help`) is consumed
|
|
85
|
+
* as another flag's argument and never seen again — so it must be detected
|
|
86
|
+
* before `parseArgs` runs. `--` ends flag scanning, so a literal `--help` after
|
|
87
|
+
* it is an operand and does not trigger help.
|
|
88
|
+
*/
|
|
89
|
+
function wantsHelp(argv) {
|
|
90
|
+
for (let i = 2; i < argv.length; i++) {
|
|
91
|
+
const a = argv[i];
|
|
92
|
+
if (a === '--') return false;
|
|
93
|
+
if (a === '--help' || a === '-h') return true;
|
|
94
|
+
}
|
|
95
|
+
return false;
|
|
96
|
+
}
|
|
97
|
+
|
|
72
98
|
function parseArgs(argv) {
|
|
73
99
|
const opts = { format: 'human', targetDir: process.cwd(), sourceUrl: null, rest: [] };
|
|
74
100
|
// argv[2] is the top-level command (`state`/`harness`/`init`/...); argv[3] begins
|
|
75
101
|
// the subcommand and its options.
|
|
76
|
-
|
|
102
|
+
//
|
|
103
|
+
// `value()` reads the argument a flag expects and rejects the case where the
|
|
104
|
+
// next token is itself a flag. Without this check `--target --format` bound
|
|
105
|
+
// the literal string "--format" as the target directory and then wrote a
|
|
106
|
+
// ledger into a directory named `--format/` — a stray write, from a typo, in
|
|
107
|
+
// an arbitrary place. A silently swallowed option is worse than a rejected
|
|
108
|
+
// one because the command still reports success.
|
|
109
|
+
//
|
|
110
|
+
// A negative number is allowed through: it is a plausible value (`--older-than-ms -1`)
|
|
111
|
+
// and cannot be mistaken for one of this CLI's flags, all of which are words.
|
|
112
|
+
//
|
|
113
|
+
// `i` is declared here, outside the loop, because `value()` must advance the
|
|
114
|
+
// shared cursor — a closure over a loop-scoped `i` would not exist yet at
|
|
115
|
+
// definition time.
|
|
116
|
+
let i = 3;
|
|
117
|
+
const value = (flag) => {
|
|
118
|
+
const next = argv[i + 1];
|
|
119
|
+
if (next === undefined || (next.startsWith('-') && !/^-\d/.test(next))) {
|
|
120
|
+
fail(opts, `Option ${flag} requires a value.`, () => 1, { ok: false, code: 'missing-option-value', option: flag });
|
|
121
|
+
}
|
|
122
|
+
i += 1;
|
|
123
|
+
return next;
|
|
124
|
+
};
|
|
125
|
+
for (i = 3; i < argv.length; i++) {
|
|
77
126
|
const a = argv[i];
|
|
78
127
|
switch (a) {
|
|
79
|
-
case '--target': case '-t': opts.targetDir =
|
|
80
|
-
case '--source': opts.sourceUrl =
|
|
81
|
-
case '--format': opts.format =
|
|
82
|
-
case '--to': opts.to =
|
|
83
|
-
case '--gate': opts.gate =
|
|
84
|
-
case '--command': opts.command =
|
|
85
|
-
case '--work-item': opts.workItemId =
|
|
86
|
-
case '--phase': opts.phase =
|
|
87
|
-
case '--run': opts.runId =
|
|
88
|
-
case '--type': opts.type =
|
|
89
|
-
case '--reason': opts.reason =
|
|
90
|
-
case '--expires-at': opts.expiresAt =
|
|
91
|
-
case '--environment': opts.environment =
|
|
92
|
-
case '--scope': opts.scope = (
|
|
93
|
-
case '--evidence-status': opts.evidenceStatus =
|
|
128
|
+
case '--target': case '-t': opts.targetDir = value(a); break;
|
|
129
|
+
case '--source': opts.sourceUrl = value(a); break;
|
|
130
|
+
case '--format': opts.format = value(a); break;
|
|
131
|
+
case '--to': opts.to = value(a); break;
|
|
132
|
+
case '--gate': opts.gate = value(a); break;
|
|
133
|
+
case '--command': opts.command = value(a); break;
|
|
134
|
+
case '--work-item': opts.workItemId = value(a); break;
|
|
135
|
+
case '--phase': opts.phase = value(a); break;
|
|
136
|
+
case '--run': opts.runId = value(a); break;
|
|
137
|
+
case '--type': opts.type = value(a); break;
|
|
138
|
+
case '--reason': opts.reason = value(a); break;
|
|
139
|
+
case '--expires-at': opts.expiresAt = value(a); break;
|
|
140
|
+
case '--environment': opts.environment = value(a); break;
|
|
141
|
+
case '--scope': opts.scope = value(a).split(',').map((s) => s.trim()).filter(Boolean); break;
|
|
142
|
+
case '--evidence-status': opts.evidenceStatus = value(a); break;
|
|
94
143
|
// Track that the flag was supplied even when its value is empty, so an
|
|
95
144
|
// empty `--files ""` is rejected rather than silently falling back to the
|
|
96
145
|
// working-tree scan (which could bind evidence to Cadet's own files).
|
|
97
|
-
case '--files': opts.filesGiven = true; opts.files = (
|
|
98
|
-
case '--story': opts.story =
|
|
99
|
-
case '--report': opts.report =
|
|
146
|
+
case '--files': opts.filesGiven = true; opts.files = value(a).split(',').map((s) => s.trim()).filter(Boolean); break;
|
|
147
|
+
case '--story': opts.story = value(a); break;
|
|
148
|
+
case '--report': opts.report = value(a); break;
|
|
100
149
|
// AR-1: the revision a gate record attests, so a gate-related fix claim
|
|
101
150
|
// can be traced to the commit that contains it.
|
|
102
|
-
case '--commit': opts.commitGiven = true; opts.commit =
|
|
103
|
-
case '--matrix': opts.matrix =
|
|
104
|
-
case '--inventory': opts.inventory =
|
|
151
|
+
case '--commit': opts.commitGiven = true; opts.commit = value(a); break;
|
|
152
|
+
case '--matrix': opts.matrix = value(a); break;
|
|
153
|
+
case '--inventory': opts.inventory = value(a); break;
|
|
105
154
|
case '--write-coverage': opts.writeCoverage = true; break;
|
|
106
155
|
case '--strict-orphans': opts.strictOrphans = true; break;
|
|
107
156
|
case '--dry-run': opts.dryRun = true; break;
|
|
108
|
-
case '--older-than-ms': opts.olderThanMs = Number(
|
|
109
|
-
case '--agents-md': opts.agentsMd =
|
|
157
|
+
case '--older-than-ms': opts.olderThanMs = Number(value(a)); break;
|
|
158
|
+
case '--agents-md': opts.agentsMd = value(a); break;
|
|
110
159
|
case '--yes': case '-y': opts.yes = true; break;
|
|
111
160
|
default: opts.rest.push(a);
|
|
112
161
|
}
|
|
@@ -265,7 +314,13 @@ async function cmdHarness(opts) {
|
|
|
265
314
|
|
|
266
315
|
if (sub === 'capabilities') {
|
|
267
316
|
const caps = detectCapabilities({ targetDir: opts.targetDir });
|
|
268
|
-
|
|
317
|
+
// The command registry is published here so an agent can *ask* which
|
|
318
|
+
// commands write instead of inferring it from a name or trusting a flag it
|
|
319
|
+
// must remember to pass. It is informational: the safety guarantee does not
|
|
320
|
+
// depend on the agent reading it, because the dispatcher enforces the
|
|
321
|
+
// registry regardless.
|
|
322
|
+
const commands = describeAllCommands();
|
|
323
|
+
if (opts.format === 'json') emit(opts, '', { ok: true, capabilities: caps, commands });
|
|
269
324
|
else {
|
|
270
325
|
console.log('Cadet-Agent capability report');
|
|
271
326
|
console.log(` CLI: ${caps.cli ? 'available' : 'unavailable'}`);
|
|
@@ -275,6 +330,11 @@ async function cmdHarness(opts) {
|
|
|
275
330
|
console.log(` Token telemetry:${caps.tokenTelemetry.provider ? ' provider' : ' estimate/unknown'}`);
|
|
276
331
|
console.log(` Cost telemetry: ${caps.costTelemetry.available ? 'available' : `unavailable (${caps.costTelemetry.reason})`}`);
|
|
277
332
|
console.log(` Note: ${caps.hook.note}`);
|
|
333
|
+
console.log(' Commands (mutating commands honour --dry-run; nothing writes without it being declared):');
|
|
334
|
+
for (const c of commands) {
|
|
335
|
+
const bound = c.requiresForUnattended.length ? ` [unattended requires ${c.requiresForUnattended.join(', ')}]` : '';
|
|
336
|
+
console.log(` ${c.mutates ? 'WRITES ' : 'read '} ${c.command}${bound}`);
|
|
337
|
+
}
|
|
278
338
|
}
|
|
279
339
|
return;
|
|
280
340
|
}
|
|
@@ -874,7 +934,61 @@ async function cmdHarness(opts) {
|
|
|
874
934
|
|
|
875
935
|
export async function run(argv) {
|
|
876
936
|
const command = argv[2];
|
|
937
|
+
|
|
938
|
+
// `--help`/`-h` is a global, read-only flag: it must be honoured at ANY depth
|
|
939
|
+
// (`harness record --help`, `state transition --help`) and must short-circuit
|
|
940
|
+
// before dispatch. Previously it was only recognised as `argv[2]`, so a nested
|
|
941
|
+
// help flag fell through into `parseArgs`'s `rest` array — and mutating
|
|
942
|
+
// subcommands acted on it. `harness record --help` appended a ledger,
|
|
943
|
+
// `harness cleanup --help` applied the retention policy and deleted run
|
|
944
|
+
// records, and `state migrate --help` wrote a `.v1.bak` backup. "Checking the
|
|
945
|
+
// help" is not a read-only operation if help is never actually checked.
|
|
946
|
+
if (wantsHelp(argv)) {
|
|
947
|
+
showHelp();
|
|
948
|
+
return;
|
|
949
|
+
}
|
|
950
|
+
|
|
877
951
|
const opts = parseArgs(argv);
|
|
952
|
+
const commandKey = resolveCommand(argv);
|
|
953
|
+
|
|
954
|
+
// Global `--dry-run`, driven by the registry rather than by each handler.
|
|
955
|
+
//
|
|
956
|
+
// This is the structural fix for the class of bug where a mutating command
|
|
957
|
+
// simply did not check the flag: `--dry-run` was parsed globally but honoured
|
|
958
|
+
// by one command, so `harness record --dry-run` wrote a ledger and
|
|
959
|
+
// `cleanup --dry-run` would have deleted records. Declaring which commands
|
|
960
|
+
// write, and honouring the flag for all of them here, means a new mutating
|
|
961
|
+
// command is covered the moment it is registered — no per-handler check to
|
|
962
|
+
// remember, and no way to forget one.
|
|
963
|
+
//
|
|
964
|
+
// A read-only command needs no interception: it is asserted not to write.
|
|
965
|
+
// A command may opt out when its dry-run is an *evaluation* rather than a
|
|
966
|
+
// no-op: `state transition --dry-run` reports the same verdict a real
|
|
967
|
+
// transition would, so its handler owns the flag. See `evaluatesOnDryRun`.
|
|
968
|
+
if (opts.dryRun === true && commandKey && COMMANDS[commandKey].mutates === true && COMMANDS[commandKey].evaluatesOnDryRun !== true) {
|
|
969
|
+
emit(
|
|
970
|
+
opts,
|
|
971
|
+
`✅ Dry run: \"${commandKey}\" would run and may write ${(COMMANDS[commandKey].writes || []).join(', ') || 'state'} — nothing was written.`,
|
|
972
|
+
{
|
|
973
|
+
ok: true,
|
|
974
|
+
dryRun: true,
|
|
975
|
+
applied: false,
|
|
976
|
+
command: commandKey,
|
|
977
|
+
mutates: true,
|
|
978
|
+
writes: COMMANDS[commandKey].writes || [],
|
|
979
|
+
},
|
|
980
|
+
);
|
|
981
|
+
return;
|
|
982
|
+
}
|
|
983
|
+
|
|
984
|
+
// A destructive command that may run unattended must carry an explicit,
|
|
985
|
+
// content-bearing bound on what it acts on. `--older-than-ms` states *what*
|
|
986
|
+
// to delete; a bare confirmation flag would only state *that* something was
|
|
987
|
+
// approved, which an agent can pass without knowing what it is approving.
|
|
988
|
+
if (commandKey) {
|
|
989
|
+
const guard = checkUnattendedRequirements(commandKey, opts);
|
|
990
|
+
if (!guard.ok) fail(opts, guard.reason, () => 1, { ok: false, command: commandKey, code: 'unattended-requirement-missing', missing: guard.missing });
|
|
991
|
+
}
|
|
878
992
|
|
|
879
993
|
// Validate the create-only policy flag early so a typo fails loudly.
|
|
880
994
|
const AGENTS_MD_MODES = ['keep', 'overwrite', 'merge'];
|
|
@@ -0,0 +1,192 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Command registry — the single source of truth for what the CLI's commands do.
|
|
3
|
+
*
|
|
4
|
+
* Why this exists: before it, dispatch was two hand-written `if (sub === …)`
|
|
5
|
+
* chains and nothing declared which commands write. Every safety property was a
|
|
6
|
+
* convention the *caller* had to remember:
|
|
7
|
+
*
|
|
8
|
+
* - `--dry-run` was parsed globally but honoured by exactly one command, so
|
|
9
|
+
* `harness record --dry-run` silently wrote a ledger. An opt-in flag only
|
|
10
|
+
* protects the caller who already knows to pass it.
|
|
11
|
+
* - Nothing asserted that a command documented as read-only performs no
|
|
12
|
+
* writes, so a write in `report`/`matrix-check` would go unnoticed.
|
|
13
|
+
* - `cleanup` deletes run records, and an unattended agent could invoke it
|
|
14
|
+
* with no bound on what it deletes.
|
|
15
|
+
*
|
|
16
|
+
* The registry inverts that: safety is a property of the command, enforced by
|
|
17
|
+
* the dispatcher and asserted by tests, not a rule the caller must recall. An
|
|
18
|
+
* agent that does not know what a command does still cannot write by accident.
|
|
19
|
+
*
|
|
20
|
+
* Contract: docs/core/HarnessContract.md C13.
|
|
21
|
+
*/
|
|
22
|
+
|
|
23
|
+
/** Every command the CLI exposes, keyed by its invocation path. */
|
|
24
|
+
export const COMMANDS = {
|
|
25
|
+
init: {
|
|
26
|
+
mutates: true,
|
|
27
|
+
summary: 'Install Cadet-Agent into the target directory.',
|
|
28
|
+
writes: ['AGENTS.md', '.cadet/**'],
|
|
29
|
+
unattended: true,
|
|
30
|
+
},
|
|
31
|
+
sync: {
|
|
32
|
+
mutates: true,
|
|
33
|
+
summary: 'Update the framework, preserving local policies and plans.',
|
|
34
|
+
writes: ['.cadet/**', 'AGENTS.md'],
|
|
35
|
+
unattended: true,
|
|
36
|
+
},
|
|
37
|
+
|
|
38
|
+
'state validate': {
|
|
39
|
+
mutates: false,
|
|
40
|
+
summary: 'Validate .cadet/state.json against the schema.',
|
|
41
|
+
},
|
|
42
|
+
'state migrate': {
|
|
43
|
+
mutates: true,
|
|
44
|
+
summary: 'Atomically migrate v1 state to the current version.',
|
|
45
|
+
writes: ['.cadet/state.json', '.cadet/state.json.v1.bak'],
|
|
46
|
+
unattended: false,
|
|
47
|
+
// A failed migration must leave the tree exactly as it found it: no backup,
|
|
48
|
+
// no partial write. The backup is an artifact of a *successful* migration,
|
|
49
|
+
// so producing one and then failing would be a write the caller did not get.
|
|
50
|
+
atomicFailure: true,
|
|
51
|
+
},
|
|
52
|
+
'state transition': {
|
|
53
|
+
mutates: true,
|
|
54
|
+
summary: 'Enforce the transition matrix and evidence; applies the transition.',
|
|
55
|
+
writes: ['.cadet/state.json'],
|
|
56
|
+
unattended: true,
|
|
57
|
+
// `--dry-run` here is not merely "don't write" — it reports the *same
|
|
58
|
+
// verdict* a real transition would (`allowed`, plus every missing or stale
|
|
59
|
+
// gate), which is the entire point of the flag. The handler therefore keeps
|
|
60
|
+
// ownership of the dry-run path and the global interception stands down.
|
|
61
|
+
// Collapsing the two meanings would replace a useful verdict with a bare
|
|
62
|
+
// "nothing was written". Contract C13.
|
|
63
|
+
evaluatesOnDryRun: true,
|
|
64
|
+
},
|
|
65
|
+
|
|
66
|
+
'harness record': {
|
|
67
|
+
mutates: true,
|
|
68
|
+
summary: 'Append a sanitized span/evidence/decision event to the run ledger.',
|
|
69
|
+
writes: ['.cadet/runs/**'],
|
|
70
|
+
unattended: true,
|
|
71
|
+
},
|
|
72
|
+
'harness confirm': {
|
|
73
|
+
mutates: true,
|
|
74
|
+
summary: 'Record manual-confirmation evidence (ledger + state).',
|
|
75
|
+
writes: ['.cadet/runs/**', '.cadet/state.json'],
|
|
76
|
+
unattended: true,
|
|
77
|
+
},
|
|
78
|
+
'harness verify': {
|
|
79
|
+
mutates: true,
|
|
80
|
+
summary: 'Run a bounded, classified verification loop.',
|
|
81
|
+
writes: ['.cadet/runs/**', '.cadet/state.json'],
|
|
82
|
+
unattended: true,
|
|
83
|
+
},
|
|
84
|
+
'harness verify-acs': {
|
|
85
|
+
mutates: true,
|
|
86
|
+
summary: 'Verify declared AC↔test coverage; may write coverage + state.',
|
|
87
|
+
writes: ['.cadet/runs/**', '.cadet/state.json', '*.coverage.json'],
|
|
88
|
+
unattended: true,
|
|
89
|
+
},
|
|
90
|
+
'harness report': {
|
|
91
|
+
mutates: false,
|
|
92
|
+
summary: 'Summarize budget consumption and failures.',
|
|
93
|
+
},
|
|
94
|
+
'harness matrix-check': {
|
|
95
|
+
mutates: false,
|
|
96
|
+
summary: 'Reconcile a TDD matrix against a compiled test inventory.',
|
|
97
|
+
},
|
|
98
|
+
'harness capabilities': {
|
|
99
|
+
mutates: false,
|
|
100
|
+
summary: 'Report available CLI/Unity/MCP/hook/token/cost telemetry.',
|
|
101
|
+
},
|
|
102
|
+
'harness cleanup': {
|
|
103
|
+
mutates: true,
|
|
104
|
+
summary: 'Apply the retention policy to .cadet/runs/, deleting run records.',
|
|
105
|
+
writes: ['.cadet/runs/**'],
|
|
106
|
+
// Destructive and irreversible for the records it removes. An unattended
|
|
107
|
+
// agent may only run it with an explicit, content-bearing bound on what it
|
|
108
|
+
// deletes — see `requiresForUnattended`. A bare "yes" would be a boolean the
|
|
109
|
+
// agent can always supply without knowing what it is approving.
|
|
110
|
+
unattended: false,
|
|
111
|
+
requiresForUnattended: ['--older-than-ms'],
|
|
112
|
+
},
|
|
113
|
+
};
|
|
114
|
+
|
|
115
|
+
/** The set of commands that write to the filesystem. */
|
|
116
|
+
export function mutatingCommands() {
|
|
117
|
+
return Object.entries(COMMANDS).filter(([, c]) => c.mutates).map(([k]) => k);
|
|
118
|
+
}
|
|
119
|
+
|
|
120
|
+
/** The set of commands that are guaranteed not to write. */
|
|
121
|
+
export function readOnlyCommands() {
|
|
122
|
+
return Object.entries(COMMANDS).filter(([, c]) => !c.mutates).map(([k]) => k);
|
|
123
|
+
}
|
|
124
|
+
|
|
125
|
+
/**
|
|
126
|
+
* Resolve the command key for a parsed invocation.
|
|
127
|
+
*
|
|
128
|
+
* Top-level commands (`init`, `sync`) are keyed by name; `state`/`harness`
|
|
129
|
+
* subcommands are keyed by `<group> <sub>`. Returns null when the invocation
|
|
130
|
+
* does not name a known command — an unknown command is a usage error, and the
|
|
131
|
+
* caller reports it rather than guessing a safety posture.
|
|
132
|
+
*/
|
|
133
|
+
export function resolveCommand(argv) {
|
|
134
|
+
const top = argv[2];
|
|
135
|
+
if (!top) return null;
|
|
136
|
+
if (top === 'state' || top === 'harness') {
|
|
137
|
+
const sub = argv[3];
|
|
138
|
+
if (!sub || sub.startsWith('-')) return null;
|
|
139
|
+
const key = `${top} ${sub}`;
|
|
140
|
+
return Object.hasOwn(COMMANDS, key) ? key : null;
|
|
141
|
+
}
|
|
142
|
+
return Object.hasOwn(COMMANDS, top) ? top : null;
|
|
143
|
+
}
|
|
144
|
+
|
|
145
|
+
/** Describe a command for machine consumption. */
|
|
146
|
+
export function describeCommand(key) {
|
|
147
|
+
const c = COMMANDS[key];
|
|
148
|
+
if (!c) return null;
|
|
149
|
+
return {
|
|
150
|
+
command: key,
|
|
151
|
+
mutates: c.mutates === true,
|
|
152
|
+
summary: c.summary,
|
|
153
|
+
writes: c.mutates ? c.writes ?? [] : [],
|
|
154
|
+
unattended: c.unattended !== false,
|
|
155
|
+
requiresForUnattended: c.requiresForUnattended ?? [],
|
|
156
|
+
// True when `--dry-run` yields a verdict from the command's own evaluator
|
|
157
|
+
// rather than a generic no-op acknowledgement.
|
|
158
|
+
evaluatesOnDryRun: c.evaluatesOnDryRun === true,
|
|
159
|
+
};
|
|
160
|
+
}
|
|
161
|
+
|
|
162
|
+
/** The full registry as a machine-readable list, sorted by command path. */
|
|
163
|
+
export function describeAllCommands() {
|
|
164
|
+
return Object.keys(COMMANDS).sort().map(describeCommand);
|
|
165
|
+
}
|
|
166
|
+
|
|
167
|
+
/**
|
|
168
|
+
* Check that an invocation carries the flags a command requires when it may run
|
|
169
|
+
* unattended. Returns `{ ok: true }` or `{ ok: false, missing, reason }`.
|
|
170
|
+
*
|
|
171
|
+
* `--dry-run` satisfies nothing here on purpose: a dry run does not perform the
|
|
172
|
+
* action, so it is never a substitute for the bound the action requires.
|
|
173
|
+
*/
|
|
174
|
+
export function checkUnattendedRequirements(key, opts) {
|
|
175
|
+
const c = COMMANDS[key];
|
|
176
|
+
if (!c || c.mutates !== true) return { ok: true };
|
|
177
|
+
const required = c.requiresForUnattended ?? [];
|
|
178
|
+
if (required.length === 0) return { ok: true };
|
|
179
|
+
|
|
180
|
+
const missing = required.filter((flag) => {
|
|
181
|
+
if (flag === '--older-than-ms') {
|
|
182
|
+
return !Number.isFinite(opts.olderThanMs);
|
|
183
|
+
}
|
|
184
|
+
return true;
|
|
185
|
+
});
|
|
186
|
+
if (missing.length === 0) return { ok: true };
|
|
187
|
+
return {
|
|
188
|
+
ok: false,
|
|
189
|
+
missing,
|
|
190
|
+
reason: `"${key}" deletes records irreversibly, so it requires ${missing.join(', ')} when run unattended.`,
|
|
191
|
+
};
|
|
192
|
+
}
|
package/src/harness/index.mjs
CHANGED
|
@@ -73,3 +73,8 @@ export {
|
|
|
73
73
|
export {
|
|
74
74
|
collectDeclaredTestNames, reconcileTestNames, inventoryFromCSharpSources,
|
|
75
75
|
} from './matrix-check.mjs';
|
|
76
|
+
|
|
77
|
+
export {
|
|
78
|
+
COMMANDS, mutatingCommands, readOnlyCommands, resolveCommand,
|
|
79
|
+
describeCommand, describeAllCommands, checkUnattendedRequirements,
|
|
80
|
+
} from './commands.mjs';
|
package/src/harness/policy.mjs
CHANGED
|
@@ -162,6 +162,7 @@ export const EXCEPTION_CATEGORIES = Object.freeze([
|
|
|
162
162
|
'unscoped-freshness',
|
|
163
163
|
'documentation-only',
|
|
164
164
|
'tooling-gap',
|
|
165
|
+
'pre-harness-story',
|
|
165
166
|
]);
|
|
166
167
|
|
|
167
168
|
/** Default expiry (in days) per category. `null` means "no default bound". */
|
|
@@ -172,6 +173,11 @@ export const EXCEPTION_EXPIRY_DAYS = Object.freeze({
|
|
|
172
173
|
'unscoped-freshness': 1,
|
|
173
174
|
'documentation-only': null, // scoped to the work item
|
|
174
175
|
'tooling-gap': 14,
|
|
176
|
+
// No default bound: this records a PERMANENT historical fact (a story closed
|
|
177
|
+
// before the harness existed, whose gates were never recorded as evidence).
|
|
178
|
+
// The gap will never close on its own, so a time-bounded exception would only
|
|
179
|
+
// re-raise the same finding every N days without anything having changed.
|
|
180
|
+
'pre-harness-story': null,
|
|
175
181
|
});
|
|
176
182
|
|
|
177
183
|
/** Categories whose exception must carry a closure review note. */
|
package/src/harness/state.mjs
CHANGED
|
@@ -231,19 +231,37 @@ export function validateState(state, context = {}) {
|
|
|
231
231
|
// It is an ERROR, not a warning, because a `done` story with no evidence is
|
|
232
232
|
// indistinguishable from a story that was never verified — which is the
|
|
233
233
|
// condition the framework exists to prevent. Projects that closed stories
|
|
234
|
-
// before the harness existed
|
|
235
|
-
//
|
|
234
|
+
// before the harness existed resolve it with a scoped `pre-harness-story`
|
|
235
|
+
// exception naming the story's work item; silently tolerating it is what let
|
|
236
|
+
// the gap grow.
|
|
236
237
|
if (isPlainObject(state.epics) && Array.isArray(state.gateEvidence)) {
|
|
237
238
|
const evidenced = new Set(
|
|
238
239
|
state.gateEvidence
|
|
239
240
|
.map((e) => (isPlainObject(e) ? e.workItemId : null))
|
|
240
241
|
.filter((id) => typeof id === 'string' && id.length > 0),
|
|
241
242
|
);
|
|
243
|
+
// A scoped exception is the FIRST-CLASS escape for a permanent historical
|
|
244
|
+
// gap. `activeExceptions` is keyed on the ACTIVE work item, which is the
|
|
245
|
+
// current story — the wrong scope here, where we walk every completed story
|
|
246
|
+
// — so the scope match is done explicitly against each story's work-item id.
|
|
247
|
+
// Only a valid, categorised exception counts: an unknown category is not a
|
|
248
|
+
// loophole, it is a typo, and `validateGateException` rejects it separately.
|
|
249
|
+
const excepted = new Set();
|
|
250
|
+
if (Array.isArray(state.changeHistory)) {
|
|
251
|
+
for (const entry of state.changeHistory) {
|
|
252
|
+
if (!isPlainObject(entry) || entry.type !== 'gate-exception') continue;
|
|
253
|
+
if (!EXCEPTION_CATEGORIES.includes(entry.category)) continue;
|
|
254
|
+
const scope = Array.isArray(entry.scope) ? entry.scope : (entry.scope ? [entry.scope] : []);
|
|
255
|
+
for (const s of scope) excepted.add(String(s));
|
|
256
|
+
}
|
|
257
|
+
}
|
|
242
258
|
for (const [epicId, epic] of Object.entries(state.epics)) {
|
|
243
259
|
if (!isPlainObject(epic) || !isPlainObject(epic.stories)) continue;
|
|
244
260
|
for (const [storyId, status] of Object.entries(epic.stories)) {
|
|
245
261
|
if (status !== 'done') continue;
|
|
246
|
-
|
|
262
|
+
const workItemId = `${epicId}::${storyId}`;
|
|
263
|
+
if (evidenced.has(workItemId)) continue;
|
|
264
|
+
if (excepted.has(workItemId)) continue;
|
|
247
265
|
errors.push({
|
|
248
266
|
path: `epics.${epicId}.stories.${storyId}`,
|
|
249
267
|
message: `story "${storyId}" is marked done but has no evidence record for its work item `
|
|
@@ -583,16 +601,22 @@ export function migrateStateFile(statePath, { backup = true } = {}) {
|
|
|
583
601
|
const tmpPath = join(tmpDir, 'state.json');
|
|
584
602
|
try {
|
|
585
603
|
writeFileSync(tmpPath, JSON.stringify(state, null, 2) + '\n', 'utf-8');
|
|
586
|
-
|
|
587
|
-
|
|
588
|
-
|
|
589
|
-
//
|
|
604
|
+
// Validate BEFORE writing anything to the real tree. The backup is an
|
|
605
|
+
// artifact of a *successful* migration, so producing one and then failing
|
|
606
|
+
// would leave a write the caller never got and did not ask for — a failed
|
|
607
|
+
// `migrate` must leave the directory exactly as it found it. Checking first
|
|
608
|
+
// also means a bad migration cannot overwrite a previous good backup.
|
|
609
|
+
//
|
|
590
610
|
// This is a structural-only check: a freshly migrated state has no on-disk
|
|
591
611
|
// evidence to bind, so freshness is intentionally not evaluated here.
|
|
592
612
|
const check = validateState(state, { structuralOnly: true });
|
|
593
613
|
if (!check.valid) {
|
|
594
614
|
throw new StateError(`migrated state failed validation: ${check.errors.map((e) => `${e.path}: ${e.message}`).join('; ')}`);
|
|
595
615
|
}
|
|
616
|
+
if (backup) {
|
|
617
|
+
copyFileSync(statePath, `${statePath}.v1.bak`);
|
|
618
|
+
}
|
|
619
|
+
// A failed rename leaves the original in place.
|
|
596
620
|
renameSync(tmpPath, statePath);
|
|
597
621
|
} finally {
|
|
598
622
|
rmSync(tmpDir, { recursive: true, force: true });
|