cadet-agent 0.35.0 → 0.37.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -1
- package/src/cli.mjs +104 -2
- package/src/harness/index.mjs +4 -0
- package/src/harness/matrix-check.mjs +122 -0
- package/src/harness/policy.mjs +6 -0
- package/src/harness/state.mjs +100 -0
- package/src/harness/verification.mjs +4 -1
package/package.json
CHANGED
package/src/cli.mjs
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import { readFileSync, writeFileSync } from 'node:fs';
|
|
2
2
|
import { fileURLToPath } from 'node:url';
|
|
3
|
-
import { dirname, join } from 'node:path';
|
|
3
|
+
import { dirname, join, resolve } from 'node:path';
|
|
4
4
|
import { install, sync } from './install.mjs';
|
|
5
5
|
import {
|
|
6
6
|
validateState, migrateStateFile, readState, writeState, evaluateTransition, applyTransition,
|
|
@@ -9,6 +9,7 @@ import {
|
|
|
9
9
|
detectRepoRole, describeRepoRole, GATES, manualConfirmation,
|
|
10
10
|
parseTestInventory, parseStoryCriteria, compareCoverage, describeCoverageGaps,
|
|
11
11
|
createEvidence, newId, computeInputTreeHash, hashCriteria,
|
|
12
|
+
collectDeclaredTestNames, reconcileTestNames,
|
|
12
13
|
} from './harness/index.mjs';
|
|
13
14
|
|
|
14
15
|
const __filename = fileURLToPath(import.meta.url);
|
|
@@ -57,10 +58,15 @@ function showHelp() {
|
|
|
57
58
|
--gate Gate name (harness verify|confirm)
|
|
58
59
|
--command Command override (harness verify)
|
|
59
60
|
--files Comma-separated relevant files to bind evidence to (harness verify|confirm)
|
|
61
|
+
--commit Revision the gate attests, as a hex SHA (harness verify|confirm)
|
|
60
62
|
--reason Why automation was unavailable (harness confirm)
|
|
61
63
|
--expires-at ISO-8601 expiry bounding the confirmation (harness confirm)
|
|
62
64
|
--environment key=value,... describing what was verified (harness confirm)
|
|
63
65
|
--scope Comma-separated scope of the confirmation (harness confirm)
|
|
66
|
+
--story Story markdown declaring the acceptance criteria (harness verify-acs)
|
|
67
|
+
--report Test report to derive the inventory from (harness verify-acs|matrix-check)
|
|
68
|
+
--matrix TDD matrix markdown to check (harness matrix-check)
|
|
69
|
+
--inventory Newline-separated test names, when no report is available (harness matrix-check)
|
|
64
70
|
--agents-md keep|overwrite|merge for an existing AGENTS.md (init/sync)
|
|
65
71
|
--yes, -y Never prompt; keep existing files (non-interactive installs)
|
|
66
72
|
--help, -h Show this help
|
|
@@ -96,6 +102,11 @@ function parseArgs(argv) {
|
|
|
96
102
|
case '--files': opts.filesGiven = true; opts.files = (argv[++i] || '').split(',').map((s) => s.trim()).filter(Boolean); break;
|
|
97
103
|
case '--story': opts.story = argv[++i]; break;
|
|
98
104
|
case '--report': opts.report = argv[++i]; break;
|
|
105
|
+
// AR-1: the revision a gate record attests, so a gate-related fix claim
|
|
106
|
+
// can be traced to the commit that contains it.
|
|
107
|
+
case '--commit': opts.commitGiven = true; opts.commit = argv[++i]; break;
|
|
108
|
+
case '--matrix': opts.matrix = argv[++i]; break;
|
|
109
|
+
case '--inventory': opts.inventory = argv[++i]; break;
|
|
99
110
|
case '--write-coverage': opts.writeCoverage = true; break;
|
|
100
111
|
case '--strict-orphans': opts.strictOrphans = true; break;
|
|
101
112
|
case '--dry-run': opts.dryRun = true; break;
|
|
@@ -361,6 +372,7 @@ async function cmdHarness(opts) {
|
|
|
361
372
|
relevantFiles,
|
|
362
373
|
rootDir: opts.targetDir,
|
|
363
374
|
approvedBy: opts.approvedBy || 'user',
|
|
375
|
+
commit: opts.commit || null,
|
|
364
376
|
at,
|
|
365
377
|
});
|
|
366
378
|
|
|
@@ -526,6 +538,7 @@ async function cmdHarness(opts) {
|
|
|
526
538
|
phase: ledger.phase || 'implementation',
|
|
527
539
|
relevantFiles,
|
|
528
540
|
rootDir: opts.targetDir,
|
|
541
|
+
commit: opts.commit || null,
|
|
529
542
|
policy,
|
|
530
543
|
budgets: ledger.tracker,
|
|
531
544
|
artifactDir: join(runsDir(opts.targetDir), 'artifacts'),
|
|
@@ -764,6 +777,95 @@ async function cmdHarness(opts) {
|
|
|
764
777
|
return;
|
|
765
778
|
}
|
|
766
779
|
|
|
780
|
+
// AR-5. Reconcile a TDD matrix's DELIVERED test-name claims against a compiled
|
|
781
|
+
// inventory. Read-only: it reports, and never writes state, so it can be run at
|
|
782
|
+
// authoring time (before anything has been implemented) as well as in a gate.
|
|
783
|
+
if (sub === 'matrix-check') {
|
|
784
|
+
if (!opts.matrix) fail(opts, 'harness matrix-check requires --matrix <path-to-matrix.md>');
|
|
785
|
+
const matrixPath = resolve(opts.targetDir, opts.matrix);
|
|
786
|
+
let matrixText;
|
|
787
|
+
try {
|
|
788
|
+
matrixText = readFileSync(matrixPath, 'utf-8');
|
|
789
|
+
} catch (err) {
|
|
790
|
+
fail(opts, `cannot read matrix ${opts.matrix}: ${err.message}`, () => 2);
|
|
791
|
+
}
|
|
792
|
+
|
|
793
|
+
// The inventory may come from a run report (strongest: it proves the test
|
|
794
|
+
// RAN) or from C# sources (a name inventory only). Prefer a report.
|
|
795
|
+
let inventory = null;
|
|
796
|
+
let inventorySource = null;
|
|
797
|
+
if (opts.report) {
|
|
798
|
+
const reportPath = resolve(opts.targetDir, opts.report);
|
|
799
|
+
let reportText;
|
|
800
|
+
try {
|
|
801
|
+
reportText = readFileSync(reportPath, 'utf-8');
|
|
802
|
+
} catch (err) {
|
|
803
|
+
fail(opts, `cannot read report ${opts.report}: ${err.message}`, () => 2);
|
|
804
|
+
}
|
|
805
|
+
const parsed = parseTestInventory(reportText);
|
|
806
|
+
if (parsed && parsed.format !== 'unknown' && parsed.names.length > 0) {
|
|
807
|
+
inventory = new Set(parsed.names);
|
|
808
|
+
inventorySource = `${opts.report} (${parsed.format})`;
|
|
809
|
+
}
|
|
810
|
+
}
|
|
811
|
+
if (!inventory && opts.inventory) {
|
|
812
|
+
const invPath = resolve(opts.targetDir, opts.inventory);
|
|
813
|
+
let raw;
|
|
814
|
+
try {
|
|
815
|
+
raw = readFileSync(invPath, 'utf-8');
|
|
816
|
+
} catch (err) {
|
|
817
|
+
fail(opts, `cannot read inventory ${opts.inventory}: ${err.message}`, () => 2);
|
|
818
|
+
}
|
|
819
|
+
inventory = new Set(String(raw).split(/\r?\n/).map((s) => s.trim()).filter(Boolean));
|
|
820
|
+
inventorySource = opts.inventory;
|
|
821
|
+
}
|
|
822
|
+
|
|
823
|
+
const collected = collectDeclaredTestNames(matrixText);
|
|
824
|
+
if (!inventory) {
|
|
825
|
+
// Without an inventory this cannot prove anything, so it must not report
|
|
826
|
+
// success. Returning the collected names is still useful at authoring time:
|
|
827
|
+
// it shows what the matrix claims, and the caller can spot a name they know
|
|
828
|
+
// they never wrote.
|
|
829
|
+
const detail = {
|
|
830
|
+
ok: false,
|
|
831
|
+
code: 'inventory-unavailable',
|
|
832
|
+
matrix: opts.matrix,
|
|
833
|
+
claims: collected.claims,
|
|
834
|
+
intents: collected.intents,
|
|
835
|
+
};
|
|
836
|
+
if (opts.format === 'json') emit(opts, '', detail);
|
|
837
|
+
else {
|
|
838
|
+
console.error('❌ No inventory supplied, so no claim can be checked. Pass --report <test-results> (preferred, proves the test ran) or --inventory <names.txt>.');
|
|
839
|
+
console.error(` The matrix declares ${collected.claims.length} delivered claim(s) across ${collected.intents.length} undelivered intention(s).`);
|
|
840
|
+
}
|
|
841
|
+
process.exit(1);
|
|
842
|
+
}
|
|
843
|
+
|
|
844
|
+
const result = reconcileTestNames(collected, inventory);
|
|
845
|
+
const ok = result.missingFromInventory.length === 0;
|
|
846
|
+
const detail = {
|
|
847
|
+
ok,
|
|
848
|
+
matrix: opts.matrix,
|
|
849
|
+
inventory: inventorySource,
|
|
850
|
+
checked: result.checked,
|
|
851
|
+
intents: collected.intents.length,
|
|
852
|
+
missingFromInventory: result.missingFromInventory,
|
|
853
|
+
unmatchedIntents: result.unmatchedIntents,
|
|
854
|
+
};
|
|
855
|
+
if (opts.format === 'json') emit(opts, '', detail);
|
|
856
|
+
else if (ok) {
|
|
857
|
+
console.log(`✅ Every delivered test-name claim exists in the inventory (${result.checked} checked from ${inventorySource}).`);
|
|
858
|
+
if (result.unmatchedIntents.length > 0) {
|
|
859
|
+
console.log(` ℹ️ ${result.unmatchedIntents.length} undelivered intention(s) now exist and the row may be stale — consider marking it DELIVERED.`);
|
|
860
|
+
}
|
|
861
|
+
} else {
|
|
862
|
+
console.error(`❌ ${result.missingFromInventory.length} delivered test-name claim(s) do not exist in the inventory (${inventorySource}):`);
|
|
863
|
+
for (const n of result.missingFromInventory) console.error(` "${n}" — declared as delivered but absent. Attach it to the criterion it proves, or correct the name.`);
|
|
864
|
+
}
|
|
865
|
+
if (!ok) process.exit(1);
|
|
866
|
+
return;
|
|
867
|
+
}
|
|
868
|
+
|
|
767
869
|
if (sub === 'cleanup') {
|
|
768
870
|
const { deleted, kept } = cleanupRuns(opts.targetDir, policy, {
|
|
769
871
|
olderThanMs: Number.isFinite(opts.olderThanMs) ? opts.olderThanMs : null,
|
|
@@ -772,7 +874,7 @@ async function cmdHarness(opts) {
|
|
|
772
874
|
return;
|
|
773
875
|
}
|
|
774
876
|
|
|
775
|
-
fail(opts, `Unknown harness subcommand: ${sub || '(none)'}. Use record|confirm|verify|verify-acs|report|cleanup|capabilities.`);
|
|
877
|
+
fail(opts, `Unknown harness subcommand: ${sub || '(none)'}. Use record|confirm|verify|verify-acs|matrix-check|report|cleanup|capabilities.`);
|
|
776
878
|
}
|
|
777
879
|
|
|
778
880
|
export async function run(argv) {
|
package/src/harness/index.mjs
CHANGED
|
@@ -69,3 +69,7 @@ export {
|
|
|
69
69
|
normalizeTestName, parseTestInventory, parseStoryCriteria, parseStoryCriteriaText,
|
|
70
70
|
compareCoverage, describeCoverageGaps,
|
|
71
71
|
} from './verify-acs.mjs';
|
|
72
|
+
|
|
73
|
+
export {
|
|
74
|
+
collectDeclaredTestNames, reconcileTestNames, inventoryFromCSharpSources,
|
|
75
|
+
} from './matrix-check.mjs';
|
|
@@ -0,0 +1,122 @@
|
|
|
1
|
+
// AR-5 — mechanical reconciliation of a TDD matrix's test-name claims against a
|
|
2
|
+
// compiled test inventory.
|
|
3
|
+
//
|
|
4
|
+
// THE DEFECT THIS EXISTS TO REMOVE. A TDD matrix row names the tests that prove
|
|
5
|
+
// an acceptance criterion. Those rows are authored during architecture, BEFORE
|
|
6
|
+
// implementation, so a name can be an intention that changes (or never happens)
|
|
7
|
+
// while nothing re-checks the row. In one real project this produced the SAME
|
|
8
|
+
// phantom-test-name defect three times, and all three were found late — by the
|
|
9
|
+
// validation gate, two stories after the claim was written. A name that does not
|
|
10
|
+
// exist reads as proof and is not.
|
|
11
|
+
//
|
|
12
|
+
// THE TWO DIRECTIONS, AND WHY THE DISTINCTION MATTERS.
|
|
13
|
+
//
|
|
14
|
+
// DELIVERED rows carry a claim: "these tests exist and prove this criterion".
|
|
15
|
+
// A name here that is absent from the inventory is a DEFECT.
|
|
16
|
+
// undelivered rows carry an intention for planned work. A name here that is
|
|
17
|
+
// absent is EXPECTED and must NOT be reported.
|
|
18
|
+
//
|
|
19
|
+
// Collapsing the two produces false failures, and a false failure is how a real
|
|
20
|
+
// check gets switched off. This module keeps them separate and makes the
|
|
21
|
+
// separation the caller's explicit choice.
|
|
22
|
+
//
|
|
23
|
+
// Deliberately dependency-free and side-effect-free: it returns findings and
|
|
24
|
+
// never throws on content, so it can run at authoring time as well as in a gate.
|
|
25
|
+
|
|
26
|
+
/** Marker that separates a row's claims from its explanatory prose. */
|
|
27
|
+
const DELIVERED_MARKER = 'DELIVERED';
|
|
28
|
+
|
|
29
|
+
/**
|
|
30
|
+
* A test name is `Identifier_LikeThis` — at least one underscore, both sides
|
|
31
|
+
* identifier-shaped. This deliberately rejects prose symbols that appear in
|
|
32
|
+
* backticks (type names such as `ViewExtent`, single letters such as `u`), which
|
|
33
|
+
* is the specific false-positive class that made an earlier checker unusable.
|
|
34
|
+
*/
|
|
35
|
+
const TEST_NAME = /^[A-Za-z][A-Za-z0-9]*_[A-Za-z0-9_]+$/;
|
|
36
|
+
|
|
37
|
+
/** Extract every backticked token that looks like a test name. */
|
|
38
|
+
function backtickedTestNames(text) {
|
|
39
|
+
const out = [];
|
|
40
|
+
for (const m of String(text).matchAll(/`([^`]+)`/g)) {
|
|
41
|
+
const name = m[1].trim();
|
|
42
|
+
if (TEST_NAME.test(name)) out.push(name);
|
|
43
|
+
}
|
|
44
|
+
return out;
|
|
45
|
+
}
|
|
46
|
+
|
|
47
|
+
/** Split a markdown table row into its cells (leading/trailing pipes dropped). */
|
|
48
|
+
function cellsOf(line) {
|
|
49
|
+
const trimmed = line.trim();
|
|
50
|
+
if (!trimmed.startsWith('|')) return null;
|
|
51
|
+
const parts = trimmed.split('|');
|
|
52
|
+
// A row looks like `| a | b | c |`; the split yields ['', ' a ', ' b ', ' c ', ''].
|
|
53
|
+
return parts.slice(1, -1).map((c) => c.trim());
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
/**
|
|
57
|
+
* Collect test names from a TDD-matrix-style markdown document.
|
|
58
|
+
*
|
|
59
|
+
* Only the "declared tests" cell — the SECOND column — is read, and within a
|
|
60
|
+
* DELIVERED row only the text BEFORE the marker. Both restrictions exist because
|
|
61
|
+
* a matrix row's later columns discuss the design in prose and legitimately name
|
|
62
|
+
* types, symbols and fractions in backticks; treating those as claims is exactly
|
|
63
|
+
* the false-failure class this module was written to avoid.
|
|
64
|
+
*
|
|
65
|
+
* @returns {{claims: string[], intents: string[]}} deduplicated, source-ordered
|
|
66
|
+
*/
|
|
67
|
+
export function collectDeclaredTestNames(markdown) {
|
|
68
|
+
const claims = [];
|
|
69
|
+
const intents = [];
|
|
70
|
+
for (const line of String(markdown).split(/\r?\n/)) {
|
|
71
|
+
const cells = cellsOf(line);
|
|
72
|
+
if (!cells || cells.length < 2) continue;
|
|
73
|
+
const declaredCell = cells[1];
|
|
74
|
+
if (!declaredCell) continue;
|
|
75
|
+
// A separator row (`| --- | --- |`) has no test names and no marker.
|
|
76
|
+
const markerAt = declaredCell.indexOf(DELIVERED_MARKER);
|
|
77
|
+
const isDelivered = markerAt !== -1;
|
|
78
|
+
const scope = isDelivered ? declaredCell.slice(0, markerAt) : declaredCell;
|
|
79
|
+
const names = backtickedTestNames(scope);
|
|
80
|
+
(isDelivered ? claims : intents).push(...names);
|
|
81
|
+
}
|
|
82
|
+
return { claims: [...new Set(claims)], intents: [...new Set(intents)] };
|
|
83
|
+
}
|
|
84
|
+
|
|
85
|
+
/**
|
|
86
|
+
* Compare collected names against a compiled inventory.
|
|
87
|
+
*
|
|
88
|
+
* @param {{claims: string[], intents: string[]}} collected
|
|
89
|
+
* @param {Set<string>|string[]} inventory compiled test method names
|
|
90
|
+
* @returns {{missingFromInventory: string[], unmatchedIntents: string[], checked: number}}
|
|
91
|
+
*/
|
|
92
|
+
export function reconcileTestNames(collected, inventory) {
|
|
93
|
+
const have = inventory instanceof Set ? inventory : new Set(inventory || []);
|
|
94
|
+
const claims = Array.isArray(collected?.claims) ? collected.claims : [];
|
|
95
|
+
const intents = Array.isArray(collected?.intents) ? collected.intents : [];
|
|
96
|
+
return {
|
|
97
|
+
// Only DELIVERED claims can be defects. An intention is allowed to be absent.
|
|
98
|
+
missingFromInventory: claims.filter((n) => !have.has(n)),
|
|
99
|
+
// Reported separately and informationally: an intention that HAS landed is
|
|
100
|
+
// not a defect, but it usually means the row is stale and should be marked
|
|
101
|
+
// DELIVERED. Surfaced so the drift is visible rather than silent.
|
|
102
|
+
unmatchedIntents: intents.filter((n) => have.has(n)),
|
|
103
|
+
checked: claims.length,
|
|
104
|
+
};
|
|
105
|
+
}
|
|
106
|
+
|
|
107
|
+
/**
|
|
108
|
+
* Extract public test-method names from C# test sources.
|
|
109
|
+
*
|
|
110
|
+
* Used to build an inventory when no test report is available — for example at
|
|
111
|
+
* authoring time, before anything has run. This is a NAME inventory, not a
|
|
112
|
+
* pass/fail one: it proves a name exists, never that the test passes. Callers
|
|
113
|
+
* that need the stronger claim must use a run report.
|
|
114
|
+
*/
|
|
115
|
+
export function inventoryFromCSharpSources(sources) {
|
|
116
|
+
const names = new Set();
|
|
117
|
+
const pattern = /public\s+(?:async\s+)?(?:void|Task)\s+([A-Za-z_][A-Za-z0-9_]*)\s*\(/g;
|
|
118
|
+
for (const text of sources) {
|
|
119
|
+
for (const m of String(text).matchAll(pattern)) names.add(m[1]);
|
|
120
|
+
}
|
|
121
|
+
return names;
|
|
122
|
+
}
|
package/src/harness/policy.mjs
CHANGED
|
@@ -162,6 +162,7 @@ export const EXCEPTION_CATEGORIES = Object.freeze([
|
|
|
162
162
|
'unscoped-freshness',
|
|
163
163
|
'documentation-only',
|
|
164
164
|
'tooling-gap',
|
|
165
|
+
'pre-harness-story',
|
|
165
166
|
]);
|
|
166
167
|
|
|
167
168
|
/** Default expiry (in days) per category. `null` means "no default bound". */
|
|
@@ -172,6 +173,11 @@ export const EXCEPTION_EXPIRY_DAYS = Object.freeze({
|
|
|
172
173
|
'unscoped-freshness': 1,
|
|
173
174
|
'documentation-only': null, // scoped to the work item
|
|
174
175
|
'tooling-gap': 14,
|
|
176
|
+
// No default bound: this records a PERMANENT historical fact (a story closed
|
|
177
|
+
// before the harness existed, whose gates were never recorded as evidence).
|
|
178
|
+
// The gap will never close on its own, so a time-bounded exception would only
|
|
179
|
+
// re-raise the same finding every N days without anything having changed.
|
|
180
|
+
'pre-harness-story': null,
|
|
175
181
|
});
|
|
176
182
|
|
|
177
183
|
/** Categories whose exception must carry a closure review note. */
|
package/src/harness/state.mjs
CHANGED
|
@@ -212,6 +212,65 @@ export function validateState(state, context = {}) {
|
|
|
212
212
|
if (lt.to !== undefined && !PHASES.includes(lt.to)) errors.push({ path: 'lastTransition.to', message: `unknown phase "${lt.to}"` });
|
|
213
213
|
}
|
|
214
214
|
}
|
|
215
|
+
|
|
216
|
+
// AR-2. A story marked `done` must have SOME evidence record of its own.
|
|
217
|
+
//
|
|
218
|
+
// WHY THIS IS NEEDED. Every gate rule in this file is scoped to the ACTIVE
|
|
219
|
+
// work item, so `state validate` could report a document as fully valid while
|
|
220
|
+
// an already-completed story had no evidence whatsoever. In one real project
|
|
221
|
+
// eight `done` stories had zero records and validation said "valid, 0 errors,
|
|
222
|
+
// 0 warnings" — the gaps were invisible until they were looked for by hand.
|
|
223
|
+
//
|
|
224
|
+
// SCOPE, DELIBERATELY NARROW. This asserts COVERAGE, not gate completeness:
|
|
225
|
+
// it asks only "is there any evidence for this story at all?". Whether every
|
|
226
|
+
// required gate was satisfied for the right phase is already enforced at
|
|
227
|
+
// transition time, against the active work item, where the phase is known.
|
|
228
|
+
// Re-deciding that here would duplicate the transition matrix and risk the
|
|
229
|
+
// two disagreeing.
|
|
230
|
+
//
|
|
231
|
+
// It is an ERROR, not a warning, because a `done` story with no evidence is
|
|
232
|
+
// indistinguishable from a story that was never verified — which is the
|
|
233
|
+
// condition the framework exists to prevent. Projects that closed stories
|
|
234
|
+
// before the harness existed resolve it with a scoped `pre-harness-story`
|
|
235
|
+
// exception naming the story's work item; silently tolerating it is what let
|
|
236
|
+
// the gap grow.
|
|
237
|
+
if (isPlainObject(state.epics) && Array.isArray(state.gateEvidence)) {
|
|
238
|
+
const evidenced = new Set(
|
|
239
|
+
state.gateEvidence
|
|
240
|
+
.map((e) => (isPlainObject(e) ? e.workItemId : null))
|
|
241
|
+
.filter((id) => typeof id === 'string' && id.length > 0),
|
|
242
|
+
);
|
|
243
|
+
// A scoped exception is the FIRST-CLASS escape for a permanent historical
|
|
244
|
+
// gap. `activeExceptions` is keyed on the ACTIVE work item, which is the
|
|
245
|
+
// current story — the wrong scope here, where we walk every completed story
|
|
246
|
+
// — so the scope match is done explicitly against each story's work-item id.
|
|
247
|
+
// Only a valid, categorised exception counts: an unknown category is not a
|
|
248
|
+
// loophole, it is a typo, and `validateGateException` rejects it separately.
|
|
249
|
+
const excepted = new Set();
|
|
250
|
+
if (Array.isArray(state.changeHistory)) {
|
|
251
|
+
for (const entry of state.changeHistory) {
|
|
252
|
+
if (!isPlainObject(entry) || entry.type !== 'gate-exception') continue;
|
|
253
|
+
if (!EXCEPTION_CATEGORIES.includes(entry.category)) continue;
|
|
254
|
+
const scope = Array.isArray(entry.scope) ? entry.scope : (entry.scope ? [entry.scope] : []);
|
|
255
|
+
for (const s of scope) excepted.add(String(s));
|
|
256
|
+
}
|
|
257
|
+
}
|
|
258
|
+
for (const [epicId, epic] of Object.entries(state.epics)) {
|
|
259
|
+
if (!isPlainObject(epic) || !isPlainObject(epic.stories)) continue;
|
|
260
|
+
for (const [storyId, status] of Object.entries(epic.stories)) {
|
|
261
|
+
if (status !== 'done') continue;
|
|
262
|
+
const workItemId = `${epicId}::${storyId}`;
|
|
263
|
+
if (evidenced.has(workItemId)) continue;
|
|
264
|
+
if (excepted.has(workItemId)) continue;
|
|
265
|
+
errors.push({
|
|
266
|
+
path: `epics.${epicId}.stories.${storyId}`,
|
|
267
|
+
message: `story "${storyId}" is marked done but has no evidence record for its work item `
|
|
268
|
+
+ `"${epicId}::${storyId}". A completed story must be backed by at least one evidence `
|
|
269
|
+
+ 'record; otherwise it is indistinguishable from one that was never verified.',
|
|
270
|
+
});
|
|
271
|
+
}
|
|
272
|
+
}
|
|
273
|
+
}
|
|
215
274
|
} else if (state.gateEvidence !== undefined) {
|
|
216
275
|
warnings.push({ path: 'gateEvidence', message: 'gateEvidence on a v1 state is ignored until migration' });
|
|
217
276
|
}
|
|
@@ -275,6 +334,15 @@ function validateEvidenceShape(ev, strict = null) {
|
|
|
275
334
|
if (ev.result !== undefined && ev.result !== null && typeof ev.result !== 'string') {
|
|
276
335
|
errors.push({ path: 'result', message: 'result must be a string or null' });
|
|
277
336
|
}
|
|
337
|
+
// AR-1. `commit` is optional (a v2-shaped record omits it) but must be a real
|
|
338
|
+
// revision identifier when present — a branch or tag name would read as a
|
|
339
|
+
// citation while being uncheckable later, which is worse than none.
|
|
340
|
+
if (ev.commit !== undefined && ev.commit !== null && !/^[0-9a-fA-F]{4,40}$/.test(String(ev.commit))) {
|
|
341
|
+
errors.push({
|
|
342
|
+
path: 'commit',
|
|
343
|
+
message: 'commit must be a 4-40 character hex revision identifier, or null',
|
|
344
|
+
});
|
|
345
|
+
}
|
|
278
346
|
if (ev.createdAt !== undefined && ev.createdAt !== null && Number.isNaN(Date.parse(ev.createdAt))) {
|
|
279
347
|
errors.push({ path: 'createdAt', message: 'createdAt must be an ISO-8601 date-time' });
|
|
280
348
|
}
|
|
@@ -552,6 +620,30 @@ export function migrateStateFile(statePath, { backup = true } = {}) {
|
|
|
552
620
|
|
|
553
621
|
// ── Evidence ────────────────────────────────────────────────────────────────
|
|
554
622
|
|
|
623
|
+
/**
|
|
624
|
+
* AR-1. Normalize and validate a commit citation.
|
|
625
|
+
*
|
|
626
|
+
* Accepts a full SHA or an abbreviated one (git's default short form is 7, but
|
|
627
|
+
* 4–40 hex characters are all unambiguous enough to store). A symbolic name such
|
|
628
|
+
* as a branch or tag is REJECTED: those move, so a record naming one cannot be
|
|
629
|
+
* checked later, which defeats the purpose of citing a revision at all.
|
|
630
|
+
*
|
|
631
|
+
* Returns null for an absent value so a v2-shaped record is unchanged.
|
|
632
|
+
*/
|
|
633
|
+
export function normalizeCommit(commit) {
|
|
634
|
+
if (commit === null || commit === undefined) return null;
|
|
635
|
+
const value = String(commit).trim();
|
|
636
|
+
if (value === '') return null;
|
|
637
|
+
if (!/^[0-9a-fA-F]{4,40}$/.test(value)) {
|
|
638
|
+
throw new StateError(
|
|
639
|
+
`commit must be a 4-40 character hex revision identifier, but was "${value}". `
|
|
640
|
+
+ 'A branch or tag name is not accepted: it moves, so the citation could not be '
|
|
641
|
+
+ 'checked later. Pass an abbreviated or full SHA.',
|
|
642
|
+
);
|
|
643
|
+
}
|
|
644
|
+
return value.toLowerCase();
|
|
645
|
+
}
|
|
646
|
+
|
|
555
647
|
/** Build an evidence record. `id` defaults to a fresh UUIDv4. */
|
|
556
648
|
export function createEvidence({
|
|
557
649
|
evidenceId,
|
|
@@ -569,12 +661,19 @@ export function createEvidence({
|
|
|
569
661
|
criteriaHash = null,
|
|
570
662
|
relevantFiles = [],
|
|
571
663
|
toolVersion = null,
|
|
664
|
+
commit = null,
|
|
572
665
|
createdAt = new Date(),
|
|
573
666
|
expiresAt = null,
|
|
574
667
|
freshnessPolicy = null,
|
|
575
668
|
source = 'automated',
|
|
576
669
|
id,
|
|
577
670
|
}) {
|
|
671
|
+
// AR-1. A gate record must be able to name the revision it attests, or a
|
|
672
|
+
// "gate-related fix claim" cannot be traced to the code it claims to cover.
|
|
673
|
+
// Validated rather than trusted: this value is persisted into state.json and
|
|
674
|
+
// read back by the Reviewer, so a malformed one would make a claim look
|
|
675
|
+
// verified while naming nothing.
|
|
676
|
+
const normalizedCommit = normalizeCommit(commit);
|
|
578
677
|
return {
|
|
579
678
|
evidenceId: evidenceId || id || undefined,
|
|
580
679
|
workItemId,
|
|
@@ -591,6 +690,7 @@ export function createEvidence({
|
|
|
591
690
|
criteriaHash: criteriaHash || hashCriteria([]),
|
|
592
691
|
relevantFiles,
|
|
593
692
|
toolVersion,
|
|
693
|
+
commit: normalizedCommit,
|
|
594
694
|
createdAt: timestamp(createdAt),
|
|
595
695
|
expiresAt: expiresAt ? timestamp(expiresAt) : null,
|
|
596
696
|
freshnessPolicy,
|
|
@@ -252,6 +252,7 @@ export async function runVerificationLoop({
|
|
|
252
252
|
relevantFiles = [],
|
|
253
253
|
criteria = [],
|
|
254
254
|
rootDir = process.cwd(),
|
|
255
|
+
commit = null,
|
|
255
256
|
policy,
|
|
256
257
|
budgets,
|
|
257
258
|
runCommandImpl = runCommand,
|
|
@@ -327,6 +328,7 @@ export async function runVerificationLoop({
|
|
|
327
328
|
inputTreeHash,
|
|
328
329
|
criteriaHash,
|
|
329
330
|
relevantFiles,
|
|
331
|
+
commit,
|
|
330
332
|
createdAt: startedAt,
|
|
331
333
|
source: 'automated',
|
|
332
334
|
});
|
|
@@ -461,7 +463,7 @@ function finalize({ status, attempts, tracker, inputTreeHash, criteriaHash, stop
|
|
|
461
463
|
export function manualConfirmation({
|
|
462
464
|
gate, workItemId, phase, projectPath, editorVersion, scope, acceptanceCriterionId = null,
|
|
463
465
|
relevantFiles = [], criteria = [], rootDir = process.cwd(), approvedBy = 'user', at = new Date(),
|
|
464
|
-
reason = null, expiresAt = null, environment = null, expiresInMs = null,
|
|
466
|
+
reason = null, expiresAt = null, environment = null, expiresInMs = null, commit = null,
|
|
465
467
|
} = {}) {
|
|
466
468
|
const inputTreeHash = computeInputTreeHash(rootDir, relevantFiles);
|
|
467
469
|
// v3 quality fields. `scope` is declared both as the free-text `result` line
|
|
@@ -504,6 +506,7 @@ export function manualConfirmation({
|
|
|
504
506
|
relevantFiles,
|
|
505
507
|
createdAt: at,
|
|
506
508
|
expiresAt: expiry,
|
|
509
|
+
commit,
|
|
507
510
|
source: 'manual-confirmation',
|
|
508
511
|
}),
|
|
509
512
|
// Present only when supplied, so a v2-shaped record is unchanged when the
|