shapeup-sdlc 3.4.0 → 3.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,296 @@
1
+ #!/usr/bin/env node
2
+ // probe requirements — the way back: a pitch clause, the criterion that graded it, the verdict.
3
+ //
4
+ // WHY THIS IS A QUERY AND NOT A SENTENCE, which is the same reason `probe owner` is one. The
5
+ // requirement matrix is read at GATE L4 and cited by GATE H's census, and both are places where a
6
+ // narrated figure is indistinguishable from a measured one. Measured on the first verdict any run
7
+ // of the spine produced: 85 of 97 criterion rows carried a `traces_to` anchor in the WorkResult and
8
+ // 0 of 97 survived into the projection ingest wrote — so a matrix assembled from memory would have
9
+ // been assembled from an edge that no longer existed on disk. Every figure below is derived from
10
+ // files: the committed registry, the board's `covers:` clauses, the verdict ledger, and the EVAL
11
+ // result's own T0 citations. Nothing is passed in and nothing is written.
12
+ //
13
+ // THE JOIN, and which half is authoritative. A requirement has EVIDENCE when an acceptance
14
+ // criterion covers it (`(covers: REQ-…)` on the AC line — the planning-time edge, reviewed at L1b)
15
+ // AND a criterion that names it passed. `traces_to[]` is the navigation path from the judge's
16
+ // criterion back to the requirement, exactly what the schema already calls it: an anchor, never a
17
+ // grading input. A criterion whose `traces_to` names a REQ that no AC covers is printed as an
18
+ // INCONSISTENCY row and counted as nothing — folding it in would derive one L4 line from two
19
+ // unreconciled sources, which is the failure `probe owner` exists to prevent.
20
+ //
21
+ // IT PROJECTS ONE RUN. `order_id`, round and attempt all repeat across runs of one feature; the
22
+ // run key is the only thing that separates them, so a projection that ignored it would silently mix
23
+ // two runs of one slug. Rows written before the key reached the ledger carry no `run_id`; they are
24
+ // reported as unknown rather than folded into the run being projected.
25
+ //
26
+ // AN EMPTY JOIN READS AS CLEARLY AS A FULL ONE. A tree with no registry, a board with no `covers:`
27
+ // and a run with no verdict are all legitimate states, and each answers "no evidence" rather than
28
+ // failing — that answer is the point of the query, not an error in it.
29
+ //
30
+ // Usage:
31
+ // node "${CLAUDE_PLUGIN_ROOT}/kernel/harness.mjs" probe requirements --slug <slug> [--run-id <id>] [--format json|table] [--cwd <dir>]
32
+ //
33
+ // Exit code: 0 = answered (an empty projection is an answer), 2 = bad argv.
34
+
35
+ import { existsSync, readFileSync, readdirSync } from "node:fs";
36
+ import { resolve, join } from "node:path";
37
+ import { runArgs, isMain } from "../lib/argv.mjs";
38
+ import { requirements as requirementsFile, evaluationDir, resultsDir, readRunId } from "../lib/paths.mjs";
39
+ import { parseRequirements, coveredReqIds } from "../verify/trace.mjs";
40
+ import { readBoard } from "../compile.mjs";
41
+ import { reqId } from "../lib/contract.mjs";
42
+
43
+ /**
44
+ * Read a file, tolerating absence — every input to this projection is optional by design.
45
+ * @param {string} p - Absolute path.
46
+ * @returns {(string|null)} The contents, or null when the file is missing or unreadable.
47
+ */
48
+ function readIf(p) {
49
+ try { return existsSync(p) ? readFileSync(p, "utf8") : null; } catch { return null; }
50
+ }
51
+
52
+ /**
53
+ * The acceptance criteria on the LOCAL board that cover each requirement.
54
+ *
55
+ * The planning-time half of the join, and the authoritative one: a `covers:` clause is written when
56
+ * the plan is still cheap to change and is reviewed at L1b, whereas `traces_to` is written by the
57
+ * judge after the fact.
58
+ *
59
+ * NOT A SECOND COVERS-CLOSURE. Whether a requirement is covered is still decided by
60
+ * `coveredReqIds` in the oracle — there is one implementation of that question and this is not it.
61
+ * This walk exists only to keep the AC the closure discards, so the matrix can print WHICH
62
+ * criterion covers the clause instead of only that one does.
63
+ *
64
+ * @param {Array<object>} board - Task entries from `readBoard` (the only parser that carries
65
+ * `acceptance_criteria`; the scheduling view does not).
66
+ * @returns {Map<string, Array<{task_id:string, ac:string}>>} REQ-id → the ACs naming it, in board
67
+ * order. A requirement no AC covers is simply absent.
68
+ */
69
+ export function coveringAcs(board) {
70
+ const out = new Map();
71
+ for (const task of board || []) {
72
+ for (const ac of task.acceptance_criteria || []) {
73
+ if (typeof ac !== "object" || !Array.isArray(ac.covers)) continue;
74
+ for (const raw of ac.covers) {
75
+ const id = reqId(raw).toUpperCase();
76
+ if (!out.has(id)) out.set(id, []);
77
+ out.get(id).push({ task_id: task.id, ac: ac.text });
78
+ }
79
+ }
80
+ }
81
+ return out;
82
+ }
83
+
84
+ /**
85
+ * Every criterion row ingest projected, tagged with the EVAL target its ledger belongs to.
86
+ *
87
+ * The target is read off the FILE NAME rather than the row: `.verdicts-<target>.jsonl` is written
88
+ * per order, so the file is what says which dispatch produced the rows, and the same file is what
89
+ * locates the WorkResult carrying that dispatch's T0 citations.
90
+ *
91
+ * @param {string} cwd - Project root.
92
+ * @param {string} slug - Feature slug.
93
+ * @returns {Array<object>} One entry per ledger line, each the stored row plus `target`. Malformed
94
+ * lines are skipped; a missing evaluation directory yields [].
95
+ */
96
+ export function readVerdictRows(cwd, slug) {
97
+ const dir = evaluationDir(cwd, slug);
98
+ let files = [];
99
+ try { files = readdirSync(dir).filter((f) => /^\.verdicts-.+\.jsonl$/.test(f)).sort(); } catch { return []; }
100
+ const rows = [];
101
+ for (const f of files) {
102
+ const target = f.replace(/^\.verdicts-/, "").replace(/\.jsonl$/, "");
103
+ for (const line of (readIf(join(dir, f)) || "").split(/\r?\n/)) {
104
+ if (!line.trim()) continue;
105
+ try { rows.push({ ...JSON.parse(line), target }); } catch { /* a truncated line is not a verdict */ }
106
+ }
107
+ }
108
+ return rows;
109
+ }
110
+
111
+ /**
112
+ * The T0 artifacts an EVAL dispatch re-hashed, read from the WorkResult it wrote.
113
+ *
114
+ * The verdict ledger records criteria, not citations, so the machine fact a generator cannot
115
+ * fabricate lives one file over — in `results/<target>.json`. Read here rather than recomputed: the
116
+ * evaluator's own citation is what the round was accepted on.
117
+ *
118
+ * @param {string} cwd - Project root.
119
+ * @param {string} slug - Feature slug.
120
+ * @param {string} target - The EVAL order suffix (`evaluate-r1`), from the ledger's file name.
121
+ * @returns {Array<{scope_id:(string|null), path:(string|null), sha256:(string|null)}>} The cited
122
+ * artifacts; [] when the result is absent, unreadable or cites none.
123
+ */
124
+ export function t0Citations(cwd, slug, target) {
125
+ const body = readIf(join(resultsDir(cwd, slug), `${target}.json`));
126
+ if (!body) return [];
127
+ try {
128
+ const cites = JSON.parse(body)?.verdict?.t0_citations;
129
+ if (!Array.isArray(cites)) return [];
130
+ return cites.map((c) => ({ scope_id: c?.scope_id ?? null, path: c?.path ?? null, sha256: c?.sha256 ?? null }));
131
+ } catch { return []; }
132
+ }
133
+
134
+ /**
135
+ * Project one run's requirement matrix from the artifacts on disk.
136
+ *
137
+ * @param {{cwd:string, slug:string, runId?:(string|null)}} opts - Project root, feature slug, and
138
+ * the run to project; omitted, it is resolved from the run's own receipt.
139
+ * @returns {{slug:string, run_id:(string|null), registry:boolean, totals:object,
140
+ * rows:Array<object>, inconsistencies:Array<object>, ledger:object}} `rows` is one entry per
141
+ * registered clause — its source, status, covering ACs, the criteria that named it and their
142
+ * verdicts, and each criterion's T0 citations — with `evidence` one of `PASS`, `no evidence` or
143
+ * `cut`. `inconsistencies` holds criteria tracing to a REQ no AC covers. `ledger` says how many
144
+ * rows were read, projected, skipped as another run's, and left unkeyed. No registry ⇒ `rows` is
145
+ * empty and `registry` is false: an empty projection, not an error.
146
+ */
147
+ export function projectRequirements({ cwd, slug, runId = undefined }) {
148
+ const regText = readIf(requirementsFile(cwd, slug));
149
+ const clauses = regText === null ? [] : parseRequirements(regText);
150
+ const board = readBoard(cwd, slug);
151
+ // `covered` decides; `acs` only says which criterion did the covering.
152
+ const covered = coveredReqIds(board);
153
+ const acs = coveringAcs(board);
154
+ const run = runId === undefined ? readRunId(cwd, slug) : runId;
155
+
156
+ const all = readVerdictRows(cwd, slug);
157
+ const ledger = { rows_read: all.length, rows_projected: 0, rows_other_run: 0, rows_unknown_run: 0 };
158
+ const mine = [];
159
+ for (const r of all) {
160
+ if (r.run_id === undefined || r.run_id === null || r.run_id === "") { ledger.rows_unknown_run++; continue; }
161
+ if (run !== null && r.run_id === run) { ledger.rows_projected++; mine.push(r); continue; }
162
+ ledger.rows_other_run++;
163
+ }
164
+
165
+ // The criteria that name each requirement, and the T0 artifacts the round that graded them cited.
166
+ const t0Cache = new Map();
167
+ /**
168
+ * The cited T0 artifacts for one EVAL target, read once per target.
169
+ * @param {string} target - The EVAL order suffix.
170
+ * @returns {Array<object>} The citation rows.
171
+ */
172
+ const t0For = (target) => {
173
+ if (!t0Cache.has(target)) t0Cache.set(target, t0Citations(cwd, slug, target));
174
+ return t0Cache.get(target);
175
+ };
176
+
177
+ const byReq = new Map();
178
+ const inconsistencies = [];
179
+ for (const r of mine) {
180
+ const anchors = Array.isArray(r.traces_to) ? r.traces_to : [];
181
+ for (const raw of anchors) {
182
+ const id = reqId(raw).toUpperCase();
183
+ const entry = {
184
+ criterion: r.criterion ?? "", dimension: r.dimension ?? "", verdict: r.verdict ?? "",
185
+ confidence: r.confidence ?? null, evidence: r.evidence ?? "", target: r.target,
186
+ run: r.run ?? null, run_id: r.run_id ?? null, t0: t0For(r.target),
187
+ };
188
+ // THE AUTHORITATIVE HALF DECIDES. An anchor pointing at a requirement no acceptance criterion
189
+ // covers is a claim the plan never made; it is printed so somebody can reconcile it, and
190
+ // counted as nothing.
191
+ if (!covered.has(id)) { inconsistencies.push({ requirement: id, ...entry, why: "no acceptance criterion covers this requirement — the anchor resolves to nothing the plan claimed" }); continue; }
192
+ if (!byReq.has(id)) byReq.set(id, []);
193
+ byReq.get(id).push(entry);
194
+ }
195
+ }
196
+
197
+ const rows = clauses.map((c) => {
198
+ const id = c.id.toUpperCase();
199
+ const covering = acs.get(id) || [];
200
+ const criteria = byReq.get(id) || [];
201
+ const cut = c.status !== "covered";
202
+ const passed = criteria.some((x) => String(x.verdict).toUpperCase() === "PASS");
203
+ return {
204
+ id: c.id,
205
+ source: c.source || "",
206
+ clause: c.clause || "",
207
+ status: c.status,
208
+ covering_acs: covering,
209
+ criteria,
210
+ t0: [...new Set(criteria.flatMap((x) => x.t0.map((t) => t.sha256).filter(Boolean)))],
211
+ evidence: cut ? "cut" : passed ? "PASS" : "no evidence",
212
+ };
213
+ });
214
+
215
+ const totals = {
216
+ total: rows.length,
217
+ pass: rows.filter((r) => r.evidence === "PASS").length,
218
+ cut: rows.filter((r) => r.evidence === "cut").length,
219
+ no_evidence: rows.filter((r) => r.evidence === "no evidence").length,
220
+ inconsistencies: inconsistencies.length,
221
+ };
222
+ return { slug, run_id: run, registry: regText !== null, totals, rows, inconsistencies, ledger };
223
+ }
224
+
225
+ /**
226
+ * The one-line summary GATE L4 prints and GATE H's census cites.
227
+ *
228
+ * @param {object} r - A report from {@link projectRequirements}.
229
+ * @returns {string} `15/17 PASS · 1 CUT (PO) · 1 no evidence (REQ-12 ← R12)`, or the empty-run
230
+ * phrasing when there is no registry to project.
231
+ */
232
+ export function summaryLine(r) {
233
+ if (!r.registry) return "n/a (no registry)";
234
+ if (!r.totals.total) return "0 requirements registered";
235
+ const gaps = r.rows.filter((x) => x.evidence === "no evidence");
236
+ const named = gaps.slice(0, 3).map((x) => (x.source ? `${x.id} ← ${x.source}` : x.id)).join(", ");
237
+ const parts = [`${r.totals.pass}/${r.totals.total} PASS`];
238
+ if (r.totals.cut) parts.push(`${r.totals.cut} CUT (PO)`);
239
+ if (r.totals.no_evidence) parts.push(`${r.totals.no_evidence} no evidence (${named}${gaps.length > 3 ? ", …" : ""})`);
240
+ if (r.totals.inconsistencies) parts.push(`${r.totals.inconsistencies} inconsistency ${r.totals.inconsistencies === 1 ? "row" : "rows"}`);
241
+ return parts.join(" · ");
242
+ }
243
+
244
+ /**
245
+ * Render the matrix as a fixed-width table for a human reading a gate block.
246
+ * @param {object} r - A report from {@link projectRequirements}.
247
+ * @returns {string} The header, one row per requirement, the summary line and any inconsistencies.
248
+ */
249
+ export function renderTable(r) {
250
+ const out = [];
251
+ const head = ["REQ", "source", "status", "covering AC", "criterion", "verdict", "T0"];
252
+ const rows = r.rows.map((x) => [
253
+ x.id, x.source || "—", x.evidence,
254
+ x.covering_acs.length ? `${x.covering_acs[0].task_id}: ${x.covering_acs[0].ac.slice(0, 40)}${x.covering_acs.length > 1 ? ` (+${x.covering_acs.length - 1})` : ""}` : "—",
255
+ x.criteria.length ? `${x.criteria[0].criterion.slice(0, 40)}${x.criteria.length > 1 ? ` (+${x.criteria.length - 1})` : ""}` : "—",
256
+ x.criteria.length ? x.criteria.map((c) => c.verdict).join(",") : "—",
257
+ x.t0.length ? x.t0.map((h) => String(h).slice(0, 12)).join(",") : "—",
258
+ ]);
259
+ if (rows.length) {
260
+ const w = head.map((h, i) => Math.max(h.length, ...rows.map((row) => String(row[i]).length)));
261
+ const line = (row) => row.map((c, i) => String(c).padEnd(w[i])).join(" ").trimEnd();
262
+ out.push(line(head), line(w.map((n) => "-".repeat(n))), ...rows.map(line), "");
263
+ }
264
+ out.push(`Requirements: ${summaryLine(r)}`);
265
+ out.push(`run_id: ${r.run_id ?? "unknown"} · ledger rows read ${r.ledger.rows_read}, projected ${r.ledger.rows_projected}, other run ${r.ledger.rows_other_run}, unkeyed (run_id: unknown) ${r.ledger.rows_unknown_run}`);
266
+ for (const i of r.inconsistencies) {
267
+ out.push(` ⚠ ${i.requirement}: "${String(i.criterion).slice(0, 60)}" ${i.verdict} — ${i.why}`);
268
+ }
269
+ return out.join("\n");
270
+ }
271
+
272
+ export const ARGV_SPEC = {
273
+ usage: "harness.mjs probe requirements --slug <slug> [--run-id <id>] [--format json|table] [--cwd <dir>]",
274
+ _: { arity: 0, max: 0, name: "(no positional operands)" },
275
+ slug: { type: "str", required: true },
276
+ "run-id": { type: "str" },
277
+ format: { type: "enum", values: ["json", "table"], default: "json" },
278
+ cwd: { type: "path" },
279
+ };
280
+
281
+ /**
282
+ * Answer "which requirement has evidence, and from which criterion" from the artifacts on disk.
283
+ *
284
+ * @param {string[]} rawArgv - The subcommand's own arguments (harness.mjs strips the verb words).
285
+ * @returns {Promise<void>} Exits 0 with the projection — an absent run, an absent registry and an
286
+ * absent verdict are all answers, never errors.
287
+ */
288
+ export async function cli(rawArgv) {
289
+ const args = runArgs(ARGV_SPEC, rawArgv);
290
+ const cwd = resolve(args.cwd || process.cwd());
291
+ const report = projectRequirements({ cwd, slug: args.slug, runId: args.runId ?? undefined });
292
+ console.log(args.format === "table" ? renderTable(report) : JSON.stringify(report, null, 2));
293
+ process.exit(0);
294
+ }
295
+
296
+ if (isMain(import.meta.url)) cli(process.argv.slice(2));
@@ -63,7 +63,7 @@ import { splitFrontmatter } from "../lib/contract.mjs";
63
63
  import { globToRegExp } from "../verify/spec.mjs";
64
64
  import {
65
65
  intake, harnessRun, wiringMap, projectProfile, scopesDir, resultsDir, ordersDir,
66
- orientDir, activeOrder, usecasesDir, breadboard, receipt, readReceipt,
66
+ orientDir, activeOrder, usecasesDir, breadboard, receipt, readReceipt, requirements,
67
67
  } from "../lib/paths.mjs";
68
68
  import { evalVerdict } from "./eval.mjs";
69
69
 
@@ -394,6 +394,12 @@ export function deriveResumeState(cwd, slug) {
394
394
  orient_dir: `.shapeup/${slug}/orient/`,
395
395
  has_orient_artifacts: hasOrientArtifacts(cwd, slug),
396
396
  has_spec_tree: hasSpecTree(cwd, slug, hr.spec_folder || null),
397
+ // A PLAIN FACT, DELIBERATELY NOT A PHASE. The requirements registry is dispatched once, before
398
+ // ANALYZE, and the orchestrator guards that one dispatch on this boolean. It is NOT an entry in
399
+ // PHASE_ARTIFACT, and adding it there would be a migration hazard rather than a tidier shape:
400
+ // that map is also `nextPhase()`'s ordered list, so every run recorded before the registry
401
+ // existed would fast-forward to the registry instead of to `build` on its next relaunch.
402
+ has_requirements: existsSync(requirements(cwd, slug)),
397
403
  has_wiring_map: existsSync(wiringMap(cwd, slug)),
398
404
  project_profile_path: projectProfile(cwd, slug),
399
405
  has_project_profile: existsSync(projectProfile(cwd, slug)),
@@ -32,7 +32,7 @@ import {
32
32
  localRoot, receipt as receiptPath, ordersDir, resultsDir, verdictsDir, trials as trialsPath,
33
33
  gates as gatesPath, scopesDir, usecasesDir, requirements as requirementsPath, wiringMap as wiringMapPath,
34
34
  } from "../lib/paths.mjs";
35
- import { readAllContracts, readContract, ucId, SCOPE_CONTRACT, WIRING_MAP } from "../lib/contract.mjs";
35
+ import { readAllContracts, readContract, ucId, reqId, SCOPE_CONTRACT, WIRING_MAP } from "../lib/contract.mjs";
36
36
  import { runIdFromReceipt } from "../lib/paths.mjs";
37
37
 
38
38
  /** The graph's home — one file per feature, beside the run trace it projects. */
@@ -263,7 +263,10 @@ export function project(cwd, slug) {
263
263
  // UseCase nodes it was built from and `--trace` stopped there instead of reaching the
264
264
  // objective. The contract now names both, so both are projectable.
265
265
  for (const uc of contract.use_cases || []) edge(id, "IMPLEMENTS", `uc:${slug}:${ucId(uc)}`);
266
- for (const req of contract.covers || []) edge(id, "COVERS", `req:${slug}:${req}`);
266
+ // Normalised to the registry's key space on the way in, the same mapping spec-lint applies:
267
+ // a contract citing the pitch's `R<n>` and a Requirement node keyed `REQ-<n>` are one node,
268
+ // and an edge drawn to the other spelling is an edge to a node the graph does not hold.
269
+ for (const req of contract.covers || []) edge(id, "COVERS", `req:${slug}:${reqId(req)}`);
267
270
  for (const dep of contract.depends_on || []) edge(id, "DEPENDS_ON", `scope:${slug}:${dep}`);
268
271
  }
269
272
  }
@@ -8,7 +8,9 @@
8
8
  // task_results[] → tick AC boxes, flip task frontmatter status, append Execution Log,
9
9
  // update tasks/_index.md row, propagate unblocks (old P3.1–P3.6)
10
10
  // discoveries[] → append to .shapeup/<slug>/discovery/ledger.md (old P3.7 / QA H.3)
11
- // verdict.criteria[] → append evaluation/.verdicts-<target>.jsonl (old evaluator B.0)
11
+ // verdict.criteria[] → append evaluation/.verdicts-<target>.jsonl (old evaluator B.0), every
12
+ // row keyed by run_id and carrying the judge's traces_to[] anchor back to
13
+ // the requirement — see step 4 for why neither may be dropped here
12
14
  // verdict.refuted[] → un-tick refuted AC boxes + set eval_verdict frontmatter (old B.2/B.2b)
13
15
  // (the leg itself) → append one leg-completion row to .shapeup/<slug>/legs.jsonl
14
16
  //
@@ -286,21 +288,32 @@ function applyResultLocked(result, { cwd, slug }) {
286
288
  }
287
289
 
288
290
  // 4. Verdict bookkeeping (old evaluator B.0/B.2/B.2b) — judge returns data, ingest writes.
291
+ //
292
+ // THE PROJECTION KEEPS THE TWO KEYS A LATER READER CANNOT RE-DERIVE. Measured on the first real
293
+ // verdict of a spined run: 85 of 97 criteria carried `traces_to` in the WorkResult and 0 of 97
294
+ // survived into this file, so the anchor from a criterion back to the requirement it grades was
295
+ // written by the judge and discarded one step later. And the row carried only the monotonic `run`
296
+ // counter, which restarts per ledger file and repeats across runs of one slug — `run_id` is the
297
+ // only key that separates them, so a projection over this file silently mixed two runs without
298
+ // it. A WorkResult carries no run key; it is read off the run's own receipt, the same derivation
299
+ // every other writer here uses.
289
300
  if (result.verdict) {
290
301
  const evalDir = join(local, "evaluation");
291
302
  mkdirSync(evalDir, { recursive: true });
292
303
  if (result.verdict.criteria?.length) {
293
304
  const target = result.order_id.split("/")[1] || "run";
294
305
  const ledger = join(evalDir, `.verdicts-${target}.jsonl`);
306
+ const runId = readRunId(cwd, slug);
295
307
  let run = 1;
296
308
  if (existsSync(ledger)) {
297
309
  const prior = readFileSync(ledger, "utf8").trim().split(/\n/).filter(Boolean).map((l) => { try { return JSON.parse(l); } catch { return null; } }).filter(Boolean);
298
310
  run = prior.reduce((mx, r) => Math.max(mx, r.run || 0), 0) + 1;
299
311
  }
300
312
  const lines = result.verdict.criteria.map((c) => JSON.stringify({
301
- run, dimension: c.dimension || "spec-conformance", criterion: c.criterion,
313
+ run, run_id: runId, dimension: c.dimension || "spec-conformance", criterion: c.criterion,
302
314
  verdict: c.verdict, confidence: c.confidence, reprobed: !!c.reprobed,
303
- evidence: c.evidence || "", at: new Date().toISOString(),
315
+ evidence: c.evidence || "", traces_to: Array.isArray(c.traces_to) ? c.traces_to : [],
316
+ at: new Date().toISOString(),
304
317
  })).join("\n");
305
318
  appendFileSync(ledger, lines + "\n");
306
319
  summary.verdict_lines = result.verdict.criteria.length;
@@ -36,6 +36,7 @@ import {
36
36
  } from "../lib/paths.mjs";
37
37
  import { readTrials } from "../verify/t0.mjs";
38
38
  import { ratchetReport } from "../probe/stats.mjs";
39
+ import { projectRequirements, summaryLine } from "../probe/requirements.mjs";
39
40
  import { collectDiff, scanDiff, summarize } from "./leftovers.mjs";
40
41
 
41
42
  /** @returns {string} Today as `YYYY-MM-DD` (UTC). */
@@ -167,7 +168,7 @@ export function section(md, heading) {
167
168
  export function buildReport(facts) {
168
169
  const {
169
170
  slug, at, verdict, qa, rounds, board, t0, artifacts, ratchet,
170
- evalCriteria, evalBugs, qaFindings, decisions, discovered, intakeSha, leftovers,
171
+ evalCriteria, evalBugs, qaFindings, decisions, discovered, intakeSha, leftovers, requirements,
171
172
  } = facts;
172
173
 
173
174
  const L = [];
@@ -214,6 +215,40 @@ export function buildReport(facts) {
214
215
  L.push("");
215
216
  }
216
217
 
218
+ // The requirement matrix — the way back from a verdict to the clause the pitch asked for, frozen
219
+ // at the one moment the run's local evidence still exists. Omitted entirely when the run has no
220
+ // registry: a table of nothing reads as "no requirements", which is a different claim from "this
221
+ // run predates the registry". Derived like every other figure here, by the same probe the L4 line
222
+ // and GATE H's census read, so the three cannot disagree.
223
+ if (requirements?.registry && requirements.rows.length) {
224
+ L.push("## Requirements", "");
225
+ L.push("One row per registered clause. A requirement has evidence when an acceptance criterion",
226
+ "covers it AND a criterion grading it passed — `covers:` is the join, the judge's anchor is the",
227
+ "path back. This is a projection, never a verdict: it never blocked this ship.", "");
228
+ L.push(`**${summaryLine(requirements)}** · run \`${requirements.run_id ?? "unknown"}\``, "");
229
+ // A clause, an AC and a criterion are all free prose, and a literal pipe in any of them breaks
230
+ // the row into columns nobody wrote — a frozen report that misrenders its own evidence.
231
+ /**
232
+ * Escape a free-prose value for a Markdown table cell.
233
+ * @param {string} s - The value.
234
+ * @returns {string} The value with every literal pipe escaped.
235
+ */
236
+ const cell = (s) => String(s).replace(/\|/g, "\\|");
237
+ L.push("| REQ | source | evidence | covering AC | criterion | T0 |", "|---|---|---|---|---|---|");
238
+ for (const r of requirements.rows) {
239
+ const ac = r.covering_acs.length ? `${r.covering_acs[0].task_id}: ${r.covering_acs[0].ac}${r.covering_acs.length > 1 ? ` (+${r.covering_acs.length - 1})` : ""}` : "—";
240
+ const crit = r.criteria.length ? `${r.criteria[0].criterion}${r.criteria.length > 1 ? ` (+${r.criteria.length - 1})` : ""} → ${r.criteria.map((c) => c.verdict).join(",")}` : "—";
241
+ const t0h = r.t0.length ? r.t0.map((h) => String(h).slice(0, 12)).join(", ") : "—";
242
+ L.push(`| ${r.id} | ${cell(r.source || "—")} | ${r.evidence} | ${cell(ac)} | ${cell(crit)} | ${t0h} |`);
243
+ }
244
+ L.push("");
245
+ if (requirements.inconsistencies.length) {
246
+ L.push("Anchored to a requirement no acceptance criterion covers — reconcile, do not count as evidence:", "");
247
+ for (const i of requirements.inconsistencies) L.push(`- ${i.requirement} ← "${i.criterion}" (${i.verdict})`);
248
+ L.push("");
249
+ }
250
+ }
251
+
217
252
  // The ratchet aggregate is derived, ~10 scalars that do not grow with the run, which is why it
218
253
  // can live in the committed tier while `metrics/` correctly stays gitignored (ADR-0001: a
219
254
  // committed shard keyed on $HOSTNAME only grows). Without this the instrument existed and was
@@ -306,6 +341,7 @@ export function generate({ cwd, slug, verdict, qa }) {
306
341
  board: boardCensus(cwd, slug),
307
342
  t0: t0Summary(cwd, slug),
308
343
  ratchet: ratchetReport(readTrials(trials(cwd, slug))),
344
+ requirements: projectRequirements({ cwd, slug }),
309
345
  artifacts: verdictArtifactCount(cwd, slug),
310
346
  evalCriteria: section(evalReport, /^#+\s.*criteria/i) || section(evalReport, /^#+\s*spec-conformance/i),
311
347
  evalBugs: section(evalReport, /^#+\s*Bugs?\b/i),
@@ -34,6 +34,11 @@
34
34
  // SCOPE-COVERS a contract's covers entry that is not a REQ-id (warn), or names a REQ that
35
35
  // is not in requirements.md (red, when a registry exists) — shape alone let a scope
36
36
  // claim coverage of a requirement that does not exist
37
+ // REQ-UNCOVERED the other direction of the same edge: a registered requirement still marked
38
+ // covered that NO acceptance criterion grades and NO scope claims. SCOPE-COVERS asks
39
+ // whether a link resolves; this asks whether a requirement has one at all. Red here
40
+ // and only advisory in trace-lint, because a requirement nothing reaches is a plan
41
+ // defect the PO can still answer at L1b — cover it, or cut it on the record
37
42
  // SCOPE-PARTITION a task claimed by more than one scope. The UC anchor is a SPEC link, not an
38
43
  // assignment: one use case is routinely implemented by several scopes, so on a
39
44
  // four-scope/one-UC cut every scope claimed every task and would build all of them.
@@ -65,9 +70,18 @@ import { parseBoard, deriveUnlocks } from "../reduce/board.mjs";
65
70
  import { runArgs } from "../lib/argv.mjs";
66
71
  import { LOCAL } from "../lib/paths.mjs";
67
72
  import { specDir, scopesDir, tasksDir, intake, sharedRoot, requirements } from "../lib/paths.mjs";
68
- import { readAllContracts, unreadableReason, ucId, scopePartitionConflicts, SCOPE_CONTRACT } from "../lib/contract.mjs";
73
+ import { readAllContracts, unreadableReason, ucId, reqId, scopePartitionConflicts, SCOPE_CONTRACT } from "../lib/contract.mjs";
74
+ import { UNREADABLE, LEGACY_LAYOUT } from "../lib/contract.mjs";
75
+ import { validate as validateAgainstSchema, SCHEMAS_DIR } from "./envelope.mjs";
69
76
  import { breadboard as stagedBreadboard } from "../lib/paths.mjs";
70
77
  import { parseBreadboard, hasBreadboardTables, idCounts } from "../lib/breadboard.mjs";
78
+ // ONE implementation of covers-closure, two reporters: trace-lint narrates it, spec-lint gates it.
79
+ // Re-deriving either here is how the advisory report and the gate start disagreeing about which
80
+ // requirement is covered. This closes the import ring spec → trace → compile → probe/resume → spec,
81
+ // which holds only while no module in it dereferences an imported binding at module-evaluation
82
+ // time — do NOT add a top-level `const x = someImportedFn()` to any of the four.
83
+ import { parseRequirements, coveredReqIds } from "./trace.mjs";
84
+ import { readBoard } from "../compile.mjs";
71
85
 
72
86
  // Inlined from hooks/sandbox-guard.mjs so this skill ships self-contained (a skill's scripts
73
87
  // must not reach outside its own folder — channels that copy only skills/ would dangle).
@@ -347,7 +361,11 @@ export function lintScopeAnchors({ scopes, specDir: specRoot, reqIds = null, tas
347
361
  else if (id && !ids.has(id)) findings.push({ rule: "SCOPE-DEPS", level: "red", scope: where, detail: `depends_on "${id}" is not a scope in this run — the scheduler drops the edge, so this scope may build before its dependency` });
348
362
  }
349
363
  for (const r of s.covers || []) {
350
- const req = String(r).trim();
364
+ // ONE KEY SPACE. A pitch numbers its requirements `R<n>` and the registry keys off
365
+ // `REQ-<n>`; `reqId` maps the first onto the second BEFORE the pattern below, so a link the
366
+ // planner actually wrote resolves instead of reading as a shape warning nobody can act on.
367
+ // A reference neither space recognises comes back verbatim and still fails the pattern.
368
+ const req = reqId(r);
351
369
  if (!/^REQ-[A-Z0-9-]+$/i.test(req)) {
352
370
  findings.push({ rule: "SCOPE-COVERS", level: "warn", scope: where, detail: `covers "${r}" is not a REQ-id — the requirement edge will not resolve` });
353
371
  continue;
@@ -372,6 +390,55 @@ export function lintScopeAnchors({ scopes, specDir: specRoot, reqIds = null, tas
372
390
  return findings;
373
391
  }
374
392
 
393
+ /**
394
+ * REQ-UNCOVERED — a live requirement that nothing in the plan reaches.
395
+ *
396
+ * THE OTHER DIRECTION OF THE COVERS EDGE. `SCOPE-COVERS` walks the links that exist and asks
397
+ * whether each one resolves; a requirement with no link at all satisfies it perfectly. Measured on
398
+ * a full run of one pitch: twenty-one requirements, every one of them with an acceptance criterion
399
+ * somewhere, and only eleven reaching a criterion the judge grades — the board is the last place a
400
+ * requirement can be dropped without anything going red, because after L1b nobody re-reads the
401
+ * pitch.
402
+ *
403
+ * WHY THE BOARD HERE IS `readBoard`, NOT `lint()`'s `tasks`. `parseBoard` (`kernel/reduce/board.mjs`)
404
+ * builds the scheduling view and its records carry no `acceptance_criteria` field at all, while
405
+ * `coveredReqIds` reads exactly that field — feed it the wrong board and the covered set is empty
406
+ * and EVERY requirement reds on EVERY run. `readBoard` (`kernel/compile.mjs`) is the parser that
407
+ * carries the criteria, and it is the only other one there may be: a second parser of the task file
408
+ * is explicitly ruled out where the first one lives.
409
+ *
410
+ * A SCOPE'S CLAIM COUNTS. The arm is about requirements nothing reaches, not about which layer
411
+ * reaches them: a clause claimed by a contract's `covers:` has an owner who answers for it at L1b,
412
+ * even before the criterion that grades it is written. `CUT (PO-approved)` is likewise an answer
413
+ * already given, not a defect — which is why `status` is read rather than assumed.
414
+ *
415
+ * @param {{clauses:Array<{id:string, clause:string, source:string, status:string}>,
416
+ * board:Array<object>, scopes:Array<{covers?:string[]}>}} input - The registry clauses
417
+ * (`parseRequirements`), the board `readBoard` parsed, and the scope contracts. An empty
418
+ * `clauses` (no registry on disk) yields no findings — absent artifact ⇒ arm skipped.
419
+ * @returns {Array<{rule:string, level:("red"|"warn"), scope:string, detail:string}>} One red per
420
+ * uncovered live requirement; [] when every one is graded, claimed or cut.
421
+ */
422
+ export function lintRequirementCoverage({ clauses = [], board = [], scopes = [] }) {
423
+ const findings = [];
424
+ const graded = coveredReqIds(board);
425
+ // The contracts speak the pitch's numbering as readily as the registry's; `reqId` lands both in
426
+ // the one key space before the comparison, exactly as SCOPE-COVERS does above.
427
+ const claimed = new Set();
428
+ for (const s of scopes) for (const r of s.covers || []) claimed.add(reqId(r).toUpperCase());
429
+ for (const c of clauses) {
430
+ if (c.status !== "covered") continue; // CUT (PO-approved) — an answer on the record, not a gap
431
+ const id = c.id.toUpperCase();
432
+ if (graded.has(c.id) || claimed.has(id)) continue;
433
+ const from = c.source ? ` ← ${c.source}` : "";
434
+ findings.push({ rule: "REQ-UNCOVERED", level: "red", scope: c.id, detail:
435
+ `${c.id}${from} is graded by no acceptance criterion and claimed by no scope — "${(c.clause || "").slice(0, 60)}" ` +
436
+ "would ship unverified and nothing downstream would say so. Cover it with an AC carrying " +
437
+ `(covers: ${c.id}), or mark it CUT (PO-approved) in requirements.md.` });
438
+ }
439
+ return findings;
440
+ }
441
+
375
442
  /**
376
443
  * Every dependency cycle among the scopes, each reported once from its lowest-sorting member.
377
444
  * @param {Array<{scope_id:string, depends_on?:string[]}>} scopes - The contracts.
@@ -663,6 +730,51 @@ export function runBreadboard(cwd, slug, intakeContent) {
663
730
  return hasBreadboardTables(intakeContent) ? intakeContent : null;
664
731
  }
665
732
 
733
+ /**
734
+ * Every scope contract whose PARSED shape fails `$defs/ScopeContract`.
735
+ *
736
+ * `kernel/lib/contract.mjs`'s own banner promised this check — "spec-lint re-validates every parsed
737
+ * contract against domain.schema.json, so a hand-edit that breaks the shape fails loudly instead of
738
+ * silently widening a sandbox" — and it did not exist. `compile` validated, spec-lint did not, so a
739
+ * contract could pass GATE L1b green and then be refused at dispatch by the one reader that checked.
740
+ *
741
+ * Measured 2026-09-19 on a real run: a planner wrote every `required_states` table cell bare
742
+ * (`loading, error, ready`) where the dialect wants `[loading, error, ready]`, so all 32 manifest
743
+ * rows across the six UI scopes parsed as strings. `verify spec` reported `red=0`; `compile` then
744
+ * refused all six with `expected array, got string`, and those scopes were never dispatched — no
745
+ * order, no leg, no T0 trial. The round reached EVAL with six of eighteen scopes missing and the
746
+ * evaluator escalated rather than grading. This arm turns that into a red at the gate, naming the
747
+ * scope and the field, with the message the compiler would otherwise produce an hour later.
748
+ *
749
+ * The validator is the one `compile` already uses; there is no second implementation here.
750
+ *
751
+ * @param {Array<{contract:object, path:string}>} contracts - Parsed contracts with their paths.
752
+ * @param {object} domainSchema - The parsed `domain.schema.json`.
753
+ * @returns {Array<{rule:string, level:string, scope:string, detail:string}>} One red per invalid
754
+ * contract; [] when the schema cannot be read (absent artifact ⇒ arm skipped).
755
+ */
756
+ export function lintContractSchema(contracts, domainSchema) {
757
+ const def = domainSchema?.$defs?.ScopeContract;
758
+ if (!def) return [];
759
+ const schema = { ...def, $defs: domainSchema.$defs };
760
+ const out = [];
761
+ for (const { contract, path } of contracts) {
762
+ const c = { ...contract };
763
+ delete c[UNREADABLE];
764
+ delete c[LEGACY_LAYOUT];
765
+ let res;
766
+ try { res = validateAgainstSchema(c, schema); } catch { continue; } // fail open, never closed
767
+ if (res?.valid) continue;
768
+ out.push({
769
+ rule: "CONTRACT-SCHEMA", level: "red", scope: contract.scope_id || path,
770
+ detail: `the contract parses, but not into the shape a WorkOrder carries — ${(res.errors || [])[0] || "schema validation failed"}. ` +
771
+ `compile refuses an order that fails its own schema, so as written this scope would be silently undispatched. ` +
772
+ `A list in a table cell is written [a, b], brackets and all.`,
773
+ });
774
+ }
775
+ return out;
776
+ }
777
+
666
778
  /**
667
779
  * Run the full spec lint (scopes + structure) for a slug.
668
780
  * @param {{cwd:string, slug:string}} opts - Working root and feature slug.
@@ -679,10 +791,21 @@ export function lint({ cwd, slug }) {
679
791
  const intakeContent = existsSync(intakePath) ? readFileSync(intakePath, "utf8") : "";
680
792
  // The REQ registry, when the tree has one — absent means covers-closure simply cannot apply.
681
793
  const reqFile = requirements(cwd, slug);
682
- const reqIds = existsSync(reqFile)
683
- ? new Set([...readFileSync(reqFile, "utf8").matchAll(/\bREQ-[A-Z0-9-]+/gi)].map((m) => m[0].toUpperCase()))
794
+ const reqText = existsSync(reqFile) ? readFileSync(reqFile, "utf8") : null;
795
+ const reqIds = reqText !== null
796
+ ? new Set([...reqText.matchAll(/\bREQ-[A-Z0-9-]+/gi)].map((m) => m[0].toUpperCase()))
684
797
  : null;
798
+ // Table rows only, and with the status/source cells REQ-UNCOVERED reports from — the id set
799
+ // above is deliberately looser (it also sees ids named in the registry's prose) and stays that
800
+ // way, because the two arms ask different questions of the same file.
801
+ const reqClauses = reqText !== null ? parseRequirements(reqText) : [];
685
802
  const repoFiles = walkFiles(cwd);
803
+ // Loaded HERE, not at module scope. `spec → trace → compile → probe/resume → spec` is a live
804
+ // import ring, and a top-level dereference of an imported binding is what would break it.
805
+ // Unreadable schema ⇒ the arm skips itself, like every other absent-artifact arm.
806
+ let domainSchema = null;
807
+ try { domainSchema = JSON.parse(readFileSync(join(SCHEMAS_DIR, "domain.schema.json"), "utf8")); } catch { /* arm skipped */ }
808
+
686
809
  const findings = [
687
810
  // A contract whose table this parser cannot see reads as a contract that declared no
688
811
  // table, and every rule below then passes for the part it could not read. Loud, not empty.
@@ -690,8 +813,11 @@ export function lint({ cwd, slug }) {
690
813
  .map(({ contract, path }) => ({ reason: unreadableReason(contract), scope: contract.scope_id || path }))
691
814
  .filter((x) => x.reason)
692
815
  .map((x) => ({ rule: "CONTRACT-UNREADABLE", level: "red", scope: x.scope, detail: `${x.reason} — the rules below could not check what they could not read` })),
816
+ ...lintContractSchema(contracts, domainSchema),
693
817
  ...lintScopes(scopes, repoFiles),
694
818
  ...lintScopeAnchors({ scopes, specDir: specRoot, reqIds, tasks }),
819
+ // `readBoard`, not the `tasks` above: only the compile-order parser carries acceptance_criteria.
820
+ ...lintRequirementCoverage({ clauses: reqClauses, board: readBoard(cwd, slug), scopes }),
695
821
  ...lintCommittedTier({ cwd, slug }),
696
822
  ...lintStructure({ specDir: specRoot, tasks, intakeContent }),
697
823
  ...(() => {