shapeup-sdlc 3.3.0 → 3.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,296 @@
1
+ #!/usr/bin/env node
2
+ // probe requirements — the way back: a pitch clause, the criterion that graded it, the verdict.
3
+ //
4
+ // WHY THIS IS A QUERY AND NOT A SENTENCE, which is the same reason `probe owner` is one. The
5
+ // requirement matrix is read at GATE L4 and cited by GATE H's census, and both are places where a
6
+ // narrated figure is indistinguishable from a measured one. Measured on the first verdict any run
7
+ // of the spine produced: 85 of 97 criterion rows carried a `traces_to` anchor in the WorkResult and
8
+ // 0 of 97 survived into the projection ingest wrote — so a matrix assembled from memory would have
9
+ // been assembled from an edge that no longer existed on disk. Every figure below is derived from
10
+ // files: the committed registry, the board's `covers:` clauses, the verdict ledger, and the EVAL
11
+ // result's own T0 citations. Nothing is passed in and nothing is written.
12
+ //
13
+ // THE JOIN, and which half is authoritative. A requirement has EVIDENCE when an acceptance
14
+ // criterion covers it (`(covers: REQ-…)` on the AC line — the planning-time edge, reviewed at L1b)
15
+ // AND a criterion that names it passed. `traces_to[]` is the navigation path from the judge's
16
+ // criterion back to the requirement, exactly what the schema already calls it: an anchor, never a
17
+ // grading input. A criterion whose `traces_to` names a REQ that no AC covers is printed as an
18
+ // INCONSISTENCY row and counted as nothing — folding it in would derive one L4 line from two
19
+ // unreconciled sources, which is the failure `probe owner` exists to prevent.
20
+ //
21
+ // IT PROJECTS ONE RUN. `order_id`, round and attempt all repeat across runs of one feature; the
22
+ // run key is the only thing that separates them, so a projection that ignored it would silently mix
23
+ // two runs of one slug. Rows written before the key reached the ledger carry no `run_id`; they are
24
+ // reported as unknown rather than folded into the run being projected.
25
+ //
26
+ // AN EMPTY JOIN READS AS CLEARLY AS A FULL ONE. A tree with no registry, a board with no `covers:`
27
+ // and a run with no verdict are all legitimate states, and each answers "no evidence" rather than
28
+ // failing — that answer is the point of the query, not an error in it.
29
+ //
30
+ // Usage:
31
+ // node "${CLAUDE_PLUGIN_ROOT}/kernel/harness.mjs" probe requirements --slug <slug> [--run-id <id>] [--format json|table] [--cwd <dir>]
32
+ //
33
+ // Exit code: 0 = answered (an empty projection is an answer), 2 = bad argv.
34
+
35
+ import { existsSync, readFileSync, readdirSync } from "node:fs";
36
+ import { resolve, join } from "node:path";
37
+ import { runArgs, isMain } from "../lib/argv.mjs";
38
+ import { requirements as requirementsFile, evaluationDir, resultsDir, readRunId } from "../lib/paths.mjs";
39
+ import { parseRequirements, coveredReqIds } from "../verify/trace.mjs";
40
+ import { readBoard } from "../compile.mjs";
41
+ import { reqId } from "../lib/contract.mjs";
42
+
43
+ /**
44
+ * Read a file, tolerating absence — every input to this projection is optional by design.
45
+ * @param {string} p - Absolute path.
46
+ * @returns {(string|null)} The contents, or null when the file is missing or unreadable.
47
+ */
48
+ function readIf(p) {
49
+ try { return existsSync(p) ? readFileSync(p, "utf8") : null; } catch { return null; }
50
+ }
51
+
52
+ /**
53
+ * The acceptance criteria on the LOCAL board that cover each requirement.
54
+ *
55
+ * The planning-time half of the join, and the authoritative one: a `covers:` clause is written when
56
+ * the plan is still cheap to change and is reviewed at L1b, whereas `traces_to` is written by the
57
+ * judge after the fact.
58
+ *
59
+ * NOT A SECOND COVERS-CLOSURE. Whether a requirement is covered is still decided by
60
+ * `coveredReqIds` in the oracle — there is one implementation of that question and this is not it.
61
+ * This walk exists only to keep the AC the closure discards, so the matrix can print WHICH
62
+ * criterion covers the clause instead of only that one does.
63
+ *
64
+ * @param {Array<object>} board - Task entries from `readBoard` (the only parser that carries
65
+ * `acceptance_criteria`; the scheduling view does not).
66
+ * @returns {Map<string, Array<{task_id:string, ac:string}>>} REQ-id → the ACs naming it, in board
67
+ * order. A requirement no AC covers is simply absent.
68
+ */
69
+ export function coveringAcs(board) {
70
+ const out = new Map();
71
+ for (const task of board || []) {
72
+ for (const ac of task.acceptance_criteria || []) {
73
+ if (typeof ac !== "object" || !Array.isArray(ac.covers)) continue;
74
+ for (const raw of ac.covers) {
75
+ const id = reqId(raw).toUpperCase();
76
+ if (!out.has(id)) out.set(id, []);
77
+ out.get(id).push({ task_id: task.id, ac: ac.text });
78
+ }
79
+ }
80
+ }
81
+ return out;
82
+ }
83
+
84
+ /**
85
+ * Every criterion row ingest projected, tagged with the EVAL target its ledger belongs to.
86
+ *
87
+ * The target is read off the FILE NAME rather than the row: `.verdicts-<target>.jsonl` is written
88
+ * per order, so the file is what says which dispatch produced the rows, and the same file is what
89
+ * locates the WorkResult carrying that dispatch's T0 citations.
90
+ *
91
+ * @param {string} cwd - Project root.
92
+ * @param {string} slug - Feature slug.
93
+ * @returns {Array<object>} One entry per ledger line, each the stored row plus `target`. Malformed
94
+ * lines are skipped; a missing evaluation directory yields [].
95
+ */
96
+ export function readVerdictRows(cwd, slug) {
97
+ const dir = evaluationDir(cwd, slug);
98
+ let files = [];
99
+ try { files = readdirSync(dir).filter((f) => /^\.verdicts-.+\.jsonl$/.test(f)).sort(); } catch { return []; }
100
+ const rows = [];
101
+ for (const f of files) {
102
+ const target = f.replace(/^\.verdicts-/, "").replace(/\.jsonl$/, "");
103
+ for (const line of (readIf(join(dir, f)) || "").split(/\r?\n/)) {
104
+ if (!line.trim()) continue;
105
+ try { rows.push({ ...JSON.parse(line), target }); } catch { /* a truncated line is not a verdict */ }
106
+ }
107
+ }
108
+ return rows;
109
+ }
110
+
111
+ /**
112
+ * The T0 artifacts an EVAL dispatch re-hashed, read from the WorkResult it wrote.
113
+ *
114
+ * The verdict ledger records criteria, not citations, so the machine fact a generator cannot
115
+ * fabricate lives one file over — in `results/<target>.json`. Read here rather than recomputed: the
116
+ * evaluator's own citation is what the round was accepted on.
117
+ *
118
+ * @param {string} cwd - Project root.
119
+ * @param {string} slug - Feature slug.
120
+ * @param {string} target - The EVAL order suffix (`evaluate-r1`), from the ledger's file name.
121
+ * @returns {Array<{scope_id:(string|null), path:(string|null), sha256:(string|null)}>} The cited
122
+ * artifacts; [] when the result is absent, unreadable or cites none.
123
+ */
124
+ export function t0Citations(cwd, slug, target) {
125
+ const body = readIf(join(resultsDir(cwd, slug), `${target}.json`));
126
+ if (!body) return [];
127
+ try {
128
+ const cites = JSON.parse(body)?.verdict?.t0_citations;
129
+ if (!Array.isArray(cites)) return [];
130
+ return cites.map((c) => ({ scope_id: c?.scope_id ?? null, path: c?.path ?? null, sha256: c?.sha256 ?? null }));
131
+ } catch { return []; }
132
+ }
133
+
134
+ /**
135
+ * Project one run's requirement matrix from the artifacts on disk.
136
+ *
137
+ * @param {{cwd:string, slug:string, runId?:(string|null)}} opts - Project root, feature slug, and
138
+ * the run to project; omitted, it is resolved from the run's own receipt.
139
+ * @returns {{slug:string, run_id:(string|null), registry:boolean, totals:object,
140
+ * rows:Array<object>, inconsistencies:Array<object>, ledger:object}} `rows` is one entry per
141
+ * registered clause — its source, status, covering ACs, the criteria that named it and their
142
+ * verdicts, and each criterion's T0 citations — with `evidence` one of `PASS`, `no evidence` or
143
+ * `cut`. `inconsistencies` holds criteria tracing to a REQ no AC covers. `ledger` says how many
144
+ * rows were read, projected, skipped as another run's, and left unkeyed. No registry ⇒ `rows` is
145
+ * empty and `registry` is false: an empty projection, not an error.
146
+ */
147
+ export function projectRequirements({ cwd, slug, runId = undefined }) {
148
+ const regText = readIf(requirementsFile(cwd, slug));
149
+ const clauses = regText === null ? [] : parseRequirements(regText);
150
+ const board = readBoard(cwd, slug);
151
+ // `covered` decides; `acs` only says which criterion did the covering.
152
+ const covered = coveredReqIds(board);
153
+ const acs = coveringAcs(board);
154
+ const run = runId === undefined ? readRunId(cwd, slug) : runId;
155
+
156
+ const all = readVerdictRows(cwd, slug);
157
+ const ledger = { rows_read: all.length, rows_projected: 0, rows_other_run: 0, rows_unknown_run: 0 };
158
+ const mine = [];
159
+ for (const r of all) {
160
+ if (r.run_id === undefined || r.run_id === null || r.run_id === "") { ledger.rows_unknown_run++; continue; }
161
+ if (run !== null && r.run_id === run) { ledger.rows_projected++; mine.push(r); continue; }
162
+ ledger.rows_other_run++;
163
+ }
164
+
165
+ // The criteria that name each requirement, and the T0 artifacts the round that graded them cited.
166
+ const t0Cache = new Map();
167
+ /**
168
+ * The cited T0 artifacts for one EVAL target, read once per target.
169
+ * @param {string} target - The EVAL order suffix.
170
+ * @returns {Array<object>} The citation rows.
171
+ */
172
+ const t0For = (target) => {
173
+ if (!t0Cache.has(target)) t0Cache.set(target, t0Citations(cwd, slug, target));
174
+ return t0Cache.get(target);
175
+ };
176
+
177
+ const byReq = new Map();
178
+ const inconsistencies = [];
179
+ for (const r of mine) {
180
+ const anchors = Array.isArray(r.traces_to) ? r.traces_to : [];
181
+ for (const raw of anchors) {
182
+ const id = reqId(raw).toUpperCase();
183
+ const entry = {
184
+ criterion: r.criterion ?? "", dimension: r.dimension ?? "", verdict: r.verdict ?? "",
185
+ confidence: r.confidence ?? null, evidence: r.evidence ?? "", target: r.target,
186
+ run: r.run ?? null, run_id: r.run_id ?? null, t0: t0For(r.target),
187
+ };
188
+ // THE AUTHORITATIVE HALF DECIDES. An anchor pointing at a requirement no acceptance criterion
189
+ // covers is a claim the plan never made; it is printed so somebody can reconcile it, and
190
+ // counted as nothing.
191
+ if (!covered.has(id)) { inconsistencies.push({ requirement: id, ...entry, why: "no acceptance criterion covers this requirement — the anchor resolves to nothing the plan claimed" }); continue; }
192
+ if (!byReq.has(id)) byReq.set(id, []);
193
+ byReq.get(id).push(entry);
194
+ }
195
+ }
196
+
197
+ const rows = clauses.map((c) => {
198
+ const id = c.id.toUpperCase();
199
+ const covering = acs.get(id) || [];
200
+ const criteria = byReq.get(id) || [];
201
+ const cut = c.status !== "covered";
202
+ const passed = criteria.some((x) => String(x.verdict).toUpperCase() === "PASS");
203
+ return {
204
+ id: c.id,
205
+ source: c.source || "",
206
+ clause: c.clause || "",
207
+ status: c.status,
208
+ covering_acs: covering,
209
+ criteria,
210
+ t0: [...new Set(criteria.flatMap((x) => x.t0.map((t) => t.sha256).filter(Boolean)))],
211
+ evidence: cut ? "cut" : passed ? "PASS" : "no evidence",
212
+ };
213
+ });
214
+
215
+ const totals = {
216
+ total: rows.length,
217
+ pass: rows.filter((r) => r.evidence === "PASS").length,
218
+ cut: rows.filter((r) => r.evidence === "cut").length,
219
+ no_evidence: rows.filter((r) => r.evidence === "no evidence").length,
220
+ inconsistencies: inconsistencies.length,
221
+ };
222
+ return { slug, run_id: run, registry: regText !== null, totals, rows, inconsistencies, ledger };
223
+ }
224
+
225
+ /**
226
+ * The one-line summary GATE L4 prints and GATE H's census cites.
227
+ *
228
+ * @param {object} r - A report from {@link projectRequirements}.
229
+ * @returns {string} `15/17 PASS · 1 CUT (PO) · 1 no evidence (REQ-12 ← R12)`, or the empty-run
230
+ * phrasing when there is no registry to project.
231
+ */
232
+ export function summaryLine(r) {
233
+ if (!r.registry) return "n/a (no registry)";
234
+ if (!r.totals.total) return "0 requirements registered";
235
+ const gaps = r.rows.filter((x) => x.evidence === "no evidence");
236
+ const named = gaps.slice(0, 3).map((x) => (x.source ? `${x.id} ← ${x.source}` : x.id)).join(", ");
237
+ const parts = [`${r.totals.pass}/${r.totals.total} PASS`];
238
+ if (r.totals.cut) parts.push(`${r.totals.cut} CUT (PO)`);
239
+ if (r.totals.no_evidence) parts.push(`${r.totals.no_evidence} no evidence (${named}${gaps.length > 3 ? ", …" : ""})`);
240
+ if (r.totals.inconsistencies) parts.push(`${r.totals.inconsistencies} inconsistency ${r.totals.inconsistencies === 1 ? "row" : "rows"}`);
241
+ return parts.join(" · ");
242
+ }
243
+
244
+ /**
245
+ * Render the matrix as a fixed-width table for a human reading a gate block.
246
+ * @param {object} r - A report from {@link projectRequirements}.
247
+ * @returns {string} The header, one row per requirement, the summary line and any inconsistencies.
248
+ */
249
+ export function renderTable(r) {
250
+ const out = [];
251
+ const head = ["REQ", "source", "status", "covering AC", "criterion", "verdict", "T0"];
252
+ const rows = r.rows.map((x) => [
253
+ x.id, x.source || "—", x.evidence,
254
+ x.covering_acs.length ? `${x.covering_acs[0].task_id}: ${x.covering_acs[0].ac.slice(0, 40)}${x.covering_acs.length > 1 ? ` (+${x.covering_acs.length - 1})` : ""}` : "—",
255
+ x.criteria.length ? `${x.criteria[0].criterion.slice(0, 40)}${x.criteria.length > 1 ? ` (+${x.criteria.length - 1})` : ""}` : "—",
256
+ x.criteria.length ? x.criteria.map((c) => c.verdict).join(",") : "—",
257
+ x.t0.length ? x.t0.map((h) => String(h).slice(0, 12)).join(",") : "—",
258
+ ]);
259
+ if (rows.length) {
260
+ const w = head.map((h, i) => Math.max(h.length, ...rows.map((row) => String(row[i]).length)));
261
+ const line = (row) => row.map((c, i) => String(c).padEnd(w[i])).join(" ").trimEnd();
262
+ out.push(line(head), line(w.map((n) => "-".repeat(n))), ...rows.map(line), "");
263
+ }
264
+ out.push(`Requirements: ${summaryLine(r)}`);
265
+ out.push(`run_id: ${r.run_id ?? "unknown"} · ledger rows read ${r.ledger.rows_read}, projected ${r.ledger.rows_projected}, other run ${r.ledger.rows_other_run}, unkeyed (run_id: unknown) ${r.ledger.rows_unknown_run}`);
266
+ for (const i of r.inconsistencies) {
267
+ out.push(` ⚠ ${i.requirement}: "${String(i.criterion).slice(0, 60)}" ${i.verdict} — ${i.why}`);
268
+ }
269
+ return out.join("\n");
270
+ }
271
+
272
+ export const ARGV_SPEC = {
273
+ usage: "harness.mjs probe requirements --slug <slug> [--run-id <id>] [--format json|table] [--cwd <dir>]",
274
+ _: { arity: 0, max: 0, name: "(no positional operands)" },
275
+ slug: { type: "str", required: true },
276
+ "run-id": { type: "str" },
277
+ format: { type: "enum", values: ["json", "table"], default: "json" },
278
+ cwd: { type: "path" },
279
+ };
280
+
281
+ /**
282
+ * Answer "which requirement has evidence, and from which criterion" from the artifacts on disk.
283
+ *
284
+ * @param {string[]} rawArgv - The subcommand's own arguments (harness.mjs strips the verb words).
285
+ * @returns {Promise<void>} Exits 0 with the projection — an absent run, an absent registry and an
286
+ * absent verdict are all answers, never errors.
287
+ */
288
+ export async function cli(rawArgv) {
289
+ const args = runArgs(ARGV_SPEC, rawArgv);
290
+ const cwd = resolve(args.cwd || process.cwd());
291
+ const report = projectRequirements({ cwd, slug: args.slug, runId: args.runId ?? undefined });
292
+ console.log(args.format === "table" ? renderTable(report) : JSON.stringify(report, null, 2));
293
+ process.exit(0);
294
+ }
295
+
296
+ if (isMain(import.meta.url)) cli(process.argv.slice(2));
@@ -63,7 +63,7 @@ import { splitFrontmatter } from "../lib/contract.mjs";
63
63
  import { globToRegExp } from "../verify/spec.mjs";
64
64
  import {
65
65
  intake, harnessRun, wiringMap, projectProfile, scopesDir, resultsDir, ordersDir,
66
- orientDir, activeOrder, usecasesDir, breadboard, receipt, readReceipt,
66
+ orientDir, activeOrder, usecasesDir, breadboard, receipt, readReceipt, requirements,
67
67
  } from "../lib/paths.mjs";
68
68
  import { evalVerdict } from "./eval.mjs";
69
69
 
@@ -394,6 +394,12 @@ export function deriveResumeState(cwd, slug) {
394
394
  orient_dir: `.shapeup/${slug}/orient/`,
395
395
  has_orient_artifacts: hasOrientArtifacts(cwd, slug),
396
396
  has_spec_tree: hasSpecTree(cwd, slug, hr.spec_folder || null),
397
+ // A PLAIN FACT, DELIBERATELY NOT A PHASE. The requirements registry is dispatched once, before
398
+ // ANALYZE, and the orchestrator guards that one dispatch on this boolean. It is NOT an entry in
399
+ // PHASE_ARTIFACT, and adding it there would be a migration hazard rather than a tidier shape:
400
+ // that map is also `nextPhase()`'s ordered list, so every run recorded before the registry
401
+ // existed would fast-forward to the registry instead of to `build` on its next relaunch.
402
+ has_requirements: existsSync(requirements(cwd, slug)),
397
403
  has_wiring_map: existsSync(wiringMap(cwd, slug)),
398
404
  project_profile_path: projectProfile(cwd, slug),
399
405
  has_project_profile: existsSync(projectProfile(cwd, slug)),
@@ -68,7 +68,11 @@ const listField = (fm, key) => {
68
68
  */
69
69
  export function parseBoard(tasksDir) {
70
70
  if (!existsSync(tasksDir)) return [];
71
- return readdirSync(tasksDir)
71
+ // SORTED, because `criticalPath` breaks ties on strict `>` and therefore keeps the FIRST chain it
72
+ // meets among equal-hours chains. Directory order is a filesystem detail (APFS happens to return
73
+ // sorted; ext4's hash order does not, and neither does a rename), so an unsorted read made a
74
+ // derived value depend on which machine ran it. Every sibling reader in the kernel already sorts.
75
+ return readdirSync(tasksDir).sort()
72
76
  .filter((f) => /^TASK-[\w.-]+\.md$/i.test(f))
73
77
  .map((f) => {
74
78
  const body = readFileSync(join(tasksDir, f), "utf8");
@@ -130,10 +134,30 @@ export function criticalPath(tasks) {
130
134
  }
131
135
  return (memo[id] = { hours: best.hours + t.hours, chain: [...best.chain, id] });
132
136
  };
137
+ // TIES BREAK ON CONTENT, NOT ON ARRIVAL. `>` alone keeps whichever equal-hours chain is MET
138
+ // FIRST, which is input order — so this function returned a different critical path for the same
139
+ // board depending only on how its task list happened to be ordered. Measured: two disjoint
140
+ // 5-hour chains, five permutations of one list, two different answers.
141
+ //
142
+ // Sorting the reader that feeds it (`parseBoard`) makes the input stable and therefore hides this
143
+ // on any one machine, but it leaves the ORDER-DEPENDENCE in place one call up — a caller with its
144
+ // own ordering, or a future second reader, re-opens it. A derived value has to be a function of
145
+ // the board, not of the walk that produced it, so the tie is resolved here: same hours, then the
146
+ // lexicographically smaller chain. Deterministic on every machine and for every caller.
147
+ /**
148
+ * Is chain `c` a better critical path than the incumbent `b`?
149
+ * @param {{hours:number, chain:string[]}} c - The candidate chain.
150
+ * @param {{hours:number, chain:string[]}} b - The incumbent best.
151
+ * @returns {boolean} True when `c` has more hours, or ties on hours and sorts first by chain
152
+ * content — so the answer is a function of the board rather than of the iteration order.
153
+ */
154
+ const better = (c, b) => c.hours > b.hours
155
+ || (c.hours === b.hours && c.chain.length > 0
156
+ && (b.chain.length === 0 || c.chain.join("\u0000") < b.chain.join("\u0000")));
133
157
  let best = { hours: 0, chain: [] };
134
158
  for (const t of tasks) {
135
159
  const c = longest(t.id);
136
- if (c.hours > best.hours) best = c;
160
+ if (better(c, best)) best = c;
137
161
  }
138
162
  return best;
139
163
  }
@@ -32,7 +32,7 @@ import {
32
32
  localRoot, receipt as receiptPath, ordersDir, resultsDir, verdictsDir, trials as trialsPath,
33
33
  gates as gatesPath, scopesDir, usecasesDir, requirements as requirementsPath, wiringMap as wiringMapPath,
34
34
  } from "../lib/paths.mjs";
35
- import { readAllContracts, readContract, ucId, SCOPE_CONTRACT, WIRING_MAP } from "../lib/contract.mjs";
35
+ import { readAllContracts, readContract, ucId, reqId, SCOPE_CONTRACT, WIRING_MAP } from "../lib/contract.mjs";
36
36
  import { runIdFromReceipt } from "../lib/paths.mjs";
37
37
 
38
38
  /** The graph's home — one file per feature, beside the run trace it projects. */
@@ -43,7 +43,7 @@ export const WORK_NODES = ["Run", "Order", "Result", "Verdict", "Trial", "GateDe
43
43
  export const DOMAIN_NODES = ["Scope", "UseCase", "Requirement", "Seam"];
44
44
 
45
45
  /** Edge types. Each names a direction that is meaningful to read backwards. */
46
- export const EDGES = ["PRODUCED", "EVALUATES", "SUPERSEDES", "COVERS", "DEPENDS_ON", "DERIVED_FROM"];
46
+ export const EDGES = ["PRODUCED", "EVALUATES", "SUPERSEDES", "COVERS", "DEPENDS_ON", "DERIVED_FROM", "IMPLEMENTS"];
47
47
 
48
48
  /**
49
49
  * Read the graph as a log and fold it into nodes and edges.
@@ -63,7 +63,10 @@ export function readGraph(cwd, slug) {
63
63
  let row;
64
64
  try { row = JSON.parse(line); } catch { continue; } // a torn line proves nothing; skip it
65
65
  lines++;
66
- if (row.k === "node" && row.id) nodes.set(row.id, { ...(nodes.get(row.id) || {}), ...row });
66
+ // LAST LINE WINS, as the banner says — a REPLACE, not a merge. Merging kept attributes from
67
+ // superseded lines alive forever, so an attribute removed from an artifact survived in the
68
+ // projection and a rebuilt graph stopped matching the maintained one.
69
+ if (row.k === "node" && row.id) nodes.set(row.id, row);
67
70
  else if (row.k === "edge" && row.from && row.to && row.t) edges.set(`${row.from}|${row.t}|${row.to}`, row);
68
71
  }
69
72
  return { nodes, edges, lines };
@@ -194,16 +197,19 @@ export function project(cwd, slug) {
194
197
  source: g.source ?? null, note: g.note ?? null, round: g.round ?? null,
195
198
  run_id: g.run_id ?? runId ?? null,
196
199
  });
197
- // A round-scoped gate's decision depends on that round's T0 verdict(s) — the most honest
198
- // shape, since that is the evidence the decision was made against. Every other gate (and a
199
- // round-scoped one with no verdict yet on disk) depends on the Run instead, so no
200
- // `GateDecision` node is ever orphaned.
200
+ // EVERY gate depends on the Run, UNCONDITIONALLY — and a round-scoped one ALSO depends on that
201
+ // round's T0 verdict(s), the evidence it was decided against.
202
+ //
203
+ // The Run edge used to be an `else` FALLBACK, and a conditional edge TARGET cannot survive an
204
+ // append-only log. Gates are crossed BEFORE the round's verdict artifact lands, so the first
205
+ // projection minted `DEPENDS_ON run` and a later one added the verdict edges beside it —
206
+ // and `appendGraph` has no tombstone, so the fallback became permanent. The incremental graph
207
+ // then carried an edge a rebuild does not imply, breaking the one property this file exists to
208
+ // promise: that it can be deleted and rebuilt identically. Unconditional is also truer — a gate
209
+ // does depend on its run. The fix is not a guard; it is removing the condition.
210
+ if (runNode) edge(id, "DEPENDS_ON", runNode);
201
211
  const roundVerdicts = (g.gate === "L2" || g.gate === "L3") ? verdictIdsByRound.get(g.round) : null;
202
- if (roundVerdicts?.length) {
203
- for (const vid of roundVerdicts) edge(id, "DEPENDS_ON", vid);
204
- } else if (runNode) {
205
- edge(id, "DEPENDS_ON", runNode);
206
- }
212
+ for (const vid of roundVerdicts || []) edge(id, "DEPENDS_ON", vid);
207
213
  }
208
214
  }
209
215
 
@@ -221,7 +227,15 @@ export function project(cwd, slug) {
221
227
  // projected to two nodes. `--trace` is supposed to reach "the execution record"; it reached
222
228
  // whichever scope happened to be written last. `baseline_trial` is chosen from the same
223
229
  // scope's prior rows, so the SUPERSEDES edge resolves inside the same partition.
224
- const trialKey = (n) => `trial:${slug}:${t.scope_id ? `${t.scope_id}:` : ""}${n}`;
230
+ //
231
+ // AND THE RUN IS PART OF THE KEY TOO, for the same reason one step out. A trial ordinal
232
+ // restarts at 1 in the next run while `trials.jsonl` is APPEND-ONLY, so both runs' rows live
233
+ // on disk together and `trial:<slug>:<scope>:1` named two of them — the second run silently
234
+ // overwrote the first run's execution record, which is the identical defect the paragraph
235
+ // above records fixing once already, one key component short. `baseline_trial` is chosen from
236
+ // the same run's rows, so SUPERSEDES still resolves inside the partition.
237
+ const trialRun = t.run_id ?? runId ?? "norun";
238
+ const trialKey = (n) => `trial:${slug}:${trialRun}:${t.scope_id ? `${t.scope_id}:` : ""}${n}`;
225
239
  const id = trialKey(t.trial);
226
240
  node(id, "Trial", {
227
241
  trial: t.trial, round: t.round ?? null, attempt: t.attempt ?? null,
@@ -249,7 +263,10 @@ export function project(cwd, slug) {
249
263
  // UseCase nodes it was built from and `--trace` stopped there instead of reaching the
250
264
  // objective. The contract now names both, so both are projectable.
251
265
  for (const uc of contract.use_cases || []) edge(id, "IMPLEMENTS", `uc:${slug}:${ucId(uc)}`);
252
- for (const req of contract.covers || []) edge(id, "COVERS", `req:${slug}:${req}`);
266
+ // Normalised to the registry's key space on the way in, the same mapping spec-lint applies:
267
+ // a contract citing the pitch's `R<n>` and a Requirement node keyed `REQ-<n>` are one node,
268
+ // and an edge drawn to the other spelling is an edge to a node the graph does not hold.
269
+ for (const req of contract.covers || []) edge(id, "COVERS", `req:${slug}:${reqId(req)}`);
253
270
  for (const dep of contract.depends_on || []) edge(id, "DEPENDS_ON", `scope:${slug}:${dep}`);
254
271
  }
255
272
  }
@@ -8,7 +8,9 @@
8
8
  // task_results[] → tick AC boxes, flip task frontmatter status, append Execution Log,
9
9
  // update tasks/_index.md row, propagate unblocks (old P3.1–P3.6)
10
10
  // discoveries[] → append to .shapeup/<slug>/discovery/ledger.md (old P3.7 / QA H.3)
11
- // verdict.criteria[] → append evaluation/.verdicts-<target>.jsonl (old evaluator B.0)
11
+ // verdict.criteria[] → append evaluation/.verdicts-<target>.jsonl (old evaluator B.0), every
12
+ // row keyed by run_id and carrying the judge's traces_to[] anchor back to
13
+ // the requirement — see step 4 for why neither may be dropped here
12
14
  // verdict.refuted[] → un-tick refuted AC boxes + set eval_verdict frontmatter (old B.2/B.2b)
13
15
  // (the leg itself) → append one leg-completion row to .shapeup/<slug>/legs.jsonl
14
16
  //
@@ -286,21 +288,32 @@ function applyResultLocked(result, { cwd, slug }) {
286
288
  }
287
289
 
288
290
  // 4. Verdict bookkeeping (old evaluator B.0/B.2/B.2b) — judge returns data, ingest writes.
291
+ //
292
+ // THE PROJECTION KEEPS THE TWO KEYS A LATER READER CANNOT RE-DERIVE. Measured on the first real
293
+ // verdict of a spined run: 85 of 97 criteria carried `traces_to` in the WorkResult and 0 of 97
294
+ // survived into this file, so the anchor from a criterion back to the requirement it grades was
295
+ // written by the judge and discarded one step later. And the row carried only the monotonic `run`
296
+ // counter, which restarts per ledger file and repeats across runs of one slug — `run_id` is the
297
+ // only key that separates them, so a projection over this file silently mixed two runs without
298
+ // it. A WorkResult carries no run key; it is read off the run's own receipt, the same derivation
299
+ // every other writer here uses.
289
300
  if (result.verdict) {
290
301
  const evalDir = join(local, "evaluation");
291
302
  mkdirSync(evalDir, { recursive: true });
292
303
  if (result.verdict.criteria?.length) {
293
304
  const target = result.order_id.split("/")[1] || "run";
294
305
  const ledger = join(evalDir, `.verdicts-${target}.jsonl`);
306
+ const runId = readRunId(cwd, slug);
295
307
  let run = 1;
296
308
  if (existsSync(ledger)) {
297
309
  const prior = readFileSync(ledger, "utf8").trim().split(/\n/).filter(Boolean).map((l) => { try { return JSON.parse(l); } catch { return null; } }).filter(Boolean);
298
310
  run = prior.reduce((mx, r) => Math.max(mx, r.run || 0), 0) + 1;
299
311
  }
300
312
  const lines = result.verdict.criteria.map((c) => JSON.stringify({
301
- run, dimension: c.dimension || "spec-conformance", criterion: c.criterion,
313
+ run, run_id: runId, dimension: c.dimension || "spec-conformance", criterion: c.criterion,
302
314
  verdict: c.verdict, confidence: c.confidence, reprobed: !!c.reprobed,
303
- evidence: c.evidence || "", at: new Date().toISOString(),
315
+ evidence: c.evidence || "", traces_to: Array.isArray(c.traces_to) ? c.traces_to : [],
316
+ at: new Date().toISOString(),
304
317
  })).join("\n");
305
318
  appendFileSync(ledger, lines + "\n");
306
319
  summary.verdict_lines = result.verdict.criteria.length;
@@ -36,6 +36,7 @@ import {
36
36
  } from "../lib/paths.mjs";
37
37
  import { readTrials } from "../verify/t0.mjs";
38
38
  import { ratchetReport } from "../probe/stats.mjs";
39
+ import { projectRequirements, summaryLine } from "../probe/requirements.mjs";
39
40
  import { collectDiff, scanDiff, summarize } from "./leftovers.mjs";
40
41
 
41
42
  /** @returns {string} Today as `YYYY-MM-DD` (UTC). */
@@ -167,7 +168,7 @@ export function section(md, heading) {
167
168
  export function buildReport(facts) {
168
169
  const {
169
170
  slug, at, verdict, qa, rounds, board, t0, artifacts, ratchet,
170
- evalCriteria, evalBugs, qaFindings, decisions, discovered, intakeSha, leftovers,
171
+ evalCriteria, evalBugs, qaFindings, decisions, discovered, intakeSha, leftovers, requirements,
171
172
  } = facts;
172
173
 
173
174
  const L = [];
@@ -214,6 +215,40 @@ export function buildReport(facts) {
214
215
  L.push("");
215
216
  }
216
217
 
218
+ // The requirement matrix — the way back from a verdict to the clause the pitch asked for, frozen
219
+ // at the one moment the run's local evidence still exists. Omitted entirely when the run has no
220
+ // registry: a table of nothing reads as "no requirements", which is a different claim from "this
221
+ // run predates the registry". Derived like every other figure here, by the same probe the L4 line
222
+ // and GATE H's census read, so the three cannot disagree.
223
+ if (requirements?.registry && requirements.rows.length) {
224
+ L.push("## Requirements", "");
225
+ L.push("One row per registered clause. A requirement has evidence when an acceptance criterion",
226
+ "covers it AND a criterion grading it passed — `covers:` is the join, the judge's anchor is the",
227
+ "path back. This is a projection, never a verdict: it never blocked this ship.", "");
228
+ L.push(`**${summaryLine(requirements)}** · run \`${requirements.run_id ?? "unknown"}\``, "");
229
+ // A clause, an AC and a criterion are all free prose, and a literal pipe in any of them breaks
230
+ // the row into columns nobody wrote — a frozen report that misrenders its own evidence.
231
+ /**
232
+ * Escape a free-prose value for a Markdown table cell.
233
+ * @param {string} s - The value.
234
+ * @returns {string} The value with every literal pipe escaped.
235
+ */
236
+ const cell = (s) => String(s).replace(/\|/g, "\\|");
237
+ L.push("| REQ | source | evidence | covering AC | criterion | T0 |", "|---|---|---|---|---|---|");
238
+ for (const r of requirements.rows) {
239
+ const ac = r.covering_acs.length ? `${r.covering_acs[0].task_id}: ${r.covering_acs[0].ac}${r.covering_acs.length > 1 ? ` (+${r.covering_acs.length - 1})` : ""}` : "—";
240
+ const crit = r.criteria.length ? `${r.criteria[0].criterion}${r.criteria.length > 1 ? ` (+${r.criteria.length - 1})` : ""} → ${r.criteria.map((c) => c.verdict).join(",")}` : "—";
241
+ const t0h = r.t0.length ? r.t0.map((h) => String(h).slice(0, 12)).join(", ") : "—";
242
+ L.push(`| ${r.id} | ${cell(r.source || "—")} | ${r.evidence} | ${cell(ac)} | ${cell(crit)} | ${t0h} |`);
243
+ }
244
+ L.push("");
245
+ if (requirements.inconsistencies.length) {
246
+ L.push("Anchored to a requirement no acceptance criterion covers — reconcile, do not count as evidence:", "");
247
+ for (const i of requirements.inconsistencies) L.push(`- ${i.requirement} ← "${i.criterion}" (${i.verdict})`);
248
+ L.push("");
249
+ }
250
+ }
251
+
217
252
  // The ratchet aggregate is derived, ~10 scalars that do not grow with the run, which is why it
218
253
  // can live in the committed tier while `metrics/` correctly stays gitignored (ADR-0001: a
219
254
  // committed shard keyed on $HOSTNAME only grows). Without this the instrument existed and was
@@ -306,6 +341,7 @@ export function generate({ cwd, slug, verdict, qa }) {
306
341
  board: boardCensus(cwd, slug),
307
342
  t0: t0Summary(cwd, slug),
308
343
  ratchet: ratchetReport(readTrials(trials(cwd, slug))),
344
+ requirements: projectRequirements({ cwd, slug }),
309
345
  artifacts: verdictArtifactCount(cwd, slug),
310
346
  evalCriteria: section(evalReport, /^#+\s.*criteria/i) || section(evalReport, /^#+\s*spec-conformance/i),
311
347
  evalBugs: section(evalReport, /^#+\s*Bugs?\b/i),