shapeup-sdlc 3.3.0 → 3.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -34,6 +34,11 @@
34
34
  // SCOPE-COVERS a contract's covers entry that is not a REQ-id (warn), or names a REQ that
35
35
  // is not in requirements.md (red, when a registry exists) — shape alone let a scope
36
36
  // claim coverage of a requirement that does not exist
37
+ // REQ-UNCOVERED the other direction of the same edge: a registered requirement still marked
38
+ // covered that NO acceptance criterion grades and NO scope claims. SCOPE-COVERS asks
39
+ // whether a link resolves; this asks whether a requirement has one at all. Red here
40
+ // and only advisory in trace-lint, because a requirement nothing reaches is a plan
41
+ // defect the PO can still answer at L1b — cover it, or cut it on the record
37
42
  // SCOPE-PARTITION a task claimed by more than one scope. The UC anchor is a SPEC link, not an
38
43
  // assignment: one use case is routinely implemented by several scopes, so on a
39
44
  // four-scope/one-UC cut every scope claimed every task and would build all of them.
@@ -65,9 +70,18 @@ import { parseBoard, deriveUnlocks } from "../reduce/board.mjs";
65
70
  import { runArgs } from "../lib/argv.mjs";
66
71
  import { LOCAL } from "../lib/paths.mjs";
67
72
  import { specDir, scopesDir, tasksDir, intake, sharedRoot, requirements } from "../lib/paths.mjs";
68
- import { readAllContracts, unreadableReason, ucId, scopePartitionConflicts, SCOPE_CONTRACT } from "../lib/contract.mjs";
73
+ import { readAllContracts, unreadableReason, ucId, reqId, scopePartitionConflicts, SCOPE_CONTRACT } from "../lib/contract.mjs";
74
+ import { UNREADABLE, LEGACY_LAYOUT } from "../lib/contract.mjs";
75
+ import { validate as validateAgainstSchema, SCHEMAS_DIR } from "./envelope.mjs";
69
76
  import { breadboard as stagedBreadboard } from "../lib/paths.mjs";
70
77
  import { parseBreadboard, hasBreadboardTables, idCounts } from "../lib/breadboard.mjs";
78
+ // ONE implementation of covers-closure, two reporters: trace-lint narrates it, spec-lint gates it.
79
+ // Re-deriving either here is how the advisory report and the gate start disagreeing about which
80
+ // requirement is covered. This closes the import ring spec → trace → compile → probe/resume → spec,
81
+ // which holds only while no module in it dereferences an imported binding at module-evaluation
82
+ // time — do NOT add a top-level `const x = someImportedFn()` to any of the four.
83
+ import { parseRequirements, coveredReqIds } from "./trace.mjs";
84
+ import { readBoard } from "../compile.mjs";
71
85
 
72
86
  // Inlined from hooks/sandbox-guard.mjs so this skill ships self-contained (a skill's scripts
73
87
  // must not reach outside its own folder — channels that copy only skills/ would dangle).
@@ -347,7 +361,11 @@ export function lintScopeAnchors({ scopes, specDir: specRoot, reqIds = null, tas
347
361
  else if (id && !ids.has(id)) findings.push({ rule: "SCOPE-DEPS", level: "red", scope: where, detail: `depends_on "${id}" is not a scope in this run — the scheduler drops the edge, so this scope may build before its dependency` });
348
362
  }
349
363
  for (const r of s.covers || []) {
350
- const req = String(r).trim();
364
+ // ONE KEY SPACE. A pitch numbers its requirements `R<n>` and the registry keys off
365
+ // `REQ-<n>`; `reqId` maps the first onto the second BEFORE the pattern below, so a link the
366
+ // planner actually wrote resolves instead of reading as a shape warning nobody can act on.
367
+ // A reference neither space recognises comes back verbatim and still fails the pattern.
368
+ const req = reqId(r);
351
369
  if (!/^REQ-[A-Z0-9-]+$/i.test(req)) {
352
370
  findings.push({ rule: "SCOPE-COVERS", level: "warn", scope: where, detail: `covers "${r}" is not a REQ-id — the requirement edge will not resolve` });
353
371
  continue;
@@ -372,6 +390,55 @@ export function lintScopeAnchors({ scopes, specDir: specRoot, reqIds = null, tas
372
390
  return findings;
373
391
  }
374
392
 
393
+ /**
394
+ * REQ-UNCOVERED — a live requirement that nothing in the plan reaches.
395
+ *
396
+ * THE OTHER DIRECTION OF THE COVERS EDGE. `SCOPE-COVERS` walks the links that exist and asks
397
+ * whether each one resolves; a requirement with no link at all satisfies it perfectly. Measured on
398
+ * a full run of one pitch: twenty-one requirements, every one of them with an acceptance criterion
399
+ * somewhere, and only eleven reaching a criterion the judge grades — the board is the last place a
400
+ * requirement can be dropped without anything going red, because after L1b nobody re-reads the
401
+ * pitch.
402
+ *
403
+ * WHY THE BOARD HERE IS `readBoard`, NOT `lint()`'s `tasks`. `parseBoard` (`kernel/reduce/board.mjs`)
404
+ * builds the scheduling view and its records carry no `acceptance_criteria` field at all, while
405
+ * `coveredReqIds` reads exactly that field — feed it the wrong board and the covered set is empty
406
+ * and EVERY requirement reds on EVERY run. `readBoard` (`kernel/compile.mjs`) is the parser that
407
+ * carries the criteria, and it is the only other one there may be: a second parser of the task file
408
+ * is explicitly ruled out where the first one lives.
409
+ *
410
+ * A SCOPE'S CLAIM COUNTS. The arm is about requirements nothing reaches, not about which layer
411
+ * reaches them: a clause claimed by a contract's `covers:` has an owner who answers for it at L1b,
412
+ * even before the criterion that grades it is written. `CUT (PO-approved)` is likewise an answer
413
+ * already given, not a defect — which is why `status` is read rather than assumed.
414
+ *
415
+ * @param {{clauses:Array<{id:string, clause:string, source:string, status:string}>,
416
+ * board:Array<object>, scopes:Array<{covers?:string[]}>}} input - The registry clauses
417
+ * (`parseRequirements`), the board `readBoard` parsed, and the scope contracts. An empty
418
+ * `clauses` (no registry on disk) yields no findings — absent artifact ⇒ arm skipped.
419
+ * @returns {Array<{rule:string, level:("red"|"warn"), scope:string, detail:string}>} One red per
420
+ * uncovered live requirement; [] when every one is graded, claimed or cut.
421
+ */
422
+ export function lintRequirementCoverage({ clauses = [], board = [], scopes = [] }) {
423
+ const findings = [];
424
+ const graded = coveredReqIds(board);
425
+ // The contracts speak the pitch's numbering as readily as the registry's; `reqId` lands both in
426
+ // the one key space before the comparison, exactly as SCOPE-COVERS does above.
427
+ const claimed = new Set();
428
+ for (const s of scopes) for (const r of s.covers || []) claimed.add(reqId(r).toUpperCase());
429
+ for (const c of clauses) {
430
+ if (c.status !== "covered") continue; // CUT (PO-approved) — an answer on the record, not a gap
431
+ const id = c.id.toUpperCase();
432
+ if (graded.has(c.id) || claimed.has(id)) continue;
433
+ const from = c.source ? ` ← ${c.source}` : "";
434
+ findings.push({ rule: "REQ-UNCOVERED", level: "red", scope: c.id, detail:
435
+ `${c.id}${from} is graded by no acceptance criterion and claimed by no scope — "${(c.clause || "").slice(0, 60)}" ` +
436
+ "would ship unverified and nothing downstream would say so. Cover it with an AC carrying " +
437
+ `(covers: ${c.id}), or mark it CUT (PO-approved) in requirements.md.` });
438
+ }
439
+ return findings;
440
+ }
441
+
375
442
  /**
376
443
  * Every dependency cycle among the scopes, each reported once from its lowest-sorting member.
377
444
  * @param {Array<{scope_id:string, depends_on?:string[]}>} scopes - The contracts.
@@ -663,6 +730,51 @@ export function runBreadboard(cwd, slug, intakeContent) {
663
730
  return hasBreadboardTables(intakeContent) ? intakeContent : null;
664
731
  }
665
732
 
733
+ /**
734
+ * Every scope contract whose PARSED shape fails `$defs/ScopeContract`.
735
+ *
736
+ * `kernel/lib/contract.mjs`'s own banner promised this check — "spec-lint re-validates every parsed
737
+ * contract against domain.schema.json, so a hand-edit that breaks the shape fails loudly instead of
738
+ * silently widening a sandbox" — and it did not exist. `compile` validated, spec-lint did not, so a
739
+ * contract could pass GATE L1b green and then be refused at dispatch by the one reader that checked.
740
+ *
741
+ * Measured 2026-09-19 on a real run: a planner wrote every `required_states` table cell bare
742
+ * (`loading, error, ready`) where the dialect wants `[loading, error, ready]`, so all 32 manifest
743
+ * rows across the six UI scopes parsed as strings. `verify spec` reported `red=0`; `compile` then
744
+ * refused all six with `expected array, got string`, and those scopes were never dispatched — no
745
+ * order, no leg, no T0 trial. The round reached EVAL with six of eighteen scopes missing and the
746
+ * evaluator escalated rather than grading. This arm turns that into a red at the gate, naming the
747
+ * scope and the field, with the message the compiler would otherwise produce an hour later.
748
+ *
749
+ * The validator is the one `compile` already uses; there is no second implementation here.
750
+ *
751
+ * @param {Array<{contract:object, path:string}>} contracts - Parsed contracts with their paths.
752
+ * @param {object} domainSchema - The parsed `domain.schema.json`.
753
+ * @returns {Array<{rule:string, level:string, scope:string, detail:string}>} One red per invalid
754
+ * contract; [] when the schema cannot be read (absent artifact ⇒ arm skipped).
755
+ */
756
+ export function lintContractSchema(contracts, domainSchema) {
757
+ const def = domainSchema?.$defs?.ScopeContract;
758
+ if (!def) return [];
759
+ const schema = { ...def, $defs: domainSchema.$defs };
760
+ const out = [];
761
+ for (const { contract, path } of contracts) {
762
+ const c = { ...contract };
763
+ delete c[UNREADABLE];
764
+ delete c[LEGACY_LAYOUT];
765
+ let res;
766
+ try { res = validateAgainstSchema(c, schema); } catch { continue; } // fail open, never closed
767
+ if (res?.valid) continue;
768
+ out.push({
769
+ rule: "CONTRACT-SCHEMA", level: "red", scope: contract.scope_id || path,
770
+ detail: `the contract parses, but not into the shape a WorkOrder carries — ${(res.errors || [])[0] || "schema validation failed"}. ` +
771
+ `compile refuses an order that fails its own schema, so as written this scope would be silently undispatched. ` +
772
+ `A list in a table cell is written [a, b], brackets and all.`,
773
+ });
774
+ }
775
+ return out;
776
+ }
777
+
666
778
  /**
667
779
  * Run the full spec lint (scopes + structure) for a slug.
668
780
  * @param {{cwd:string, slug:string}} opts - Working root and feature slug.
@@ -679,10 +791,21 @@ export function lint({ cwd, slug }) {
679
791
  const intakeContent = existsSync(intakePath) ? readFileSync(intakePath, "utf8") : "";
680
792
  // The REQ registry, when the tree has one — absent means covers-closure simply cannot apply.
681
793
  const reqFile = requirements(cwd, slug);
682
- const reqIds = existsSync(reqFile)
683
- ? new Set([...readFileSync(reqFile, "utf8").matchAll(/\bREQ-[A-Z0-9-]+/gi)].map((m) => m[0].toUpperCase()))
794
+ const reqText = existsSync(reqFile) ? readFileSync(reqFile, "utf8") : null;
795
+ const reqIds = reqText !== null
796
+ ? new Set([...reqText.matchAll(/\bREQ-[A-Z0-9-]+/gi)].map((m) => m[0].toUpperCase()))
684
797
  : null;
798
+ // Table rows only, and with the status/source cells REQ-UNCOVERED reports from — the id set
799
+ // above is deliberately looser (it also sees ids named in the registry's prose) and stays that
800
+ // way, because the two arms ask different questions of the same file.
801
+ const reqClauses = reqText !== null ? parseRequirements(reqText) : [];
685
802
  const repoFiles = walkFiles(cwd);
803
+ // Loaded HERE, not at module scope. `spec → trace → compile → probe/resume → spec` is a live
804
+ // import ring, and a top-level dereference of an imported binding is what would break it.
805
+ // Unreadable schema ⇒ the arm skips itself, like every other absent-artifact arm.
806
+ let domainSchema = null;
807
+ try { domainSchema = JSON.parse(readFileSync(join(SCHEMAS_DIR, "domain.schema.json"), "utf8")); } catch { /* arm skipped */ }
808
+
686
809
  const findings = [
687
810
  // A contract whose table this parser cannot see reads as a contract that declared no
688
811
  // table, and every rule below then passes for the part it could not read. Loud, not empty.
@@ -690,8 +813,11 @@ export function lint({ cwd, slug }) {
690
813
  .map(({ contract, path }) => ({ reason: unreadableReason(contract), scope: contract.scope_id || path }))
691
814
  .filter((x) => x.reason)
692
815
  .map((x) => ({ rule: "CONTRACT-UNREADABLE", level: "red", scope: x.scope, detail: `${x.reason} — the rules below could not check what they could not read` })),
816
+ ...lintContractSchema(contracts, domainSchema),
693
817
  ...lintScopes(scopes, repoFiles),
694
818
  ...lintScopeAnchors({ scopes, specDir: specRoot, reqIds, tasks }),
819
+ // `readBoard`, not the `tasks` above: only the compile-order parser carries acceptance_criteria.
820
+ ...lintRequirementCoverage({ clauses: reqClauses, board: readBoard(cwd, slug), scopes }),
695
821
  ...lintCommittedTier({ cwd, slug }),
696
822
  ...lintStructure({ specDir: specRoot, tasks, intakeContent }),
697
823
  ...(() => {
@@ -211,8 +211,16 @@ export function traceLint(slug, { cwd, gate = false }) {
211
211
  const findings = [];
212
212
 
213
213
  // 1. Covers-closure.
214
+ //
215
+ // THE WHOLE ARM IS GATED ON THE REGISTRY EXISTING — both halves of it, and the second half is the
216
+ // one that was missing. With no `requirements.md` there are no clauses, so `REQ-UNCOVERED` cannot
217
+ // fire; but `dangling` is derived from the BOARD, which needs no registry to carry a `covers:`
218
+ // clause, so a tree with no registry reported "covers-closure not applicable" in the same breath
219
+ // as a red finding for every `covers:` on the board. An arm that reports itself skipped and emits
220
+ // findings anyway is not skipped, and the report says the opposite of what the findings do.
214
221
  const reqPath = join(shared, "requirements.md");
215
- const clauses = existsSync(reqPath) ? parseRequirements(readFileSync(reqPath, "utf8")) : [];
222
+ const closureChecked = existsSync(reqPath);
223
+ const clauses = closureChecked ? parseRequirements(readFileSync(reqPath, "utf8")) : [];
216
224
  const board = readBoard(cwd, slug);
217
225
  const covered = coveredReqIds(board);
218
226
  const knownIds = new Set(clauses.map((c) => c.id));
@@ -227,12 +235,13 @@ export function traceLint(slug, { cwd, gate = false }) {
227
235
  findings.push({ severity: "red", code: "REQ-UNCOVERED", req: id,
228
236
  message: `${id} (status: covered) is named by no AC's covers: — the clause "${(c?.clause || "").slice(0, 60)}" would silently vanish. Cover it with an AC, or mark it CUT (PO-approved).` });
229
237
  }
230
- for (const id of dangling) {
231
- findings.push({ severity: "red", code: "COVERS-DANGLING", req: id,
232
- message: `an AC declares (covers: ${id}) but ${id} is not in requirements.md — a covers: link must resolve to a registered REQ.` });
238
+ if (closureChecked) {
239
+ for (const id of dangling) {
240
+ findings.push({ severity: "red", code: "COVERS-DANGLING", req: id,
241
+ message: `an AC declares (covers: ${id}) but ${id} is not in requirements.md — a covers: link must resolve to a registered REQ.` });
242
+ }
233
243
  }
234
244
 
235
- const closureChecked = existsSync(reqPath);
236
245
  const coversClosure = {
237
246
  checked: closureChecked,
238
247
  requirements_total: clauses.length,
@@ -240,8 +249,8 @@ export function traceLint(slug, { cwd, gate = false }) {
240
249
  cut_status: cut.length,
241
250
  covered_by_ac: [...covered].filter((id) => knownIds.has(id)).length,
242
251
  uncovered,
243
- dangling_covers: dangling,
244
- pass: uncovered.length === 0 && dangling.length === 0,
252
+ dangling_covers: closureChecked ? dangling : [],
253
+ pass: uncovered.length === 0 && (!closureChecked || dangling.length === 0),
245
254
  skipped_reason: closureChecked ? null : "no requirements.md registry — covers-closure not applicable (non-regression on pre-spine specs).",
246
255
  };
247
256
 
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "shapeup-sdlc",
3
- "version": "3.3.0",
3
+ "version": "3.5.0",
4
4
  "description": "Shape Up for coding agents \u2014 with gates the agent can't talk its way past. Harness for Claude Code.",
5
5
  "bin": {
6
6
  "shapeup-sdlc": "bin/init.mjs"
@@ -98,6 +98,21 @@ its phase; templates live in `assets/templates/`.
98
98
  (red). An invariant-backed regression task still anchors to its owning UC — there is no
99
99
  second path to green.
100
100
 
101
+ **The requirement edge is written on the AC line, or it does not exist.** An acceptance
102
+ criterion that grades a registry requirement ends with `(covers: REQ-…)` — the trailing clause,
103
+ in the checkbox text, not a mention in prose. Measured on two runs of one pitch: every
104
+ requirement had an acceptance criterion somewhere on the board and only half reached a criterion
105
+ the judge grades, because a board AC reaches the judge through the refuted list alone — it can
106
+ yield a FAIL and can never yield a PASS. An AC nothing cites by id produces no evidence for the
107
+ requirement it was written for.
108
+
109
+ **A requirement with no natural use-case home still becomes a task.** Contrast, localisation, a
110
+ performance ceiling, a test surface — a non-functional clause has no actor+action and so no UC of
111
+ its own, and the habit is to record it in the risk register, where nothing grades it. Give it a
112
+ task whose AC reaches the committed spec (the invariant, the contract field or the Test Surface
113
+ row that states it) and carries its `(covers: REQ-…)`. A line in the risk table is a note; a
114
+ covered AC is a requirement the run can be measured against.
115
+
101
116
  ---
102
117
 
103
118
  ## The other three operations — same craft, different payload + whitelist
@@ -106,7 +121,7 @@ second path to green.
106
121
  |---|---|---|
107
122
  | `reconcile` | Verify `ledger.feature == payload.feature` (mismatch → STOP). Map each `[+]` Keep item → its owning UC; new task continues numbering (never renumber); `~`/Cut → synthesis "Hammered Out" row, no file. A Keep item asserting a new invariant → APPEND `[INV-NN]` + TS-INV row to that UC (append-only sections in your substrate). A new actor/action with no UC → `status: "escalated"` + a `deviations[]` spec-ambiguity entry: spawning a UC mid-cycle is silent re-shaping, the PO decides. Finish with board-derive (appetite overflow → report) + spec-lint | re-run phases 1–5; edit UC Steps; resolve the appetite HAMMER yourself |
108
123
  | `retrofit-surface` | Append `## Test Surface` (derived rows only, after Error Cases) to each UC of a pre-surface spec; an all-sources-empty UC gets the explicit empty-sources line | touch anything else — append-only substrate |
109
- | `coverage` | Extract **atomic** customer requirement clauses from `payload.requirements` (default: the pitch) and write the SHARED `shapeup/<slug>/requirements.md` registry: one `\| REQ-id \| clause (verbatim) \| source \| status \| note \|` row per clause. Split compound sentences into one testable clause each — a clause lost *inside* a bigger sentence is a requirement nothing can be traced to. **Assign REQ-ids ONCE and freeze them** (they behave like scope_id, never TASK-NNN — every `covers:` link rots otherwise): re-running, append new clauses with fresh ids, mark a removed clause `CUT (PO-approved)`, never renumber or delete. Status starts `covered` (a live requirement); only the PO sets `CUT`. The REQ source itself is frozen — the registry is a separate derived file | edit the REQ source; renumber existing REQ-ids; delete a dropped clause instead of marking it CUT; invent a requirement not in the source |
124
+ | `coverage` | Extract **atomic** customer requirement clauses from `payload.requirements` (default: the pitch) and write the SHARED `shapeup/<slug>/requirements.md` registry: one `\| REQ-id \| clause (verbatim) \| source \| status \| note \|` row per clause. Split compound sentences into one testable clause each — a clause lost *inside* a bigger sentence is a requirement nothing can be traced to. **Assign REQ-ids ONCE and freeze them** (they behave like scope_id, never TASK-NNN — every `covers:` link rots otherwise): re-running, append new clauses with fresh ids, mark a removed clause `CUT (PO-approved)`, never renumber or delete. Status starts `covered` (a live requirement); only the PO sets `CUT`. The REQ source itself is frozen — the registry is a separate derived file. **Numbering.** A source clause already carrying an `R<n>` keeps its number — `R12` → `REQ-12` — and its `source` cell records where it came from verbatim (`shaping.md R12`), because that cell is the only thing that survives a re-run. A clause with no R-id takes the next free number ABOVE the highest `R<n>` in the source, so it can never collide with one added later. Splitting a compound clause keeps `REQ-12` for the first atomic part and records `shaping.md R12 (split 2/3)` for the rest — a requirement graded in parts is why splitting matters at all. On a re-run, match an existing id by its frozen `source` cell and clause text, **never** by re-deriving the number from the source's current order | edit the REQ source; renumber existing REQ-ids; delete a dropped clause instead of marking it CUT; invent a requirement not in the source; re-point an existing REQ-id because the source's R-numbers shifted |
110
125
  ---
111
126
 
112
127
  ## Anti-rationalization table
@@ -41,6 +41,16 @@ the ship report's census table.
41
41
  scalars and [a, b] lists, a `## Affordances` table for affordance_manifest, and a
42
42
  short `## Why this slice` paragraph. A reviewer must be able to read the substrate
43
43
  in a PR; regeneration preserves prose under headings you do not own.
44
+ ► affordance_manifest lives in the TABLE and NOWHERE ELSE. Do not also write it in
45
+ the frontmatter: nothing reads it there, so the copy is discarded — and a run has
46
+ been lost to exactly that. The frontmatter copy said required_states: [idle], the
47
+ table cell said a bare idle, the table won, the value was a string where an array
48
+ was required, and four scopes were never dispatched — the board green, the contract
49
+ lint-clean, each leg reporting done with no error, every round, until EVAL refused
50
+ to grade a round whose scopes had never run. A contract declaring no affordances
51
+ writes `affordance_manifest: []` in frontmatter and no table.
52
+ ► A LIST INSIDE A TABLE CELL IS WRITTEN `[a, b]`, brackets and all — `[idle]`, never
53
+ `idle`. A bare word in that cell is a string, and required_states is an array.
44
54
  scope_id, topology_type — the stable join key is the scope
45
55
  use_cases[] — the UC ids this scope implements.
46
56
  THE ONLY LINK YOU WRITE TO THE
@@ -53,7 +63,12 @@ the ship report's census table.
53
63
  the board's own use_case_refs
54
64
  covers[] — optional REQ-ids from
55
65
  requirements.md this scope answers
56
- for; stable, never renumbered
66
+ for; stable, never renumbered.
67
+ WRITE THE REGISTRY'S OWN KEY:
68
+ `REQ-12`, not the pitch's `R12`.
69
+ Both resolve — readers normalise —
70
+ but one spelling in the committed
71
+ contract is one thing to read
57
72
  depends_on[] — scope_ids this scope builds AFTER.
58
73
  This is the build ORDER — declare
59
74
  it whenever one scope consumes
@@ -72,6 +72,14 @@ H0.1 Unresolved scopes (breaker cases only):
72
72
  H0.2 QA findings (qa-edge-hunter's hunt-report.md, when present) — all `~` by default.
73
73
  H0.3 Discovered-task ledger entries still open (discovery/ledger.md, `[+]`/`~` unresolved).
74
74
  H0.4 Attempt-budget hammer proposals (scopes that exhausted their T0 attempts during BUILD).
75
+ H0.4b Requirements with no PASS evidence — the pitch clauses the run never showed working. Run
76
+ node "${CLAUDE_PLUGIN_ROOT}/kernel/harness.mjs" probe requirements --slug <slug> --format table
77
+ and take its `no evidence` rows; cite the row, the same way H0.0 cites ownership. Each is a
78
+ census item carrying its source clause (`REQ-12 ← shaping.md R12`). A `cut` row is an answer
79
+ the PO already gave — not an item. An inconsistency row (a criterion anchored to a
80
+ requirement no acceptance criterion covers) is reported to the PO as a reconciliation, never
81
+ counted as evidence and never promoted as a finding. No registry on disk → this input is
82
+ empty and the census is unchanged (absent artifact ⇒ arm skipped).
75
83
  H0.5 Classify every item: MUST-HAVE (the pitch's core problem is unsolved without it) vs
76
84
  NICE-TO-HAVE (`~`, improves but doesn't block the core promise). Default to NICE-TO-HAVE
77
85
  unless the item traces directly to a pitch boundary or a scope's business_goal — a
@@ -81,7 +89,8 @@ H0.5 Classify every item: MUST-HAVE (the pitch's core problem is unsolved witho
81
89
  **GATE H0 Output:**
82
90
  ```
83
91
  ⏸ GATE H0 — Census
84
- Must-have (unresolved) : [N] — [list, each with source: scope | QA | discovered | advisor-overflow]
92
+ Must-have (unresolved) : [N] — [list, each with source: scope | QA | discovered | requirement |
93
+ advisor-overflow]
85
94
  Nice-to-have (~) : [M]
86
95
  Carry candidates : [scopes still uphill/downhill, or exhausted attempt budget]
87
96
  ```
@@ -152,13 +152,23 @@ round. Write it so someone without your context can answer it in one reply.
152
152
  "t0_citations": [ { "scope_id": "cart", "path": "…/t0/verdicts/r2-a3.json", "sha256": "…" } ],
153
153
  "criteria": [ { "criterion": "UC-01 step 3", "dimension": "spec-conformance",
154
154
  "verdict": "FAIL", "confidence": "high", "reprobed": true,
155
- "evidence": "Pay click throws — apps/web/checkout/Pay.tsx:84" } ],
155
+ "evidence": "Pay click throws — apps/web/checkout/Pay.tsx:84",
156
+ "traces_to": ["REQ-4"] } ],
156
157
  "refuted": [ { "task_id": "TASK-007", "ac": "<the checkbox text your evidence disproves>" } ],
157
158
  "bugs": [ /* report-schema bug entries */ ]
158
159
  }
159
160
  }
160
161
  ```
161
162
 
163
+ **`traces_to` is copied, not invented.** Fill it from the `(covers: REQ-…)` clause of the
164
+ acceptance criteria your criterion grades: the AC already carries the link, written when the plan
165
+ was reviewed, and you record which requirement your criterion maps back to. An AC with no `covers:`
166
+ clause yields no anchor — leave the array empty rather than guessing, and never read the pitch to
167
+ supply one. This changes nothing you grade: the anchor is a navigation path, never a grading input,
168
+ and a criterion passes or fails on its evidence exactly as before. It matters downstream because
169
+ the requirement matrix at GATE L4 and the census at GATE H are projected from these anchors; a
170
+ verdict that drops them grades the build and says nothing about what the pitch asked for.
171
+
162
172
  **Every FAIL criterion's `evidence` MUST carry a `file:line` locator** — schema-enforced, not
163
173
  advice: the envelope is validated against `work-result.schema.json` at ingest and a locatorless
164
174
  FAIL is rejected before any write. A PASS may cite plain output. (Observed, not theorized: a
@@ -175,6 +185,7 @@ separation is the whole point of the architecture.
175
185
  ## Verification checklist
176
186
 
177
187
  - [ ] Every criterion traces to committed spec text (UC/domain-model/contract/Done-when/Non-Go)
188
+ - [ ] `traces_to` copied from the graded ACs' `covers:` clauses — empty where they carry none
178
189
  - [ ] Every PASS cites a confirming probe; every FAIL cites evidence or "NO EVIDENCE"
179
190
  - [ ] Every FAIL was re-probed once; confidence assigned per the ledger rule
180
191
  - [ ] Scoped spec → T0 citations present with recomputed sha256 (else the run returned `failed`)
@@ -297,6 +297,12 @@ Scope contracts present:
297
297
  - scope-summary "Done when" headline statements
298
298
  - the Deferred Places from ux-behavior.md (breadboard Places this shape will not build) —
299
299
  each one needs the PO's yes; a rejected deferral goes back to the planner as a screen
300
+ - the REQ → AC table from requirements.md, one row per registered requirement: REQ-id, the
301
+ source clause it came from (`REQ-12 ← shaping.md R12`), and the acceptance criterion that
302
+ grades it — or the scope that claims it, or CUT (PO-approved). Omitted entirely when the run
303
+ has no registry. A requirement with none of the three is already a red below; this table is
304
+ what the PO reads to answer it — cover it, or cut it on the record. Printed, never asked:
305
+ the table decides nothing at this gate
300
306
  No scope contracts (pre-v0.3.0, unchanged from v0.2.6):
301
307
  Read tasks/_index.md (LOCAL root). Print:
302
308
  - task count by package/variant (.shared / .be / .web / .mobile / .e2e)
@@ -312,7 +318,15 @@ this is the orchestrator's own re-confirmation before committing to a build sequ
312
318
  waiting to happen), PA1 (directory-aligned scope), PA2 (size cap), SCOPE-ANCHOR (a scope
313
319
  naming no committed use case, or one that does not resolve), TIER-DIRECTION (a committed
314
320
  contract naming LOCAL task ids), SCOPE-DEPS (a build-order id naming a scope that is not
315
- in this run), BREADBOARD-PLACE (a breadboard Place with UI affordances has no
321
+ in this run), REQ-UNCOVERED (a requirement in requirements.md that no acceptance criterion
322
+ grades and no scope claims — the PO's two ways out are an AC carrying `(covers: REQ-…)` or
323
+ `CUT (PO-approved)` in the registry; silent on a run with no registry),
324
+ CONTRACT-SCHEMA (a scope contract that parses but not into the shape a WorkOrder carries —
325
+ most often a list written bare in a table cell where the dialect wants `[a, b]`; without
326
+ this the compiler refuses the order later and the scope is never dispatched at all, with
327
+ the board green and the leg reporting done), CONTRACT-UNREADABLE (a table the parser could
328
+ not see, or a table field also declared in frontmatter where nothing reads it),
329
+ BREADBOARD-PLACE (a breadboard Place with UI affordances has no
316
330
  `## Screen: … (P#)` in ux-behavior.md and is not deferred), BREADBOARD-UI (a U# not
317
331
  specified on a screen of its own Place). Any red → HARD STOP, past a 🔴 at the
318
332
  architect's own checkpoint. The breadboard reds are the planner's to fix — add the screen
@@ -513,8 +527,15 @@ Feature : [slug] — [SHIPPED (deployed) | BUILT & VERIFIED — deploy pending
513
527
  Rounds : [r] (build+eval cycles)
514
528
  Verdict : PASS (dims: [spec-conformance]; not evaluated: [security, performance])
515
529
  QA : [hunt done — N findings, M promoted+fixed, rest ~ | skipped (--no-qa) | n/a (pre-QA spec)]
530
+ Requirements: [15/17 PASS · 1 CUT (PO) · 1 no evidence (REQ-12 ← R12) | n/a (no registry)]
516
531
  Ledger : harness-run.md
517
532
  ```
533
+ The Requirements line is TRANSCRIBED, never composed — run
534
+ `node "${CLAUDE_PLUGIN_ROOT}/kernel/harness.mjs" probe requirements --slug <slug> --format table`
535
+ and copy its `Requirements:` summary. It joins each registered clause to the acceptance criterion
536
+ that covers it and to the criterion the judge graded, over the run named in its own output. It
537
+ decides nothing here: a requirement with no PASS evidence is a fact GATE H's census and the
538
+ baseline comparison weigh, not a ship blocker.
518
539
  Question (max 1): "Anything to record before I close the run? (y/n) or provide feedback for the next sprint."
519
540
  On confirm:
520
541
  - If the PO provides substantive feedback (not just 'y' or empty) → automatically delegate via Agent (model: exec — see references/protocol.md "Invocation mechanism"): Skill(shapeup-sdlc-plugin:coach) with the provided feedback for RLHF. The coach runs its own GATE COACH-1 to have the PO categorize each rule, then files it under the responsible skill in `shapeup/knowledge-base/<skill>.md` (committed → team-shared). Coachable: `task-executor`, `ba-pitch-analyzer`, `qa-edge-hunter`, `orient`, `scope-architect`, `solution-architect` (each reads its own file at the top of its next run) and `tech-lead` (workflow guidance, read at the next GATE L0). Guidance never decides a gate: a filed rule may add a question or a check to a gate block, never an answer. The tech lead does not categorize the feedback itself — that is the coach's gate, by design (no assumptions).
@@ -341,6 +341,7 @@
341
341
  "spec_folder",
342
342
  "feature",
343
343
  "discovered_ledger",
344
+ "requirements",
344
345
  "kb_rules_path"
345
346
  ],
346
347
  "scope-architect": [
@@ -2611,6 +2612,10 @@
2611
2612
  "type": "boolean",
2612
2613
  "description": "ANALYZE finished: the spec folder's usecases/ carries at least one use case that is not _index.md. WIRE reads these — one wiring-map entry per use case — which is why ANALYZE precedes WIRE in the phase chain: dispatched against an empty spec folder, WIRE escalates on every launch."
2613
2614
  },
2615
+ "has_requirements": {
2616
+ "type": "boolean",
2617
+ "description": "The requirements registry is on disk: shapeup/<slug>/requirements.md exists. A PLAIN FACT, not a phase — the orchestrator guards its single `coverage` dispatch on this boolean, and it is deliberately absent from kernel/probe/resume.mjs's PHASE_ARTIFACT map, which doubles as nextPhase()'s ordered list: an entry there would fast-forward every run recorded before the registry existed to the registry instead of to build."
2618
+ },
2614
2619
  "has_wiring_map": {
2615
2620
  "type": "boolean",
2616
2621
  "description": "WIRE finished: shapeup/<slug>/wiring-map.md exists."
@@ -407,6 +407,9 @@ const RESUME = {
407
407
  eval_dimensions: { type: "array", items: { type: "string" } },
408
408
  has_orient_artifacts: { type: "boolean" },
409
409
  has_spec_tree: { type: "boolean" },
410
+ // The requirements registry — a fact, not a phase. See the COVERAGE block below for why it is
411
+ // guarded on this bare boolean and never asked about through `probe resume --require`.
412
+ has_requirements: { type: "boolean" },
410
413
  has_wiring_map: { type: "boolean" },
411
414
  has_project_profile: { type: "boolean" },
412
415
  // THE SHAPE THE KERNEL WRITES, not a convenient one. `probe resume` emits `{scope_id, path}` per
@@ -488,7 +491,7 @@ const ORIENT = {
488
491
  required: ["ok", "artifact_written", "spiked_area", "spike_result"],
489
492
  };
490
493
 
491
- /** analyze / wire — "did the artifact land?" is all a gate needs from them. */
494
+ /** coverage / analyze / wire — "did the artifact land?" is all a gate needs from them. */
492
495
  const PHASE_OK = {
493
496
  type: "object",
494
497
  properties: { ok: { type: "boolean" }, artifact_written: { type: "boolean" }, detail: { type: "string" } },
@@ -1006,8 +1009,47 @@ if (!rs.has_orient_artifacts) {
1006
1009
  if (g.stop) return withWarnings(g.stop);
1007
1010
  }
1008
1011
 
1009
- // ---- ANALYZE (spec tree + board) — ahead of WIRE, which reads its use cases -------------------
1012
+ // ---- COVERAGE (the requirements registry) — ahead of ANALYZE, whose ACs cite its ids ----------
1013
+ //
1014
+ // WHY IT RUNS AT ALL, and why here. The pitch's own requirement list is the one statement of what
1015
+ // the run was asked for, and until it is extracted into `shapeup/<slug>/requirements.md` there is
1016
+ // no stable key an acceptance criterion, a scope contract or a verdict can point back to. Measured
1017
+ // on a full run: the planner produced the requirement edge on the board and the judge never saw
1018
+ // it, because nothing on either side shared a key space. So the registry is written BEFORE the
1019
+ // board, not beside it — ANALYZE's acceptance criteria cite `REQ-<n>` ids, which have to exist
1020
+ // before they can be cited.
1021
+ //
1022
+ // IT IS NOT A PHASE, AND THAT IS THE WHOLE DESIGN OF THIS BLOCK.
1023
+ // · No `phase("Coverage")`: the dispatch belongs to the planning stretch the Analyze group
1024
+ // already covers (`setRunStatus("mapping")` spans it), so it never renders as an empty group
1025
+ // on a relaunch — the failure a per-phase progress box would otherwise have to pay a leg to
1026
+ // avoid, and there is no leg to spend here (see the next point).
1027
+ // · No `requirePhase()` / no `fastForward()`: both route to `probe resume --require`, whose
1028
+ // `--require` is an ENUM over `PHASE_ARTIFACT`'s keys. `coverage` is not one, so the call exits
1029
+ // 2, and this file reads any exit other than 6 as "the predicate was never asked" — a dispatch
1030
+ // that worked would abort the run. The skip is therefore narrated, and guarded on the bare
1031
+ // `has_requirements` boolean the resume state carries.
1032
+ // · Not in `PHASE_ARTIFACT` either: that map is also `nextPhase()`'s ordered list, so adding it
1033
+ // would fast-forward every pre-registry run to the registry instead of to `build`.
1010
1034
  phase("Analyze");
1035
+ if (!rs.has_requirements) {
1036
+ log(`COVERAGE — dispatching (slug ${slug})`);
1037
+ await setRunStatus("mapping", "Analyze");
1038
+ const c = await worker({
1039
+ skill: "ba-pitch-analyzer", operation: "coverage", schema: PHASE_OK, phase: "Analyze", label: "coverage",
1040
+ payload: { requirements: rs.intake_path, feature: slug },
1041
+ extra:
1042
+ "Extract the pitch's requirement clauses into the SHARED requirements registry, one atomic " +
1043
+ "clause per row. A clause carrying an R-id keeps its number as REQ-<n> and records the R-id " +
1044
+ "verbatim in its source cell; ids are assigned once and never renumbered.",
1045
+ });
1046
+ if (c.__failed) return diedAt("COVERAGE", c);
1047
+ await advisory(`reduce graph --slug ${slug}`, "Analyze", "graph:coverage");
1048
+ } else {
1049
+ log(`COVERAGE — a requirements registry is already on disk; not re-dispatching it`);
1050
+ }
1051
+
1052
+ // ---- ANALYZE (spec tree + board) — ahead of WIRE, which reads its use cases -------------------
1011
1053
  if (!rs.has_spec_tree) {
1012
1054
  log(`ANALYZE — dispatching (slug ${slug})`);
1013
1055
  // "mapping", not "analyzing": the kernel's RUN_STATUSES enum is deliberately COARSER than this
@@ -1193,6 +1235,17 @@ await advisory(`reduce hill --slug ${slug}`, "MapScopes", "hill-derive");
1193
1235
  const lastEval = rs.eval_rounds_done?.length ? Math.max(...rs.eval_rounds_done) : 0;
1194
1236
  let round = lastEval + 1;
1195
1237
  let verdict = null;
1238
+ // DERIVED FROM DISK AT EVERY LAUNCH, never accumulated only in memory. These two lists ARE GATE H's
1239
+ // census: `scope-hammer` is dispatched with `hammer_proposals`, and AGENTS.md makes a scope that
1240
+ // exhausted its attempt budget a queued GATE H proposal. As bare `const []` they were emptied by the
1241
+ // one event that most needs them intact — a gate PAUSE is a `return`, so the PO's answer is followed
1242
+ // by a FRESH LAUNCH whose accumulators start empty and whose round loop is then fast-forwarded past
1243
+ // the rounds that filled them. The PO was handed an empty cut list for a run that had genuinely
1244
+ // exhausted scopes. Unattended (`ci`) runs never pause, which is why no archived trace shows it.
1245
+ //
1246
+ // This file has now paid for the same class three times — `findings` in the temporal dead zone and
1247
+ // `payload.bugs` "lived in a variable, which a relaunch between two rounds resets to empty" are the
1248
+ // other two. The cure is the same each time and it is not a bigger variable: re-derive the fact.
1196
1249
  const allGreen = [];
1197
1250
  const allHammer = [];
1198
1251
  // OUTSIDE the loop, because its whole purpose is to cross a round boundary: round r's verdict is
@@ -1248,6 +1301,16 @@ while (verdict !== "pass" && round <= maxRounds) {
1248
1301
  const alreadyGreen = new Set(g?.green_scopes_by_round?.[String(round)] || []);
1249
1302
  if (alreadyGreen.size) log(`BUILD r${round} — ${alreadyGreen.size} scope(s) already green in the graph, skipping them`);
1250
1303
 
1304
+ // REBUILD THE CENSUS FROM THE GRAPH THIS QUERY JUST RETURNED. Everything a prior launch learned is
1305
+ // already on disk: a scope green in ANY round is green work, and a scope the run has touched but
1306
+ // never got green is what GATE H has to be shown. One query, already made, re-read — so a relaunch
1307
+ // after a paused gate carries the same census the launch that paused it would have.
1308
+ const greenEver = new Set(Object.values(g?.green_scopes_by_round || {}).flat());
1309
+ for (const sid of greenEver) if (!allGreen.includes(sid)) allGreen.push(sid);
1310
+ for (const sid of (g?.scopes || [])) {
1311
+ if (!greenEver.has(sid) && !allHammer.includes(sid) && scopes.some((x) => x.scope_id === sid)) allHammer.push(sid);
1312
+ }
1313
+
1251
1314
  // SCOPES FAN OUT. A scope contract is the definition of an independent subtask — disjoint
1252
1315
  // substrate, own fixtures, own ratchet — so the loop that ran them one at a time was leaving the
1253
1316
  // whole point of the contract on the floor. `pipeline()` has NO barrier between its stages: a
@@ -1281,13 +1344,25 @@ while (verdict !== "pass" && round <= maxRounds) {
1281
1344
  async (pre, s) => (pre?.pending ? buildScope(s, round) : pre),
1282
1345
  async (res, s) => {
1283
1346
  if (!res || res.__failed) return res;
1284
- if (res.resumed || !res.green) return res;
1285
- const confirmed = await query(`probe t0 --slug ${slug} --scope ${s.scope_id} --round ${round}`,
1286
- T0CHECK, "Build", `t0confirm:${s.scope_id}-r${round}`);
1287
- if (!confirmed?.green) {
1288
- log(`BUILD r${round} — ${s.scope_id} reported green but no T0 verdict is on disk for this ` +
1289
- `round; treating it as not green (the evaluator cites that artifact, and it is not there).`);
1290
- return { ...res, green: false, reason: "reported green with no T0 verdict artifact on disk" };
1347
+ if (!res.green) return res;
1348
+ // THE T0 RE-READ IS SKIPPED FOR A RESUMED SCOPE; THE LEG CHECK BELOW IS NOT.
1349
+ //
1350
+ // `resumed` means the graph already reported this scope green for this round, so re-reading
1351
+ // its T0 artifact would only confirm what the resume derivation just read. But these two
1352
+ // stages answer DIFFERENT questions, and the second one is precisely the question a resumed
1353
+ // scope is most likely to fail: a relaunch happens because the previous launch DIED, and a
1354
+ // leg that died between writing its result and running `reduce ingest` leaves exactly this
1355
+ // state — green T0 on disk, result never applied, board still `pending`. The short-circuit
1356
+ // used to cover both stages, so the one scope class known to be at risk was the one class
1357
+ // nobody asked, and the late-ingest repair nine lines below could never fire for it.
1358
+ if (!res.resumed) {
1359
+ const confirmed = await query(`probe t0 --slug ${slug} --scope ${s.scope_id} --round ${round}`,
1360
+ T0CHECK, "Build", `t0confirm:${s.scope_id}-r${round}`);
1361
+ if (!confirmed?.green) {
1362
+ log(`BUILD r${round} — ${s.scope_id} reported green but no T0 verdict is on disk for this ` +
1363
+ `round; treating it as not green (the evaluator cites that artifact, and it is not there).`);
1364
+ return { ...res, green: false, reason: "reported green with no T0 verdict artifact on disk" };
1365
+ }
1291
1366
  }
1292
1367
  // AND ITS RESULT HAS TO HAVE REACHED THE BOARD. A green T0 says the worker's fixtures ran and
1293
1368
  // passed; it says nothing about whether the WorkResult was applied, and this stage used to ask
@@ -1341,8 +1416,10 @@ while (verdict !== "pass" && round <= maxRounds) {
1341
1416
  }
1342
1417
  }
1343
1418
 
1344
- allGreen.push(...roundGreen);
1345
- allHammer.push(...roundHammer);
1419
+ for (const sid of roundGreen) if (!allGreen.includes(sid)) allGreen.push(sid);
1420
+ // A scope that went green this round is no longer a cut candidate, however it was queued earlier.
1421
+ for (const sid of roundGreen) { const i = allHammer.indexOf(sid); if (i !== -1) allHammer.splice(i, 1); }
1422
+ for (const sid of roundHammer) if (!allHammer.includes(sid) && !allGreen.includes(sid)) allHammer.push(sid);
1346
1423
 
1347
1424
  // INNER breaker: nothing green and something queued → GATE H. The census is scope-hammer's job.
1348
1425
  if (roundGreen.length === 0 && roundHammer.length > 0) {