shapeup-sdlc 3.4.0 → 3.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (43) hide show
  1. package/.claude-plugin/plugin.json +1 -1
  2. package/AGENTS.md +16 -4
  3. package/README.md +7 -3
  4. package/SECURITY.md +1 -1
  5. package/hooks/sandbox-guard.mjs +69 -6
  6. package/kernel/compile.mjs +33 -12
  7. package/kernel/harness.mjs +10 -4
  8. package/kernel/init/run-args.mjs +206 -0
  9. package/kernel/init/run.mjs +10 -0
  10. package/kernel/lib/contract.mjs +68 -1
  11. package/kernel/lib/paths.mjs +10 -0
  12. package/kernel/probe/concurrency.mjs +31 -6
  13. package/kernel/probe/digest.mjs +15 -1
  14. package/kernel/probe/owner.mjs +4 -1
  15. package/kernel/probe/requirements.mjs +296 -0
  16. package/kernel/probe/resume.mjs +195 -6
  17. package/kernel/probe/rounds.mjs +104 -0
  18. package/kernel/reduce/graph.mjs +5 -2
  19. package/kernel/reduce/ingest.mjs +69 -15
  20. package/kernel/reduce/ship.mjs +52 -31
  21. package/kernel/reduce/snapshot.mjs +23 -2
  22. package/kernel/report/export.mjs +54 -2
  23. package/kernel/report/facts.mjs +24 -2
  24. package/{skills/tech-lead → kernel}/schemas/domain.schema.json +20 -12
  25. package/kernel/verify/envelope.mjs +2 -2
  26. package/kernel/verify/skills.mjs +1 -1
  27. package/kernel/verify/spec.mjs +130 -4
  28. package/kernel/verify/trace.mjs +16 -7
  29. package/package.json +1 -1
  30. package/skills/ba-pitch-analyzer/SKILL.md +16 -1
  31. package/skills/coach/SKILL.md +8 -2
  32. package/skills/hill-chart/SKILL.md +3 -4
  33. package/skills/scope-architect/SKILL.md +16 -1
  34. package/skills/scope-hammer/SKILL.md +11 -2
  35. package/skills/spec-evaluator/SKILL.md +12 -1
  36. package/skills/tech-lead/SKILL.md +10 -10
  37. package/skills/tech-lead/references/gates.md +70 -12
  38. package/skills/tech-lead/references/protocol.md +4 -2
  39. package/skills/tech-lead/workflows/shapeup-run.js +176 -38
  40. package/skills/translator/SKILL.md +1 -1
  41. /package/{skills/tech-lead → kernel}/schemas/gate-answers.schema.json +0 -0
  42. /package/{skills/tech-lead → kernel}/schemas/work-order.schema.json +0 -0
  43. /package/{skills/tech-lead → kernel}/schemas/work-result.schema.json +0 -0
@@ -211,8 +211,16 @@ export function traceLint(slug, { cwd, gate = false }) {
211
211
  const findings = [];
212
212
 
213
213
  // 1. Covers-closure.
214
+ //
215
+ // THE WHOLE ARM IS GATED ON THE REGISTRY EXISTING — both halves of it, and the second half is the
216
+ // one that was missing. With no `requirements.md` there are no clauses, so `REQ-UNCOVERED` cannot
217
+ // fire; but `dangling` is derived from the BOARD, which needs no registry to carry a `covers:`
218
+ // clause, so a tree with no registry reported "covers-closure not applicable" in the same breath
219
+ // as a red finding for every `covers:` on the board. An arm that reports itself skipped and emits
220
+ // findings anyway is not skipped, and the report says the opposite of what the findings do.
214
221
  const reqPath = join(shared, "requirements.md");
215
- const clauses = existsSync(reqPath) ? parseRequirements(readFileSync(reqPath, "utf8")) : [];
222
+ const closureChecked = existsSync(reqPath);
223
+ const clauses = closureChecked ? parseRequirements(readFileSync(reqPath, "utf8")) : [];
216
224
  const board = readBoard(cwd, slug);
217
225
  const covered = coveredReqIds(board);
218
226
  const knownIds = new Set(clauses.map((c) => c.id));
@@ -227,12 +235,13 @@ export function traceLint(slug, { cwd, gate = false }) {
227
235
  findings.push({ severity: "red", code: "REQ-UNCOVERED", req: id,
228
236
  message: `${id} (status: covered) is named by no AC's covers: — the clause "${(c?.clause || "").slice(0, 60)}" would silently vanish. Cover it with an AC, or mark it CUT (PO-approved).` });
229
237
  }
230
- for (const id of dangling) {
231
- findings.push({ severity: "red", code: "COVERS-DANGLING", req: id,
232
- message: `an AC declares (covers: ${id}) but ${id} is not in requirements.md — a covers: link must resolve to a registered REQ.` });
238
+ if (closureChecked) {
239
+ for (const id of dangling) {
240
+ findings.push({ severity: "red", code: "COVERS-DANGLING", req: id,
241
+ message: `an AC declares (covers: ${id}) but ${id} is not in requirements.md — a covers: link must resolve to a registered REQ.` });
242
+ }
233
243
  }
234
244
 
235
- const closureChecked = existsSync(reqPath);
236
245
  const coversClosure = {
237
246
  checked: closureChecked,
238
247
  requirements_total: clauses.length,
@@ -240,8 +249,8 @@ export function traceLint(slug, { cwd, gate = false }) {
240
249
  cut_status: cut.length,
241
250
  covered_by_ac: [...covered].filter((id) => knownIds.has(id)).length,
242
251
  uncovered,
243
- dangling_covers: dangling,
244
- pass: uncovered.length === 0 && dangling.length === 0,
252
+ dangling_covers: closureChecked ? dangling : [],
253
+ pass: uncovered.length === 0 && (!closureChecked || dangling.length === 0),
245
254
  skipped_reason: closureChecked ? null : "no requirements.md registry — covers-closure not applicable (non-regression on pre-spine specs).",
246
255
  };
247
256
 
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "shapeup-sdlc",
3
- "version": "3.4.0",
3
+ "version": "3.6.0",
4
4
  "description": "Shape Up for coding agents \u2014 with gates the agent can't talk its way past. Harness for Claude Code.",
5
5
  "bin": {
6
6
  "shapeup-sdlc": "bin/init.mjs"
@@ -98,6 +98,21 @@ its phase; templates live in `assets/templates/`.
98
98
  (red). An invariant-backed regression task still anchors to its owning UC — there is no
99
99
  second path to green.
100
100
 
101
+ **The requirement edge is written on the AC line, or it does not exist.** An acceptance
102
+ criterion that grades a registry requirement ends with `(covers: REQ-…)` — the trailing clause,
103
+ in the checkbox text, not a mention in prose. Measured on two runs of one pitch: every
104
+ requirement had an acceptance criterion somewhere on the board and only half reached a criterion
105
+ the judge grades, because a board AC reaches the judge through the refuted list alone — it can
106
+ yield a FAIL and can never yield a PASS. An AC nothing cites by id produces no evidence for the
107
+ requirement it was written for.
108
+
109
+ **A requirement with no natural use-case home still becomes a task.** Contrast, localisation, a
110
+ performance ceiling, a test surface — a non-functional clause has no actor+action and so no UC of
111
+ its own, and the habit is to record it in the risk register, where nothing grades it. Give it a
112
+ task whose AC reaches the committed spec (the invariant, the contract field or the Test Surface
113
+ row that states it) and carries its `(covers: REQ-…)`. A line in the risk table is a note; a
114
+ covered AC is a requirement the run can be measured against.
115
+
101
116
  ---
102
117
 
103
118
  ## The other three operations — same craft, different payload + whitelist
@@ -106,7 +121,7 @@ second path to green.
106
121
  |---|---|---|
107
122
  | `reconcile` | Verify `ledger.feature == payload.feature` (mismatch → STOP). Map each `[+]` Keep item → its owning UC; new task continues numbering (never renumber); `~`/Cut → synthesis "Hammered Out" row, no file. A Keep item asserting a new invariant → APPEND `[INV-NN]` + TS-INV row to that UC (append-only sections in your substrate). A new actor/action with no UC → `status: "escalated"` + a `deviations[]` spec-ambiguity entry: spawning a UC mid-cycle is silent re-shaping, the PO decides. Finish with board-derive (appetite overflow → report) + spec-lint | re-run phases 1–5; edit UC Steps; resolve the appetite HAMMER yourself |
108
123
  | `retrofit-surface` | Append `## Test Surface` (derived rows only, after Error Cases) to each UC of a pre-surface spec; an all-sources-empty UC gets the explicit empty-sources line | touch anything else — append-only substrate |
109
- | `coverage` | Extract **atomic** customer requirement clauses from `payload.requirements` (default: the pitch) and write the SHARED `shapeup/<slug>/requirements.md` registry: one `\| REQ-id \| clause (verbatim) \| source \| status \| note \|` row per clause. Split compound sentences into one testable clause each — a clause lost *inside* a bigger sentence is a requirement nothing can be traced to. **Assign REQ-ids ONCE and freeze them** (they behave like scope_id, never TASK-NNN — every `covers:` link rots otherwise): re-running, append new clauses with fresh ids, mark a removed clause `CUT (PO-approved)`, never renumber or delete. Status starts `covered` (a live requirement); only the PO sets `CUT`. The REQ source itself is frozen — the registry is a separate derived file | edit the REQ source; renumber existing REQ-ids; delete a dropped clause instead of marking it CUT; invent a requirement not in the source |
124
+ | `coverage` | Extract **atomic** customer requirement clauses from `payload.requirements` (default: the pitch) and write the SHARED `shapeup/<slug>/requirements.md` registry: one `\| REQ-id \| clause (verbatim) \| source \| status \| note \|` row per clause. Split compound sentences into one testable clause each — a clause lost *inside* a bigger sentence is a requirement nothing can be traced to. **Assign REQ-ids ONCE and freeze them** (they behave like scope_id, never TASK-NNN — every `covers:` link rots otherwise): re-running, append new clauses with fresh ids, mark a removed clause `CUT (PO-approved)`, never renumber or delete. Status starts `covered` (a live requirement); only the PO sets `CUT`. The REQ source itself is frozen — the registry is a separate derived file. **Numbering.** A source clause already carrying an `R<n>` keeps its number — `R12` → `REQ-12` — and its `source` cell records where it came from verbatim (`shaping.md R12`), because that cell is the only thing that survives a re-run. A clause with no R-id takes the next free number ABOVE the highest `R<n>` in the source, so it can never collide with one added later. Splitting a compound clause keeps `REQ-12` for the first atomic part and records `shaping.md R12 (split 2/3)` for the rest — a requirement graded in parts is why splitting matters at all. On a re-run, match an existing id by its frozen `source` cell and clause text, **never** by re-deriving the number from the source's current order | edit the REQ source; renumber existing REQ-ids; delete a dropped clause instead of marking it CUT; invent a requirement not in the source; re-point an existing REQ-id because the source's R-numbers shifted |
110
125
  ---
111
126
 
112
127
  ## Anti-rationalization table
@@ -93,7 +93,7 @@ never lands in any worker's KB.
93
93
  Orchestrated, this skill is dispatched like every worker: a **WorkOrder** in (`--order <path>`,
94
94
  operation `coach` or `scan`), a **WorkResult** out. Standalone, the raw feedback is passed
95
95
  directly; it maps onto the one payload field registered for this worker in the central domain
96
- registry (`skills/tech-lead/schemas/domain.schema.json`, `x-payload-by-worker`):
96
+ registry (`kernel/schemas/domain.schema.json`, `x-payload-by-worker`):
97
97
 
98
98
  | Payload field | Standalone form | Meaning |
99
99
  |---|---|---|
@@ -117,7 +117,13 @@ fields nobody used" → "Prefer the minimum DTO that satisfies the AC; don't add
117
117
  fields"). Keep the originating why — a rule without its reason gets ignored or misapplied.
118
118
 
119
119
  ### Step 2 — ⏸ GATE COACH-1: Categorize (ASK, never assume)
120
- This is the load-bearing gate. **Do not infer which skill a rule belongs to** — a
120
+ This is the load-bearing gate. **Resolve it first** — `node
121
+ "${CLAUDE_PLUGIN_ROOT}/kernel/harness.mjs" gate --resolve COACH-1 --slug <slug>
122
+ [--file <path>|--preset <name>]` — so the ledger carries a row for the decision this gate makes,
123
+ same as every other gate in the run. Exit 0 (`decision=skip`) — an unattended lane with no live PO;
124
+ record nothing and stop here, the same outcome the CI preset's own note already documents. Exit 4
125
+ (`ask`) — proceed with the categorization below, which IS the PO conversation this decision opens.
126
+ **Do not infer which skill a rule belongs to** — a
121
127
  miscategorized rule lands in a file the wrong worker reads (or no worker reads). Present every
122
128
  candidate rule and ask the PO to assign each one. Emit this block, then stop and wait:
123
129
 
@@ -10,7 +10,7 @@ re-runs a computation that would erase true history.**
10
10
 
11
11
  You are not a worker: no WorkOrder, no WorkResult, invoked directly by the user (or by `/hill`)
12
12
  exactly like `shapeup` is. There is nothing to declare in
13
- `skills/tech-lead/schemas/domain.schema.json` and nothing to teach `harness compile` or
13
+ `kernel/schemas/domain.schema.json` and nothing to teach `harness compile` or
14
14
  `harness reduce ingest` — those steps exist only for dispatched workers.
15
15
 
16
16
  ## What you read
@@ -71,9 +71,8 @@ Build one `{ scope_id, phase }` object per file.
71
71
  ## Rendering — the injection contract
72
72
 
73
73
  The engine ships at `assets/dashboard.template.html` — a complete, self-contained HTML page
74
- (inline CSS/JS, no external fetch beyond Google Fonts, no build step — the same convention as
75
- this repo's own `docs/visualize/*.html`). Do not rewrite it from a text description; read it,
76
- fill in real data, and write the result.
74
+ (inline CSS/JS, no external fetch beyond Google Fonts, no build step). Do not rewrite it from a
75
+ text description; read it, fill in real data, and write the result.
77
76
 
78
77
  1. For each discovered slug, build one entry:
79
78
 
@@ -41,6 +41,16 @@ the ship report's census table.
41
41
  scalars and [a, b] lists, a `## Affordances` table for affordance_manifest, and a
42
42
  short `## Why this slice` paragraph. A reviewer must be able to read the substrate
43
43
  in a PR; regeneration preserves prose under headings you do not own.
44
+ ► affordance_manifest lives in the TABLE and NOWHERE ELSE. Do not also write it in
45
+ the frontmatter: nothing reads it there, so the copy is discarded — and a run has
46
+ been lost to exactly that. The frontmatter copy said required_states: [idle], the
47
+ table cell said a bare idle, the table won, the value was a string where an array
48
+ was required, and four scopes were never dispatched — the board green, the contract
49
+ lint-clean, each leg reporting done with no error, every round, until EVAL refused
50
+ to grade a round whose scopes had never run. A contract declaring no affordances
51
+ writes `affordance_manifest: []` in frontmatter and no table.
52
+ ► A LIST INSIDE A TABLE CELL IS WRITTEN `[a, b]`, brackets and all — `[idle]`, never
53
+ `idle`. A bare word in that cell is a string, and required_states is an array.
44
54
  scope_id, topology_type — the stable join key is the scope
45
55
  use_cases[] — the UC ids this scope implements.
46
56
  THE ONLY LINK YOU WRITE TO THE
@@ -53,7 +63,12 @@ the ship report's census table.
53
63
  the board's own use_case_refs
54
64
  covers[] — optional REQ-ids from
55
65
  requirements.md this scope answers
56
- for; stable, never renumbered
66
+ for; stable, never renumbered.
67
+ WRITE THE REGISTRY'S OWN KEY:
68
+ `REQ-12`, not the pitch's `R12`.
69
+ Both resolve — readers normalise —
70
+ but one spelling in the committed
71
+ contract is one thing to read
57
72
  depends_on[] — scope_ids this scope builds AFTER.
58
73
  This is the build ORDER — declare
59
74
  it whenever one scope consumes
@@ -72,6 +72,14 @@ H0.1 Unresolved scopes (breaker cases only):
72
72
  H0.2 QA findings (qa-edge-hunter's hunt-report.md, when present) — all `~` by default.
73
73
  H0.3 Discovered-task ledger entries still open (discovery/ledger.md, `[+]`/`~` unresolved).
74
74
  H0.4 Attempt-budget hammer proposals (scopes that exhausted their T0 attempts during BUILD).
75
+ H0.4b Requirements with no PASS evidence — the pitch clauses the run never showed working. Run
76
+ node "${CLAUDE_PLUGIN_ROOT}/kernel/harness.mjs" probe requirements --slug <slug> --format table
77
+ and take its `no evidence` rows; cite the row, the same way H0.0 cites ownership. Each is a
78
+ census item carrying its source clause (`REQ-12 ← shaping.md R12`). A `cut` row is an answer
79
+ the PO already gave — not an item. An inconsistency row (a criterion anchored to a
80
+ requirement no acceptance criterion covers) is reported to the PO as a reconciliation, never
81
+ counted as evidence and never promoted as a finding. No registry on disk → this input is
82
+ empty and the census is unchanged (absent artifact ⇒ arm skipped).
75
83
  H0.5 Classify every item: MUST-HAVE (the pitch's core problem is unsolved without it) vs
76
84
  NICE-TO-HAVE (`~`, improves but doesn't block the core promise). Default to NICE-TO-HAVE
77
85
  unless the item traces directly to a pitch boundary or a scope's business_goal — a
@@ -81,7 +89,8 @@ H0.5 Classify every item: MUST-HAVE (the pitch's core problem is unsolved witho
81
89
  **GATE H0 Output:**
82
90
  ```
83
91
  ⏸ GATE H0 — Census
84
- Must-have (unresolved) : [N] — [list, each with source: scope | QA | discovered | advisor-overflow]
92
+ Must-have (unresolved) : [N] — [list, each with source: scope | QA | discovered | requirement |
93
+ advisor-overflow]
85
94
  Nice-to-have (~) : [M]
86
95
  Carry candidates : [scopes still uphill/downhill, or exhausted attempt budget]
87
96
  ```
@@ -141,7 +150,7 @@ the harness (this is neither the generator nor the evaluator).
141
150
  Orchestrated, this skill is dispatched like every worker: a **WorkOrder** in (`--order <path>`,
142
151
  operation `hammer`), a **WorkResult** out. The standalone flags below map 1:1 onto the payload
143
152
  fields registered for this worker in the central domain registry
144
- (`skills/tech-lead/schemas/domain.schema.json`, `x-payload-by-worker`):
153
+ (`kernel/schemas/domain.schema.json`, `x-payload-by-worker`):
145
154
 
146
155
  | Payload field | Standalone flag | Meaning |
147
156
  |---|---|---|
@@ -152,13 +152,23 @@ round. Write it so someone without your context can answer it in one reply.
152
152
  "t0_citations": [ { "scope_id": "cart", "path": "…/t0/verdicts/r2-a3.json", "sha256": "…" } ],
153
153
  "criteria": [ { "criterion": "UC-01 step 3", "dimension": "spec-conformance",
154
154
  "verdict": "FAIL", "confidence": "high", "reprobed": true,
155
- "evidence": "Pay click throws — apps/web/checkout/Pay.tsx:84" } ],
155
+ "evidence": "Pay click throws — apps/web/checkout/Pay.tsx:84",
156
+ "traces_to": ["REQ-4"] } ],
156
157
  "refuted": [ { "task_id": "TASK-007", "ac": "<the checkbox text your evidence disproves>" } ],
157
158
  "bugs": [ /* report-schema bug entries */ ]
158
159
  }
159
160
  }
160
161
  ```
161
162
 
163
+ **`traces_to` is copied, not invented.** Fill it from the `(covers: REQ-…)` clause of the
164
+ acceptance criteria your criterion grades: the AC already carries the link, written when the plan
165
+ was reviewed, and you record which requirement your criterion maps back to. An AC with no `covers:`
166
+ clause yields no anchor — leave the array empty rather than guessing, and never read the pitch to
167
+ supply one. This changes nothing you grade: the anchor is a navigation path, never a grading input,
168
+ and a criterion passes or fails on its evidence exactly as before. It matters downstream because
169
+ the requirement matrix at GATE L4 and the census at GATE H are projected from these anchors; a
170
+ verdict that drops them grades the build and says nothing about what the pitch asked for.
171
+
162
172
  **Every FAIL criterion's `evidence` MUST carry a `file:line` locator** — schema-enforced, not
163
173
  advice: the envelope is validated against `work-result.schema.json` at ingest and a locatorless
164
174
  FAIL is rejected before any write. A PASS may cite plain output. (Observed, not theorized: a
@@ -175,6 +185,7 @@ separation is the whole point of the architecture.
175
185
  ## Verification checklist
176
186
 
177
187
  - [ ] Every criterion traces to committed spec text (UC/domain-model/contract/Done-when/Non-Go)
188
+ - [ ] `traces_to` copied from the graded ACs' `covers:` clauses — empty where they carry none
178
189
  - [ ] Every PASS cites a confirming probe; every FAIL cites evidence or "NO EVIDENCE"
179
190
  - [ ] Every FAIL was re-probed once; confidence assigned per the ledger rule
180
191
  - [ ] Scoped spec → T0 citations present with recomputed sha256 (else the run returned `failed`)
@@ -60,15 +60,15 @@ check the lane:
60
60
  legacy loop instead — `references/protocol.md` (BUILD(r)/EVAL) + `references/protocol.md`
61
61
  carry the full step-by-step for both the tiny lane and a scope-less BUILD loop, verbatim, non-
62
62
  regression. Stop reading this file here for that run.
63
- - **Otherwise** (the common case — a scoped spec, any auto level): build `RunArgs`
64
- (`domain.schema.json` `$defs/RunArgs` — `{slug, runId, autoLevel, answers, lane,
65
- models:{exec,eval,qa}, budgets:{maxRounds,attemptBudget,wallClockS}, pluginRoot, startedAt}`,
66
- plus every switch the operator typed — `references/gates.md` GATE L0.9 has the flag→field table,
67
- and a flag that stops here is a flag that was accepted and ignored). **Write that exact object to
68
- `.shapeup/<slug>/run-args.json` before launching**, fresh on every launch and relaunch: the flags
69
- reach the workflow as a value in memory, so it is the run's only evidence of what it was launched
70
- with, and a run that cannot state its own configuration cannot have a claim about it checked.
71
- Then launch with the **`Workflow` tool** — naming `init run`'s staged copy, never the install path:
63
+ - **Otherwise** (the common case — a scoped spec, any auto level): resolve every switch the
64
+ operator typed (`references/gates.md` GATE L0.9b has the flag→field table) and run the kernel's
65
+ sole `RunArgs` writer: `node "${CLAUDE_PLUGIN_ROOT}/kernel/harness.mjs" init run-args --slug
66
+ <slug> --auto-level <level> --exec-model <n> [--eval-model <n>] [--qa-model <n>] --max-rounds <N>
67
+ --attempts <N> --plugin-root "${CLAUDE_PLUGIN_ROOT}" [--answers <a>] [--lane <l>] [--no-eval]
68
+ [--no-qa] [--adversarial-verify] [--parallel-scopes <N>]`. It writes `.shapeup/<slug>/run-args.json`
69
+ fresh on every launch/relaunch and prints that identical object — **pass it to `Workflow`
70
+ verbatim, never re-type it**. Then launch with the **`Workflow` tool** — naming `init run`'s
71
+ staged copy, never the install path:
72
72
 
73
73
  ```
74
74
  Workflow({
@@ -114,7 +114,7 @@ did not actually receive from the PO — an unattended lane with no answer for a
114
114
 
115
115
  FIRST freeze the evidence — run state is gitignored, so `shapeup/<slug>/REPORT.md` (already
116
116
  written by `shapeup-run.js` via `harness reduce ship`, or write it now on a `gate_h` close) is all a
117
- teammate sees. Then emit:
117
+ teammate sees. Then RESOLVE the gate — `references/gates.md` GATE L4 has the call — and emit:
118
118
 
119
119
  ```
120
120
  ⏸ GATE L4 — Ship Sign-Off
@@ -113,11 +113,13 @@ Collect (explicit — never inferred):
113
113
  gate to be its first execution.
114
114
  ```
115
115
 
116
- **L0.9b — the launch record.** Every switch the operator typed becomes a `RunArgs` field, or it
117
- does nothing at all: the workflow cannot read a config file and cannot ask a follow-up, so a flag
118
- that stops at the skill boundary was accepted and ignored. That is not hypothetical — `--no-qa` was
119
- documented in seven places across the shipped set and inert in all of them, because no line of this
120
- protocol ever put `noQa` into the record.
116
+ **L0.9b — the launch record.** Every switch the operator typed to *this launch* becomes a `RunArgs`
117
+ field, or it does nothing at all: the workflow cannot read a config file and cannot ask a follow-up,
118
+ so a flag that stops at the skill boundary was accepted and ignored. That is not hypothetical —
119
+ `--no-qa` was documented in seven places across the shipped set and inert in all of them, because no
120
+ line of this protocol ever put `noQa` into the record. `--wall-clock-budget` is the one flag below
121
+ that is not a `RunArgs` field at all — it is consumed earlier, at `init run` itself, and never
122
+ needed to reach this launch; see its row for where it actually lands.
121
123
 
122
124
  | Flag | `RunArgs` field |
123
125
  |---|---|
@@ -125,14 +127,19 @@ protocol ever put `noQa` into the record.
125
127
  | `--no-qa` | `noQa: true` |
126
128
  | `--parallel-scopes N` | `maxParallelScopes: N` — how many scopes build at once (default 4; `1` = sequential) |
127
129
  | `--adversarial-verify` | `adversarialVerify: true` |
128
- | `--rounds N` / `--attempts N` / `--wall-clock-budget S` | `budgets.{maxRounds,attemptBudget,wallClockS}` |
130
+ | `--rounds N` / `--attempts N` | `budgets.{maxRounds,attemptBudget}` |
129
131
  | `--gate-answers <set>` | `answers` |
132
+ | `--wall-clock-budget S` | *(not a `RunArgs` field)* — typed once, on the `harness init run` command line itself, not on this launch; it lands straight in the run receipt as `wall_clock_budget_s`, and the deadline breaker reads that receipt field directly — consumed by `kernel/verify/budget.mjs` as `wall_clock_budget_s`. `budgets` declares only `maxRounds`/`attemptBudget` — the schema, `SKILL.md`'s own RunArgs contract line and this script's own header comment all agree there is no third member |
130
133
  | `--orch-model/--exec-model/--eval-model/--qa-model` | `models.{…}` (L0.8) |
131
134
 
132
- The assembled object is written to `.shapeup/<slug>/run-args.json` before the launch, fresh on every
133
- launch and relaunch. It is the only artifact that records what a run was configured with; the ship
134
- report, a resumed session and any later measurement all read it, and none of them can recover a
135
- value that only ever existed as an argument.
135
+ `harness init run-args` (invoked at Step 2 of `SKILL.md`) is the sole writer of the assembled
136
+ object: it takes the resolved values above, writes `.shapeup/<slug>/run-args.json` fresh on every
137
+ launch and relaunch, and prints the same object back so the launch never re-assembles it by hand. It
138
+ is the only artifact that records what a run was configured with; the ship report, a resumed session
139
+ and any later measurement all read it, and none of them can recover a value that only ever existed
140
+ as an argument. Step 2 is not merely advisory: `shapeup-run.js`'s own Preflight refuses to dispatch
141
+ ORIENT (or anything past it) when this file is missing at the run's local root — a launch that
142
+ skipped this step aborts there rather than proceeding on a silent default.
136
143
 
137
144
  **L0.0 — intake precondition (before any other L0 collection):**
138
145
  ```
@@ -161,7 +168,14 @@ Model matrix : orch=[model] exec=[model] eval=[model] qa=[model] digester=[scrip
161
168
  Budgets : round_budget=[N] (outer) attempt_budget=[N] (inner, per scope)
162
169
  Knowledge : [tech-lead.md — N workflow rules, M suggested values (confirmed above) | none — `/retro --scan` or `/retro --research <stack>` seeds it (optional)]
163
170
  ```
164
- Do NOT start ORIENT until confirmed (interactive/auto). Under --unattended, proceed.
171
+ **Resolve it** — this gate is this skill's own (the workflow never sees it), so it is this skill
172
+ that runs the same tool every other gate resolves through, not a paragraph read as a stand-in for
173
+ one: `node "${CLAUDE_PLUGIN_ROOT}/kernel/harness.mjs" gate --resolve L0 --slug <slug>
174
+ [--file <path>|--preset <name>]`. Exit 4 (`ask`) is the confirmation this block already asks for —
175
+ put it to the PO and wait, same as the paragraph above always meant. Exit 0 (`decision=proceed`) —
176
+ continue straight to ORIENT, which is what `--unattended`'s pre-answered set resolves to. Exit 5
177
+ (`abort`) — stop; do not launch. Either way, the gate's own ledger row is what lets a later reader
178
+ see the decision that opened the run, not only the decisions that closed it.
165
179
 
166
180
  ---
167
181
 
@@ -297,6 +311,12 @@ Scope contracts present:
297
311
  - scope-summary "Done when" headline statements
298
312
  - the Deferred Places from ux-behavior.md (breadboard Places this shape will not build) —
299
313
  each one needs the PO's yes; a rejected deferral goes back to the planner as a screen
314
+ - the REQ → AC table from requirements.md, one row per registered requirement: REQ-id, the
315
+ source clause it came from (`REQ-12 ← shaping.md R12`), and the acceptance criterion that
316
+ grades it — or the scope that claims it, or CUT (PO-approved). Omitted entirely when the run
317
+ has no registry. A requirement with none of the three is already a red below; this table is
318
+ what the PO reads to answer it — cover it, or cut it on the record. Printed, never asked:
319
+ the table decides nothing at this gate
300
320
  No scope contracts (pre-v0.3.0, unchanged from v0.2.6):
301
321
  Read tasks/_index.md (LOCAL root). Print:
302
322
  - task count by package/variant (.shared / .be / .web / .mobile / .e2e)
@@ -312,7 +332,15 @@ this is the orchestrator's own re-confirmation before committing to a build sequ
312
332
  waiting to happen), PA1 (directory-aligned scope), PA2 (size cap), SCOPE-ANCHOR (a scope
313
333
  naming no committed use case, or one that does not resolve), TIER-DIRECTION (a committed
314
334
  contract naming LOCAL task ids), SCOPE-DEPS (a build-order id naming a scope that is not
315
- in this run), BREADBOARD-PLACE (a breadboard Place with UI affordances has no
335
+ in this run), REQ-UNCOVERED (a requirement in requirements.md that no acceptance criterion
336
+ grades and no scope claims — the PO's two ways out are an AC carrying `(covers: REQ-…)` or
337
+ `CUT (PO-approved)` in the registry; silent on a run with no registry),
338
+ CONTRACT-SCHEMA (a scope contract that parses but not into the shape a WorkOrder carries —
339
+ most often a list written bare in a table cell where the dialect wants `[a, b]`; without
340
+ this the compiler refuses the order later and the scope is never dispatched at all, with
341
+ the board green and the leg reporting done), CONTRACT-UNREADABLE (a table the parser could
342
+ not see, or a table field also declared in frontmatter where nothing reads it),
343
+ BREADBOARD-PLACE (a breadboard Place with UI affordances has no
316
344
  `## Screen: … (P#)` in ux-behavior.md and is not deferred), BREADBOARD-UI (a U# not
317
345
  specified on a screen of its own Place). Any red → HARD STOP, past a 🔴 at the
318
346
  architect's own checkpoint. The breadboard reds are the planner's to fix — add the screen
@@ -513,12 +541,42 @@ Feature : [slug] — [SHIPPED (deployed) | BUILT & VERIFIED — deploy pending
513
541
  Rounds : [r] (build+eval cycles)
514
542
  Verdict : PASS (dims: [spec-conformance]; not evaluated: [security, performance])
515
543
  QA : [hunt done — N findings, M promoted+fixed, rest ~ | skipped (--no-qa) | n/a (pre-QA spec)]
544
+ Requirements: [15/17 PASS · 1 CUT (PO) · 1 no evidence (REQ-12 ← R12) | n/a (no registry)]
516
545
  Ledger : harness-run.md
517
546
  ```
547
+ The Requirements line is TRANSCRIBED, never composed — run
548
+ `node "${CLAUDE_PLUGIN_ROOT}/kernel/harness.mjs" probe requirements --slug <slug> --format table`
549
+ and copy its `Requirements:` summary. It joins each registered clause to the acceptance criterion
550
+ that covers it and to the criterion the judge graded, over the run named in its own output. It
551
+ decides nothing here: a requirement with no PASS evidence is a fact GATE H's census and the
552
+ baseline comparison weigh, not a ship blocker.
518
553
  Question (max 1): "Anything to record before I close the run? (y/n) or provide feedback for the next sprint."
519
554
  On confirm:
520
555
  - If the PO provides substantive feedback (not just 'y' or empty) → automatically delegate via Agent (model: exec — see references/protocol.md "Invocation mechanism"): Skill(shapeup-sdlc-plugin:coach) with the provided feedback for RLHF. The coach runs its own GATE COACH-1 to have the PO categorize each rule, then files it under the responsible skill in `shapeup/knowledge-base/<skill>.md` (committed → team-shared). Coachable: `task-executor`, `ba-pitch-analyzer`, `qa-edge-hunter`, `orient`, `scope-architect`, `solution-architect` (each reads its own file at the top of its next run) and `tech-lead` (workflow guidance, read at the next GATE L0). Guidance never decides a gate: a filed rule may add a question or a check to a gate block, never an answer. The tech lead does not categorize the feedback itself — that is the coach's gate, by design (no assumptions).
521
556
  - Then output → `✅ [slug] [shipped & deployed | built & verified, deploy pending] — [r] rounds, verdict PASS.`
522
557
 
558
+ **Resolve the gate itself before any of the above** — this is the decision that shipped the run,
559
+ and without it the trace holds no record of that decision at all: `node
560
+ "${CLAUDE_PLUGIN_ROOT}/kernel/harness.mjs" gate --resolve L4 --slug <slug>
561
+ [--file <path>|--preset <name>]`. Exit 0 (`decision=ship|hold`) — render the block above and close
562
+ the run: `node "${CLAUDE_PLUGIN_ROOT}/kernel/harness.mjs" probe resume --slug <slug> --close shipped
563
+ --cause "verdict=<verdict> rounds=<r> decision=<ship|hold>"`. Always issue this call — a `gate_h`
564
+ close is the ordinary case where `shapeup-run.js` handed off without closing the run, and the
565
+ GATE H → L4 path (scope-hammer's census, then this gate) is the one this instruction exists for.
566
+ The close itself is a once-only fact IN THE KERNEL (`closeRun`'s own guard reads a `closed_status:`
567
+ line that only `closeRun` ever writes — never the mutable `status:` line every phase rewrites, this
568
+ call included), not a conditional this instruction has to get right: if this run_id was NOT already
569
+ closed, this call performs the close, fresh. If it was already closed `shipped` (or `aborted`) and
570
+ the cause text is byte-identical to what is already on the ledger, this call is a true idempotent
571
+ no-op. If it was already closed with the SAME status but a genuinely different cause — a run closed
572
+ more than once across relaunches, the ordinary shape a `gate_h` hand-off after an earlier abort takes
573
+ — this call SUPERSEDES it: the new cause is written, the prior one is folded into the same
574
+ `close_cause` line rather than lost, and the kernel call itself still exits 0 (only the RunReturn a
575
+ launch's own `withWarnings` wraps carries the resulting `state_warning` — a prose-driven close like
576
+ this one has no RunReturn to attach it to, so read `close_cause` by hand if this branch matters to
577
+ you). If it was already closed with a DIFFERENT status altogether, this call is refused outright —
578
+ cause intact, never silently flipped. Exit 4 (`ask`) — the block above IS that stop; put it to the
579
+ PO and wait, same as always. L4's answer set carries no `abort`.
580
+
523
581
  ---
524
582
 
@@ -630,7 +630,7 @@ trace. See `references/gates.md` — GATE L0.1.
630
630
  ## Central domain registry
631
631
 
632
632
  Every record type and payload field that crosses a skill boundary is defined exactly once in
633
- `skills/tech-lead/schemas/domain.schema.json` — the envelope schemas (`work-order.schema.json`,
633
+ `kernel/schemas/domain.schema.json` — the envelope schemas (`work-order.schema.json`,
634
634
  `work-result.schema.json`) only `$ref` it. The registry annotates each entity's tier
635
635
  (SHARED/LOCAL), location, sole writer, and readers, carries the machine-readable ERD (`x-erd`),
636
636
  and maps which payload fields each worker may rely on (`x-payload-by-worker`).
@@ -686,13 +686,15 @@ lens: lite | standard | cross-context
686
686
  eval_dimensions: [spec-conformance] # the set from GATE L0.5 (init-run --dimensions); every EVAL order is compiled from THIS line
687
687
  max_rounds: 3
688
688
  auto_level: interactive | auto | unattended
689
- status: orienting | mapping | building | evaluating | shipped | escalated
689
+ status: orienting | mapping | building | evaluating | shipped | escalated | aborted
690
690
  final_verdict: ~ | pass | fail | not-evaluated
691
691
  rounds_used: [N]
692
692
  discovered_rounds: [N]
693
693
  deploy: ~ | deployed | pending-po
694
694
  started_at: [ISO]
695
695
  closed_at: ~ | [ISO]
696
+ close_cause: ~ | [why the run ended at that terminal status — `probe resume --close` writes this and closed_at together]
697
+ closed_status: ~ | [the terminal status actually closed — written ONLY by `probe resume --close`, never by `--set-status`, so it is immune to `status:` above being rewritten by ordinary phase traffic after the close]
696
698
  ---
697
699
  ```
698
700