shapeup-sdlc 3.4.0 → 3.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude-plugin/plugin.json +1 -1
- package/AGENTS.md +16 -4
- package/README.md +7 -3
- package/SECURITY.md +1 -1
- package/hooks/sandbox-guard.mjs +69 -6
- package/kernel/compile.mjs +33 -12
- package/kernel/harness.mjs +10 -4
- package/kernel/init/run-args.mjs +206 -0
- package/kernel/init/run.mjs +10 -0
- package/kernel/lib/contract.mjs +68 -1
- package/kernel/lib/paths.mjs +10 -0
- package/kernel/probe/concurrency.mjs +31 -6
- package/kernel/probe/digest.mjs +15 -1
- package/kernel/probe/owner.mjs +4 -1
- package/kernel/probe/requirements.mjs +296 -0
- package/kernel/probe/resume.mjs +195 -6
- package/kernel/probe/rounds.mjs +104 -0
- package/kernel/reduce/graph.mjs +5 -2
- package/kernel/reduce/ingest.mjs +69 -15
- package/kernel/reduce/ship.mjs +52 -31
- package/kernel/reduce/snapshot.mjs +23 -2
- package/kernel/report/export.mjs +54 -2
- package/kernel/report/facts.mjs +24 -2
- package/{skills/tech-lead → kernel}/schemas/domain.schema.json +20 -12
- package/kernel/verify/envelope.mjs +2 -2
- package/kernel/verify/skills.mjs +1 -1
- package/kernel/verify/spec.mjs +130 -4
- package/kernel/verify/trace.mjs +16 -7
- package/package.json +1 -1
- package/skills/ba-pitch-analyzer/SKILL.md +16 -1
- package/skills/coach/SKILL.md +8 -2
- package/skills/hill-chart/SKILL.md +3 -4
- package/skills/scope-architect/SKILL.md +16 -1
- package/skills/scope-hammer/SKILL.md +11 -2
- package/skills/spec-evaluator/SKILL.md +12 -1
- package/skills/tech-lead/SKILL.md +10 -10
- package/skills/tech-lead/references/gates.md +70 -12
- package/skills/tech-lead/references/protocol.md +4 -2
- package/skills/tech-lead/workflows/shapeup-run.js +176 -38
- package/skills/translator/SKILL.md +1 -1
- /package/{skills/tech-lead → kernel}/schemas/gate-answers.schema.json +0 -0
- /package/{skills/tech-lead → kernel}/schemas/work-order.schema.json +0 -0
- /package/{skills/tech-lead → kernel}/schemas/work-result.schema.json +0 -0
package/kernel/verify/trace.mjs
CHANGED
|
@@ -211,8 +211,16 @@ export function traceLint(slug, { cwd, gate = false }) {
|
|
|
211
211
|
const findings = [];
|
|
212
212
|
|
|
213
213
|
// 1. Covers-closure.
|
|
214
|
+
//
|
|
215
|
+
// THE WHOLE ARM IS GATED ON THE REGISTRY EXISTING — both halves of it, and the second half is the
|
|
216
|
+
// one that was missing. With no `requirements.md` there are no clauses, so `REQ-UNCOVERED` cannot
|
|
217
|
+
// fire; but `dangling` is derived from the BOARD, which needs no registry to carry a `covers:`
|
|
218
|
+
// clause, so a tree with no registry reported "covers-closure not applicable" in the same breath
|
|
219
|
+
// as a red finding for every `covers:` on the board. An arm that reports itself skipped and emits
|
|
220
|
+
// findings anyway is not skipped, and the report says the opposite of what the findings do.
|
|
214
221
|
const reqPath = join(shared, "requirements.md");
|
|
215
|
-
const
|
|
222
|
+
const closureChecked = existsSync(reqPath);
|
|
223
|
+
const clauses = closureChecked ? parseRequirements(readFileSync(reqPath, "utf8")) : [];
|
|
216
224
|
const board = readBoard(cwd, slug);
|
|
217
225
|
const covered = coveredReqIds(board);
|
|
218
226
|
const knownIds = new Set(clauses.map((c) => c.id));
|
|
@@ -227,12 +235,13 @@ export function traceLint(slug, { cwd, gate = false }) {
|
|
|
227
235
|
findings.push({ severity: "red", code: "REQ-UNCOVERED", req: id,
|
|
228
236
|
message: `${id} (status: covered) is named by no AC's covers: — the clause "${(c?.clause || "").slice(0, 60)}" would silently vanish. Cover it with an AC, or mark it CUT (PO-approved).` });
|
|
229
237
|
}
|
|
230
|
-
|
|
231
|
-
|
|
232
|
-
|
|
238
|
+
if (closureChecked) {
|
|
239
|
+
for (const id of dangling) {
|
|
240
|
+
findings.push({ severity: "red", code: "COVERS-DANGLING", req: id,
|
|
241
|
+
message: `an AC declares (covers: ${id}) but ${id} is not in requirements.md — a covers: link must resolve to a registered REQ.` });
|
|
242
|
+
}
|
|
233
243
|
}
|
|
234
244
|
|
|
235
|
-
const closureChecked = existsSync(reqPath);
|
|
236
245
|
const coversClosure = {
|
|
237
246
|
checked: closureChecked,
|
|
238
247
|
requirements_total: clauses.length,
|
|
@@ -240,8 +249,8 @@ export function traceLint(slug, { cwd, gate = false }) {
|
|
|
240
249
|
cut_status: cut.length,
|
|
241
250
|
covered_by_ac: [...covered].filter((id) => knownIds.has(id)).length,
|
|
242
251
|
uncovered,
|
|
243
|
-
dangling_covers: dangling,
|
|
244
|
-
pass: uncovered.length === 0 && dangling.length === 0,
|
|
252
|
+
dangling_covers: closureChecked ? dangling : [],
|
|
253
|
+
pass: uncovered.length === 0 && (!closureChecked || dangling.length === 0),
|
|
245
254
|
skipped_reason: closureChecked ? null : "no requirements.md registry — covers-closure not applicable (non-regression on pre-spine specs).",
|
|
246
255
|
};
|
|
247
256
|
|
package/package.json
CHANGED
|
@@ -98,6 +98,21 @@ its phase; templates live in `assets/templates/`.
|
|
|
98
98
|
(red). An invariant-backed regression task still anchors to its owning UC — there is no
|
|
99
99
|
second path to green.
|
|
100
100
|
|
|
101
|
+
**The requirement edge is written on the AC line, or it does not exist.** An acceptance
|
|
102
|
+
criterion that grades a registry requirement ends with `(covers: REQ-…)` — the trailing clause,
|
|
103
|
+
in the checkbox text, not a mention in prose. Measured on two runs of one pitch: every
|
|
104
|
+
requirement had an acceptance criterion somewhere on the board and only half reached a criterion
|
|
105
|
+
the judge grades, because a board AC reaches the judge through the refuted list alone — it can
|
|
106
|
+
yield a FAIL and can never yield a PASS. An AC nothing cites by id produces no evidence for the
|
|
107
|
+
requirement it was written for.
|
|
108
|
+
|
|
109
|
+
**A requirement with no natural use-case home still becomes a task.** Contrast, localisation, a
|
|
110
|
+
performance ceiling, a test surface — a non-functional clause has no actor+action and so no UC of
|
|
111
|
+
its own, and the habit is to record it in the risk register, where nothing grades it. Give it a
|
|
112
|
+
task whose AC reaches the committed spec (the invariant, the contract field or the Test Surface
|
|
113
|
+
row that states it) and carries its `(covers: REQ-…)`. A line in the risk table is a note; a
|
|
114
|
+
covered AC is a requirement the run can be measured against.
|
|
115
|
+
|
|
101
116
|
---
|
|
102
117
|
|
|
103
118
|
## The other three operations — same craft, different payload + whitelist
|
|
@@ -106,7 +121,7 @@ second path to green.
|
|
|
106
121
|
|---|---|---|
|
|
107
122
|
| `reconcile` | Verify `ledger.feature == payload.feature` (mismatch → STOP). Map each `[+]` Keep item → its owning UC; new task continues numbering (never renumber); `~`/Cut → synthesis "Hammered Out" row, no file. A Keep item asserting a new invariant → APPEND `[INV-NN]` + TS-INV row to that UC (append-only sections in your substrate). A new actor/action with no UC → `status: "escalated"` + a `deviations[]` spec-ambiguity entry: spawning a UC mid-cycle is silent re-shaping, the PO decides. Finish with board-derive (appetite overflow → report) + spec-lint | re-run phases 1–5; edit UC Steps; resolve the appetite HAMMER yourself |
|
|
108
123
|
| `retrofit-surface` | Append `## Test Surface` (derived rows only, after Error Cases) to each UC of a pre-surface spec; an all-sources-empty UC gets the explicit empty-sources line | touch anything else — append-only substrate |
|
|
109
|
-
| `coverage` | Extract **atomic** customer requirement clauses from `payload.requirements` (default: the pitch) and write the SHARED `shapeup/<slug>/requirements.md` registry: one `\| REQ-id \| clause (verbatim) \| source \| status \| note \|` row per clause. Split compound sentences into one testable clause each — a clause lost *inside* a bigger sentence is a requirement nothing can be traced to. **Assign REQ-ids ONCE and freeze them** (they behave like scope_id, never TASK-NNN — every `covers:` link rots otherwise): re-running, append new clauses with fresh ids, mark a removed clause `CUT (PO-approved)`, never renumber or delete. Status starts `covered` (a live requirement); only the PO sets `CUT`. The REQ source itself is frozen — the registry is a separate derived file | edit the REQ source; renumber existing REQ-ids; delete a dropped clause instead of marking it CUT; invent a requirement not in the source |
|
|
124
|
+
| `coverage` | Extract **atomic** customer requirement clauses from `payload.requirements` (default: the pitch) and write the SHARED `shapeup/<slug>/requirements.md` registry: one `\| REQ-id \| clause (verbatim) \| source \| status \| note \|` row per clause. Split compound sentences into one testable clause each — a clause lost *inside* a bigger sentence is a requirement nothing can be traced to. **Assign REQ-ids ONCE and freeze them** (they behave like scope_id, never TASK-NNN — every `covers:` link rots otherwise): re-running, append new clauses with fresh ids, mark a removed clause `CUT (PO-approved)`, never renumber or delete. Status starts `covered` (a live requirement); only the PO sets `CUT`. The REQ source itself is frozen — the registry is a separate derived file. **Numbering.** A source clause already carrying an `R<n>` keeps its number — `R12` → `REQ-12` — and its `source` cell records where it came from verbatim (`shaping.md R12`), because that cell is the only thing that survives a re-run. A clause with no R-id takes the next free number ABOVE the highest `R<n>` in the source, so it can never collide with one added later. Splitting a compound clause keeps `REQ-12` for the first atomic part and records `shaping.md R12 (split 2/3)` for the rest — a requirement graded in parts is why splitting matters at all. On a re-run, match an existing id by its frozen `source` cell and clause text, **never** by re-deriving the number from the source's current order | edit the REQ source; renumber existing REQ-ids; delete a dropped clause instead of marking it CUT; invent a requirement not in the source; re-point an existing REQ-id because the source's R-numbers shifted |
|
|
110
125
|
---
|
|
111
126
|
|
|
112
127
|
## Anti-rationalization table
|
package/skills/coach/SKILL.md
CHANGED
|
@@ -93,7 +93,7 @@ never lands in any worker's KB.
|
|
|
93
93
|
Orchestrated, this skill is dispatched like every worker: a **WorkOrder** in (`--order <path>`,
|
|
94
94
|
operation `coach` or `scan`), a **WorkResult** out. Standalone, the raw feedback is passed
|
|
95
95
|
directly; it maps onto the one payload field registered for this worker in the central domain
|
|
96
|
-
registry (`
|
|
96
|
+
registry (`kernel/schemas/domain.schema.json`, `x-payload-by-worker`):
|
|
97
97
|
|
|
98
98
|
| Payload field | Standalone form | Meaning |
|
|
99
99
|
|---|---|---|
|
|
@@ -117,7 +117,13 @@ fields nobody used" → "Prefer the minimum DTO that satisfies the AC; don't add
|
|
|
117
117
|
fields"). Keep the originating why — a rule without its reason gets ignored or misapplied.
|
|
118
118
|
|
|
119
119
|
### Step 2 — ⏸ GATE COACH-1: Categorize (ASK, never assume)
|
|
120
|
-
This is the load-bearing gate. **
|
|
120
|
+
This is the load-bearing gate. **Resolve it first** — `node
|
|
121
|
+
"${CLAUDE_PLUGIN_ROOT}/kernel/harness.mjs" gate --resolve COACH-1 --slug <slug>
|
|
122
|
+
[--file <path>|--preset <name>]` — so the ledger carries a row for the decision this gate makes,
|
|
123
|
+
same as every other gate in the run. Exit 0 (`decision=skip`) — an unattended lane with no live PO;
|
|
124
|
+
record nothing and stop here, the same outcome the CI preset's own note already documents. Exit 4
|
|
125
|
+
(`ask`) — proceed with the categorization below, which IS the PO conversation this decision opens.
|
|
126
|
+
**Do not infer which skill a rule belongs to** — a
|
|
121
127
|
miscategorized rule lands in a file the wrong worker reads (or no worker reads). Present every
|
|
122
128
|
candidate rule and ask the PO to assign each one. Emit this block, then stop and wait:
|
|
123
129
|
|
|
@@ -10,7 +10,7 @@ re-runs a computation that would erase true history.**
|
|
|
10
10
|
|
|
11
11
|
You are not a worker: no WorkOrder, no WorkResult, invoked directly by the user (or by `/hill`)
|
|
12
12
|
exactly like `shapeup` is. There is nothing to declare in
|
|
13
|
-
`
|
|
13
|
+
`kernel/schemas/domain.schema.json` and nothing to teach `harness compile` or
|
|
14
14
|
`harness reduce ingest` — those steps exist only for dispatched workers.
|
|
15
15
|
|
|
16
16
|
## What you read
|
|
@@ -71,9 +71,8 @@ Build one `{ scope_id, phase }` object per file.
|
|
|
71
71
|
## Rendering — the injection contract
|
|
72
72
|
|
|
73
73
|
The engine ships at `assets/dashboard.template.html` — a complete, self-contained HTML page
|
|
74
|
-
(inline CSS/JS, no external fetch beyond Google Fonts, no build step
|
|
75
|
-
|
|
76
|
-
fill in real data, and write the result.
|
|
74
|
+
(inline CSS/JS, no external fetch beyond Google Fonts, no build step). Do not rewrite it from a
|
|
75
|
+
text description; read it, fill in real data, and write the result.
|
|
77
76
|
|
|
78
77
|
1. For each discovered slug, build one entry:
|
|
79
78
|
|
|
@@ -41,6 +41,16 @@ the ship report's census table.
|
|
|
41
41
|
scalars and [a, b] lists, a `## Affordances` table for affordance_manifest, and a
|
|
42
42
|
short `## Why this slice` paragraph. A reviewer must be able to read the substrate
|
|
43
43
|
in a PR; regeneration preserves prose under headings you do not own.
|
|
44
|
+
► affordance_manifest lives in the TABLE and NOWHERE ELSE. Do not also write it in
|
|
45
|
+
the frontmatter: nothing reads it there, so the copy is discarded — and a run has
|
|
46
|
+
been lost to exactly that. The frontmatter copy said required_states: [idle], the
|
|
47
|
+
table cell said a bare idle, the table won, the value was a string where an array
|
|
48
|
+
was required, and four scopes were never dispatched — the board green, the contract
|
|
49
|
+
lint-clean, each leg reporting done with no error, every round, until EVAL refused
|
|
50
|
+
to grade a round whose scopes had never run. A contract declaring no affordances
|
|
51
|
+
writes `affordance_manifest: []` in frontmatter and no table.
|
|
52
|
+
► A LIST INSIDE A TABLE CELL IS WRITTEN `[a, b]`, brackets and all — `[idle]`, never
|
|
53
|
+
`idle`. A bare word in that cell is a string, and required_states is an array.
|
|
44
54
|
scope_id, topology_type — the stable join key is the scope
|
|
45
55
|
use_cases[] — the UC ids this scope implements.
|
|
46
56
|
THE ONLY LINK YOU WRITE TO THE
|
|
@@ -53,7 +63,12 @@ the ship report's census table.
|
|
|
53
63
|
the board's own use_case_refs
|
|
54
64
|
covers[] — optional REQ-ids from
|
|
55
65
|
requirements.md this scope answers
|
|
56
|
-
for; stable, never renumbered
|
|
66
|
+
for; stable, never renumbered.
|
|
67
|
+
WRITE THE REGISTRY'S OWN KEY:
|
|
68
|
+
`REQ-12`, not the pitch's `R12`.
|
|
69
|
+
Both resolve — readers normalise —
|
|
70
|
+
but one spelling in the committed
|
|
71
|
+
contract is one thing to read
|
|
57
72
|
depends_on[] — scope_ids this scope builds AFTER.
|
|
58
73
|
This is the build ORDER — declare
|
|
59
74
|
it whenever one scope consumes
|
|
@@ -72,6 +72,14 @@ H0.1 Unresolved scopes (breaker cases only):
|
|
|
72
72
|
H0.2 QA findings (qa-edge-hunter's hunt-report.md, when present) — all `~` by default.
|
|
73
73
|
H0.3 Discovered-task ledger entries still open (discovery/ledger.md, `[+]`/`~` unresolved).
|
|
74
74
|
H0.4 Attempt-budget hammer proposals (scopes that exhausted their T0 attempts during BUILD).
|
|
75
|
+
H0.4b Requirements with no PASS evidence — the pitch clauses the run never showed working. Run
|
|
76
|
+
node "${CLAUDE_PLUGIN_ROOT}/kernel/harness.mjs" probe requirements --slug <slug> --format table
|
|
77
|
+
and take its `no evidence` rows; cite the row, the same way H0.0 cites ownership. Each is a
|
|
78
|
+
census item carrying its source clause (`REQ-12 ← shaping.md R12`). A `cut` row is an answer
|
|
79
|
+
the PO already gave — not an item. An inconsistency row (a criterion anchored to a
|
|
80
|
+
requirement no acceptance criterion covers) is reported to the PO as a reconciliation, never
|
|
81
|
+
counted as evidence and never promoted as a finding. No registry on disk → this input is
|
|
82
|
+
empty and the census is unchanged (absent artifact ⇒ arm skipped).
|
|
75
83
|
H0.5 Classify every item: MUST-HAVE (the pitch's core problem is unsolved without it) vs
|
|
76
84
|
NICE-TO-HAVE (`~`, improves but doesn't block the core promise). Default to NICE-TO-HAVE
|
|
77
85
|
unless the item traces directly to a pitch boundary or a scope's business_goal — a
|
|
@@ -81,7 +89,8 @@ H0.5 Classify every item: MUST-HAVE (the pitch's core problem is unsolved witho
|
|
|
81
89
|
**GATE H0 Output:**
|
|
82
90
|
```
|
|
83
91
|
⏸ GATE H0 — Census
|
|
84
|
-
Must-have (unresolved) : [N] — [list, each with source: scope | QA | discovered |
|
|
92
|
+
Must-have (unresolved) : [N] — [list, each with source: scope | QA | discovered | requirement |
|
|
93
|
+
advisor-overflow]
|
|
85
94
|
Nice-to-have (~) : [M]
|
|
86
95
|
Carry candidates : [scopes still uphill/downhill, or exhausted attempt budget]
|
|
87
96
|
```
|
|
@@ -141,7 +150,7 @@ the harness (this is neither the generator nor the evaluator).
|
|
|
141
150
|
Orchestrated, this skill is dispatched like every worker: a **WorkOrder** in (`--order <path>`,
|
|
142
151
|
operation `hammer`), a **WorkResult** out. The standalone flags below map 1:1 onto the payload
|
|
143
152
|
fields registered for this worker in the central domain registry
|
|
144
|
-
(`
|
|
153
|
+
(`kernel/schemas/domain.schema.json`, `x-payload-by-worker`):
|
|
145
154
|
|
|
146
155
|
| Payload field | Standalone flag | Meaning |
|
|
147
156
|
|---|---|---|
|
|
@@ -152,13 +152,23 @@ round. Write it so someone without your context can answer it in one reply.
|
|
|
152
152
|
"t0_citations": [ { "scope_id": "cart", "path": "…/t0/verdicts/r2-a3.json", "sha256": "…" } ],
|
|
153
153
|
"criteria": [ { "criterion": "UC-01 step 3", "dimension": "spec-conformance",
|
|
154
154
|
"verdict": "FAIL", "confidence": "high", "reprobed": true,
|
|
155
|
-
"evidence": "Pay click throws — apps/web/checkout/Pay.tsx:84"
|
|
155
|
+
"evidence": "Pay click throws — apps/web/checkout/Pay.tsx:84",
|
|
156
|
+
"traces_to": ["REQ-4"] } ],
|
|
156
157
|
"refuted": [ { "task_id": "TASK-007", "ac": "<the checkbox text your evidence disproves>" } ],
|
|
157
158
|
"bugs": [ /* report-schema bug entries */ ]
|
|
158
159
|
}
|
|
159
160
|
}
|
|
160
161
|
```
|
|
161
162
|
|
|
163
|
+
**`traces_to` is copied, not invented.** Fill it from the `(covers: REQ-…)` clause of the
|
|
164
|
+
acceptance criteria your criterion grades: the AC already carries the link, written when the plan
|
|
165
|
+
was reviewed, and you record which requirement your criterion maps back to. An AC with no `covers:`
|
|
166
|
+
clause yields no anchor — leave the array empty rather than guessing, and never read the pitch to
|
|
167
|
+
supply one. This changes nothing you grade: the anchor is a navigation path, never a grading input,
|
|
168
|
+
and a criterion passes or fails on its evidence exactly as before. It matters downstream because
|
|
169
|
+
the requirement matrix at GATE L4 and the census at GATE H are projected from these anchors; a
|
|
170
|
+
verdict that drops them grades the build and says nothing about what the pitch asked for.
|
|
171
|
+
|
|
162
172
|
**Every FAIL criterion's `evidence` MUST carry a `file:line` locator** — schema-enforced, not
|
|
163
173
|
advice: the envelope is validated against `work-result.schema.json` at ingest and a locatorless
|
|
164
174
|
FAIL is rejected before any write. A PASS may cite plain output. (Observed, not theorized: a
|
|
@@ -175,6 +185,7 @@ separation is the whole point of the architecture.
|
|
|
175
185
|
## Verification checklist
|
|
176
186
|
|
|
177
187
|
- [ ] Every criterion traces to committed spec text (UC/domain-model/contract/Done-when/Non-Go)
|
|
188
|
+
- [ ] `traces_to` copied from the graded ACs' `covers:` clauses — empty where they carry none
|
|
178
189
|
- [ ] Every PASS cites a confirming probe; every FAIL cites evidence or "NO EVIDENCE"
|
|
179
190
|
- [ ] Every FAIL was re-probed once; confidence assigned per the ledger rule
|
|
180
191
|
- [ ] Scoped spec → T0 citations present with recomputed sha256 (else the run returned `failed`)
|
|
@@ -60,15 +60,15 @@ check the lane:
|
|
|
60
60
|
legacy loop instead — `references/protocol.md` (BUILD(r)/EVAL) + `references/protocol.md`
|
|
61
61
|
carry the full step-by-step for both the tiny lane and a scope-less BUILD loop, verbatim, non-
|
|
62
62
|
regression. Stop reading this file here for that run.
|
|
63
|
-
- **Otherwise** (the common case — a scoped spec, any auto level):
|
|
64
|
-
(`
|
|
65
|
-
|
|
66
|
-
|
|
67
|
-
|
|
68
|
-
`.shapeup/<slug>/run-args.json`
|
|
69
|
-
|
|
70
|
-
|
|
71
|
-
|
|
63
|
+
- **Otherwise** (the common case — a scoped spec, any auto level): resolve every switch the
|
|
64
|
+
operator typed (`references/gates.md` GATE L0.9b has the flag→field table) and run the kernel's
|
|
65
|
+
sole `RunArgs` writer: `node "${CLAUDE_PLUGIN_ROOT}/kernel/harness.mjs" init run-args --slug
|
|
66
|
+
<slug> --auto-level <level> --exec-model <n> [--eval-model <n>] [--qa-model <n>] --max-rounds <N>
|
|
67
|
+
--attempts <N> --plugin-root "${CLAUDE_PLUGIN_ROOT}" [--answers <a>] [--lane <l>] [--no-eval]
|
|
68
|
+
[--no-qa] [--adversarial-verify] [--parallel-scopes <N>]`. It writes `.shapeup/<slug>/run-args.json`
|
|
69
|
+
fresh on every launch/relaunch and prints that identical object — **pass it to `Workflow`
|
|
70
|
+
verbatim, never re-type it**. Then launch with the **`Workflow` tool** — naming `init run`'s
|
|
71
|
+
staged copy, never the install path:
|
|
72
72
|
|
|
73
73
|
```
|
|
74
74
|
Workflow({
|
|
@@ -114,7 +114,7 @@ did not actually receive from the PO — an unattended lane with no answer for a
|
|
|
114
114
|
|
|
115
115
|
FIRST freeze the evidence — run state is gitignored, so `shapeup/<slug>/REPORT.md` (already
|
|
116
116
|
written by `shapeup-run.js` via `harness reduce ship`, or write it now on a `gate_h` close) is all a
|
|
117
|
-
teammate sees. Then emit:
|
|
117
|
+
teammate sees. Then RESOLVE the gate — `references/gates.md` GATE L4 has the call — and emit:
|
|
118
118
|
|
|
119
119
|
```
|
|
120
120
|
⏸ GATE L4 — Ship Sign-Off
|
|
@@ -113,11 +113,13 @@ Collect (explicit — never inferred):
|
|
|
113
113
|
gate to be its first execution.
|
|
114
114
|
```
|
|
115
115
|
|
|
116
|
-
**L0.9b — the launch record.** Every switch the operator typed becomes a `RunArgs`
|
|
117
|
-
does nothing at all: the workflow cannot read a config file and cannot ask a follow-up,
|
|
118
|
-
that stops at the skill boundary was accepted and ignored. That is not hypothetical —
|
|
119
|
-
documented in seven places across the shipped set and inert in all of them, because no
|
|
120
|
-
protocol ever put `noQa` into the record.
|
|
116
|
+
**L0.9b — the launch record.** Every switch the operator typed to *this launch* becomes a `RunArgs`
|
|
117
|
+
field, or it does nothing at all: the workflow cannot read a config file and cannot ask a follow-up,
|
|
118
|
+
so a flag that stops at the skill boundary was accepted and ignored. That is not hypothetical —
|
|
119
|
+
`--no-qa` was documented in seven places across the shipped set and inert in all of them, because no
|
|
120
|
+
line of this protocol ever put `noQa` into the record. `--wall-clock-budget` is the one flag below
|
|
121
|
+
that is not a `RunArgs` field at all — it is consumed earlier, at `init run` itself, and never
|
|
122
|
+
needed to reach this launch; see its row for where it actually lands.
|
|
121
123
|
|
|
122
124
|
| Flag | `RunArgs` field |
|
|
123
125
|
|---|---|
|
|
@@ -125,14 +127,19 @@ protocol ever put `noQa` into the record.
|
|
|
125
127
|
| `--no-qa` | `noQa: true` |
|
|
126
128
|
| `--parallel-scopes N` | `maxParallelScopes: N` — how many scopes build at once (default 4; `1` = sequential) |
|
|
127
129
|
| `--adversarial-verify` | `adversarialVerify: true` |
|
|
128
|
-
| `--rounds N` / `--attempts N`
|
|
130
|
+
| `--rounds N` / `--attempts N` | `budgets.{maxRounds,attemptBudget}` |
|
|
129
131
|
| `--gate-answers <set>` | `answers` |
|
|
132
|
+
| `--wall-clock-budget S` | *(not a `RunArgs` field)* — typed once, on the `harness init run` command line itself, not on this launch; it lands straight in the run receipt as `wall_clock_budget_s`, and the deadline breaker reads that receipt field directly — consumed by `kernel/verify/budget.mjs` as `wall_clock_budget_s`. `budgets` declares only `maxRounds`/`attemptBudget` — the schema, `SKILL.md`'s own RunArgs contract line and this script's own header comment all agree there is no third member |
|
|
130
133
|
| `--orch-model/--exec-model/--eval-model/--qa-model` | `models.{…}` (L0.8) |
|
|
131
134
|
|
|
132
|
-
|
|
133
|
-
|
|
134
|
-
|
|
135
|
-
|
|
135
|
+
`harness init run-args` (invoked at Step 2 of `SKILL.md`) is the sole writer of the assembled
|
|
136
|
+
object: it takes the resolved values above, writes `.shapeup/<slug>/run-args.json` fresh on every
|
|
137
|
+
launch and relaunch, and prints the same object back so the launch never re-assembles it by hand. It
|
|
138
|
+
is the only artifact that records what a run was configured with; the ship report, a resumed session
|
|
139
|
+
and any later measurement all read it, and none of them can recover a value that only ever existed
|
|
140
|
+
as an argument. Step 2 is not merely advisory: `shapeup-run.js`'s own Preflight refuses to dispatch
|
|
141
|
+
ORIENT (or anything past it) when this file is missing at the run's local root — a launch that
|
|
142
|
+
skipped this step aborts there rather than proceeding on a silent default.
|
|
136
143
|
|
|
137
144
|
**L0.0 — intake precondition (before any other L0 collection):**
|
|
138
145
|
```
|
|
@@ -161,7 +168,14 @@ Model matrix : orch=[model] exec=[model] eval=[model] qa=[model] digester=[scrip
|
|
|
161
168
|
Budgets : round_budget=[N] (outer) attempt_budget=[N] (inner, per scope)
|
|
162
169
|
Knowledge : [tech-lead.md — N workflow rules, M suggested values (confirmed above) | none — `/retro --scan` or `/retro --research <stack>` seeds it (optional)]
|
|
163
170
|
```
|
|
164
|
-
|
|
171
|
+
**Resolve it** — this gate is this skill's own (the workflow never sees it), so it is this skill
|
|
172
|
+
that runs the same tool every other gate resolves through, not a paragraph read as a stand-in for
|
|
173
|
+
one: `node "${CLAUDE_PLUGIN_ROOT}/kernel/harness.mjs" gate --resolve L0 --slug <slug>
|
|
174
|
+
[--file <path>|--preset <name>]`. Exit 4 (`ask`) is the confirmation this block already asks for —
|
|
175
|
+
put it to the PO and wait, same as the paragraph above always meant. Exit 0 (`decision=proceed`) —
|
|
176
|
+
continue straight to ORIENT, which is what `--unattended`'s pre-answered set resolves to. Exit 5
|
|
177
|
+
(`abort`) — stop; do not launch. Either way, the gate's own ledger row is what lets a later reader
|
|
178
|
+
see the decision that opened the run, not only the decisions that closed it.
|
|
165
179
|
|
|
166
180
|
---
|
|
167
181
|
|
|
@@ -297,6 +311,12 @@ Scope contracts present:
|
|
|
297
311
|
- scope-summary "Done when" headline statements
|
|
298
312
|
- the Deferred Places from ux-behavior.md (breadboard Places this shape will not build) —
|
|
299
313
|
each one needs the PO's yes; a rejected deferral goes back to the planner as a screen
|
|
314
|
+
- the REQ → AC table from requirements.md, one row per registered requirement: REQ-id, the
|
|
315
|
+
source clause it came from (`REQ-12 ← shaping.md R12`), and the acceptance criterion that
|
|
316
|
+
grades it — or the scope that claims it, or CUT (PO-approved). Omitted entirely when the run
|
|
317
|
+
has no registry. A requirement with none of the three is already a red below; this table is
|
|
318
|
+
what the PO reads to answer it — cover it, or cut it on the record. Printed, never asked:
|
|
319
|
+
the table decides nothing at this gate
|
|
300
320
|
No scope contracts (pre-v0.3.0, unchanged from v0.2.6):
|
|
301
321
|
Read tasks/_index.md (LOCAL root). Print:
|
|
302
322
|
- task count by package/variant (.shared / .be / .web / .mobile / .e2e)
|
|
@@ -312,7 +332,15 @@ this is the orchestrator's own re-confirmation before committing to a build sequ
|
|
|
312
332
|
waiting to happen), PA1 (directory-aligned scope), PA2 (size cap), SCOPE-ANCHOR (a scope
|
|
313
333
|
naming no committed use case, or one that does not resolve), TIER-DIRECTION (a committed
|
|
314
334
|
contract naming LOCAL task ids), SCOPE-DEPS (a build-order id naming a scope that is not
|
|
315
|
-
in this run),
|
|
335
|
+
in this run), REQ-UNCOVERED (a requirement in requirements.md that no acceptance criterion
|
|
336
|
+
grades and no scope claims — the PO's two ways out are an AC carrying `(covers: REQ-…)` or
|
|
337
|
+
`CUT (PO-approved)` in the registry; silent on a run with no registry),
|
|
338
|
+
CONTRACT-SCHEMA (a scope contract that parses but not into the shape a WorkOrder carries —
|
|
339
|
+
most often a list written bare in a table cell where the dialect wants `[a, b]`; without
|
|
340
|
+
this the compiler refuses the order later and the scope is never dispatched at all, with
|
|
341
|
+
the board green and the leg reporting done), CONTRACT-UNREADABLE (a table the parser could
|
|
342
|
+
not see, or a table field also declared in frontmatter where nothing reads it),
|
|
343
|
+
BREADBOARD-PLACE (a breadboard Place with UI affordances has no
|
|
316
344
|
`## Screen: … (P#)` in ux-behavior.md and is not deferred), BREADBOARD-UI (a U# not
|
|
317
345
|
specified on a screen of its own Place). Any red → HARD STOP, past a 🔴 at the
|
|
318
346
|
architect's own checkpoint. The breadboard reds are the planner's to fix — add the screen
|
|
@@ -513,12 +541,42 @@ Feature : [slug] — [SHIPPED (deployed) | BUILT & VERIFIED — deploy pending
|
|
|
513
541
|
Rounds : [r] (build+eval cycles)
|
|
514
542
|
Verdict : PASS (dims: [spec-conformance]; not evaluated: [security, performance])
|
|
515
543
|
QA : [hunt done — N findings, M promoted+fixed, rest ~ | skipped (--no-qa) | n/a (pre-QA spec)]
|
|
544
|
+
Requirements: [15/17 PASS · 1 CUT (PO) · 1 no evidence (REQ-12 ← R12) | n/a (no registry)]
|
|
516
545
|
Ledger : harness-run.md
|
|
517
546
|
```
|
|
547
|
+
The Requirements line is TRANSCRIBED, never composed — run
|
|
548
|
+
`node "${CLAUDE_PLUGIN_ROOT}/kernel/harness.mjs" probe requirements --slug <slug> --format table`
|
|
549
|
+
and copy its `Requirements:` summary. It joins each registered clause to the acceptance criterion
|
|
550
|
+
that covers it and to the criterion the judge graded, over the run named in its own output. It
|
|
551
|
+
decides nothing here: a requirement with no PASS evidence is a fact GATE H's census and the
|
|
552
|
+
baseline comparison weigh, not a ship blocker.
|
|
518
553
|
Question (max 1): "Anything to record before I close the run? (y/n) or provide feedback for the next sprint."
|
|
519
554
|
On confirm:
|
|
520
555
|
- If the PO provides substantive feedback (not just 'y' or empty) → automatically delegate via Agent (model: exec — see references/protocol.md "Invocation mechanism"): Skill(shapeup-sdlc-plugin:coach) with the provided feedback for RLHF. The coach runs its own GATE COACH-1 to have the PO categorize each rule, then files it under the responsible skill in `shapeup/knowledge-base/<skill>.md` (committed → team-shared). Coachable: `task-executor`, `ba-pitch-analyzer`, `qa-edge-hunter`, `orient`, `scope-architect`, `solution-architect` (each reads its own file at the top of its next run) and `tech-lead` (workflow guidance, read at the next GATE L0). Guidance never decides a gate: a filed rule may add a question or a check to a gate block, never an answer. The tech lead does not categorize the feedback itself — that is the coach's gate, by design (no assumptions).
|
|
521
556
|
- Then output → `✅ [slug] [shipped & deployed | built & verified, deploy pending] — [r] rounds, verdict PASS.`
|
|
522
557
|
|
|
558
|
+
**Resolve the gate itself before any of the above** — this is the decision that shipped the run,
|
|
559
|
+
and without it the trace holds no record of that decision at all: `node
|
|
560
|
+
"${CLAUDE_PLUGIN_ROOT}/kernel/harness.mjs" gate --resolve L4 --slug <slug>
|
|
561
|
+
[--file <path>|--preset <name>]`. Exit 0 (`decision=ship|hold`) — render the block above and close
|
|
562
|
+
the run: `node "${CLAUDE_PLUGIN_ROOT}/kernel/harness.mjs" probe resume --slug <slug> --close shipped
|
|
563
|
+
--cause "verdict=<verdict> rounds=<r> decision=<ship|hold>"`. Always issue this call — a `gate_h`
|
|
564
|
+
close is the ordinary case where `shapeup-run.js` handed off without closing the run, and the
|
|
565
|
+
GATE H → L4 path (scope-hammer's census, then this gate) is the one this instruction exists for.
|
|
566
|
+
The close itself is a once-only fact IN THE KERNEL (`closeRun`'s own guard reads a `closed_status:`
|
|
567
|
+
line that only `closeRun` ever writes — never the mutable `status:` line every phase rewrites, this
|
|
568
|
+
call included), not a conditional this instruction has to get right: if this run_id was NOT already
|
|
569
|
+
closed, this call performs the close, fresh. If it was already closed `shipped` (or `aborted`) and
|
|
570
|
+
the cause text is byte-identical to what is already on the ledger, this call is a true idempotent
|
|
571
|
+
no-op. If it was already closed with the SAME status but a genuinely different cause — a run closed
|
|
572
|
+
more than once across relaunches, the ordinary shape a `gate_h` hand-off after an earlier abort takes
|
|
573
|
+
— this call SUPERSEDES it: the new cause is written, the prior one is folded into the same
|
|
574
|
+
`close_cause` line rather than lost, and the kernel call itself still exits 0 (only the RunReturn a
|
|
575
|
+
launch's own `withWarnings` wraps carries the resulting `state_warning` — a prose-driven close like
|
|
576
|
+
this one has no RunReturn to attach it to, so read `close_cause` by hand if this branch matters to
|
|
577
|
+
you). If it was already closed with a DIFFERENT status altogether, this call is refused outright —
|
|
578
|
+
cause intact, never silently flipped. Exit 4 (`ask`) — the block above IS that stop; put it to the
|
|
579
|
+
PO and wait, same as always. L4's answer set carries no `abort`.
|
|
580
|
+
|
|
523
581
|
---
|
|
524
582
|
|
|
@@ -630,7 +630,7 @@ trace. See `references/gates.md` — GATE L0.1.
|
|
|
630
630
|
## Central domain registry
|
|
631
631
|
|
|
632
632
|
Every record type and payload field that crosses a skill boundary is defined exactly once in
|
|
633
|
-
`
|
|
633
|
+
`kernel/schemas/domain.schema.json` — the envelope schemas (`work-order.schema.json`,
|
|
634
634
|
`work-result.schema.json`) only `$ref` it. The registry annotates each entity's tier
|
|
635
635
|
(SHARED/LOCAL), location, sole writer, and readers, carries the machine-readable ERD (`x-erd`),
|
|
636
636
|
and maps which payload fields each worker may rely on (`x-payload-by-worker`).
|
|
@@ -686,13 +686,15 @@ lens: lite | standard | cross-context
|
|
|
686
686
|
eval_dimensions: [spec-conformance] # the set from GATE L0.5 (init-run --dimensions); every EVAL order is compiled from THIS line
|
|
687
687
|
max_rounds: 3
|
|
688
688
|
auto_level: interactive | auto | unattended
|
|
689
|
-
status: orienting | mapping | building | evaluating | shipped | escalated
|
|
689
|
+
status: orienting | mapping | building | evaluating | shipped | escalated | aborted
|
|
690
690
|
final_verdict: ~ | pass | fail | not-evaluated
|
|
691
691
|
rounds_used: [N]
|
|
692
692
|
discovered_rounds: [N]
|
|
693
693
|
deploy: ~ | deployed | pending-po
|
|
694
694
|
started_at: [ISO]
|
|
695
695
|
closed_at: ~ | [ISO]
|
|
696
|
+
close_cause: ~ | [why the run ended at that terminal status — `probe resume --close` writes this and closed_at together]
|
|
697
|
+
closed_status: ~ | [the terminal status actually closed — written ONLY by `probe resume --close`, never by `--set-status`, so it is immune to `status:` above being rewritten by ordinary phase traffic after the close]
|
|
696
698
|
---
|
|
697
699
|
```
|
|
698
700
|
|