shapeup-sdlc 3.15.0 → 3.16.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude-plugin/plugin.json +1 -1
- package/AGENTS.md +1 -1
- package/kernel/probe/eval.mjs +52 -2
- package/kernel/reduce/ingest.mjs +10 -7
- package/kernel/reduce/ship.mjs +16 -1
- package/package.json +1 -1
- package/skills/qa-edge-hunter/SKILL.md +5 -1
- package/skills/spec-evaluator/SKILL.md +6 -0
- package/skills/tech-lead/references/protocol.md +3 -2
- package/skills/tech-lead/workflows/shapeup-run.js +14 -4
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "shapeup-sdlc-plugin",
|
|
3
3
|
"displayName": "ShapeUp SDLC Plugin",
|
|
4
|
-
"version": "3.
|
|
4
|
+
"version": "3.16.0",
|
|
5
5
|
"description": "Shape Up SDLC harness for Claude Code: shaping, intake, orient, scope-mapping, building (T0-verified, sandboxed, scope-contracted), evaluation and QA skills orchestrated by a tech-lead.",
|
|
6
6
|
"author": {
|
|
7
7
|
"name": "Liberty Nguyen",
|
package/AGENTS.md
CHANGED
|
@@ -36,7 +36,7 @@ Betting Table: PO decides; rejected pitches loop back to raw idea.
|
|
|
36
36
|
| Wire | ⏸ **L1a.5** — Wiring Review ✚ | `/solution-architect` (`wire`): sole writer of committed `wiring-map.md` — per-UC engine → seam → entry-point call site → affordance, per `project-profile.md` |
|
|
37
37
|
| Map Scopes | ⏸ **L1b** — Board Review (+ substrate disjointness lint) | `/scope-architect` (scope contracts ✦ — sole writer); traceability oracle advisory ✚. A registered requirement that no acceptance criterion grades and no scope claims is **red** here, and L1b prints the `REQ → AC` table: the two ways out are an AC carrying `(covers: REQ-…)` or the PO marking the clause `CUT (PO-approved)`. Red only where the plan is still cheap to change — after L1b nobody re-reads the pitch |
|
|
38
38
|
| Build Vertically | ⏸ **L2** — Board 100% ✅ + T0-green ✦ | per dispatch: compile order → `/task-executor` (--order) → ingest result; T0-verified per attempt (fixtures + DB probe ✦), substrate-sandboxed ✦. Scopes build **concurrently** ✦ — `--parallel-scopes N` caps it (default 4), a scope is released the moment its own dependencies are green, and a scope green in this round is skipped rather than rebuilt. Then the **round build gate** ⚙: the ledger's run command, then the profile's `build_probe` and `launch_probe`, run once per round before EVAL — a red gate ends the round with no verdict and its failing step is compiled into the next round's orders as bugs; a `mobile` profile with no `launch_probe` is warned about every round, so the install/launch risk has an owner |
|
|
39
|
-
| EVAL (once per round) | ⏸ **L3** — Verdict | `/spec-evaluator` (--order), only over a round whose build gate ⚙ is not red: spec- + test-surface-conformance ★, T0 citation ✦; refuted boxes/verdict applied by ingest. The order names the round's build gate artifact and the profile's `launch_probe`, so a `[ui]` row is graded on the app the gate launched rather than recorded as no evidence — a project with no `launch_probe` still gets no on-device grade |
|
|
39
|
+
| EVAL (once per round) | ⏸ **L3** — Verdict | `/spec-evaluator` (--order), only over a round whose build gate ⚙ is not red: spec- + test-surface-conformance ★, T0 citation ✦; refuted boxes/verdict applied by ingest. The order names the round's build gate artifact and the profile's `launch_probe`, so a `[ui]` row is graded on the app the gate launched rather than recorded as no evidence — a project with no `launch_probe` still gets no on-device grade. A PASS names every Test Surface row as its own criterion: one that grades the surface as a group is refused, and the judge is sent back once with the reason |
|
|
40
40
|
| FAIL → round r+1 | — | regression rule ★: bugs + full Test Surface of touched UC. Every criterion the verdict graded FAIL reaches the round, whether or not the judge filed a bug for it, addressed to the scope that owns its use case — a round with a FAIL verdict and nothing to fix is how a run stalls. A row whose check the scope rewrote between a failing trial and a passing one is listed for the judge, who reads it against the row before the pass counts |
|
|
41
41
|
|
|
42
42
|
✦ = requires scope contracts (`shapeup/<slug>/scopes/*.md`); ✚ = requires the spine artifacts (`requirements.md`, `wiring-map.md`, `project-profile.md`). Traceability stays advisory until `covers:` is populated. Absent artifact ⇒ arm skipped (non-regression).
|
package/kernel/probe/eval.mjs
CHANGED
|
@@ -35,7 +35,7 @@ import { existsSync, readFileSync, readdirSync, realpathSync } from "node:fs";
|
|
|
35
35
|
import { join, resolve, resolve as resolvePath, sep } from "node:path";
|
|
36
36
|
import { createHash } from "node:crypto";
|
|
37
37
|
import { runArgs } from "../lib/argv.mjs";
|
|
38
|
-
import { resultsDir, scopesDir, readRunId, verdictsDir } from "../lib/paths.mjs";
|
|
38
|
+
import { resultsDir, scopesDir, readRunId, verdictsDir, specDir } from "../lib/paths.mjs";
|
|
39
39
|
|
|
40
40
|
/** Longest `reason` reported. A deviation is prose written by a worker and can run to paragraphs. */
|
|
41
41
|
const REASON_MAX = 400;
|
|
@@ -203,6 +203,56 @@ export function verdictProblem(verdict) {
|
|
|
203
203
|
return null;
|
|
204
204
|
}
|
|
205
205
|
|
|
206
|
+
/**
|
|
207
|
+
* The Test Surface row ids the spec declares, in spec order.
|
|
208
|
+
*
|
|
209
|
+
* @param {string} cwd - Project root.
|
|
210
|
+
* @param {string} slug - Feature slug.
|
|
211
|
+
* @returns {string[]} Every `TS-…` id that opens a table row in a use-case file; [] with no spec tree.
|
|
212
|
+
*/
|
|
213
|
+
export function surfaceRows(cwd, slug) {
|
|
214
|
+
const dir = join(specDir(cwd, slug), "usecases");
|
|
215
|
+
let files = [];
|
|
216
|
+
try { files = readdirSync(dir).filter((f) => /^UC-.*\.md$/.test(f)).sort(); } catch { return []; }
|
|
217
|
+
const rows = [];
|
|
218
|
+
for (const f of files) {
|
|
219
|
+
let text = "";
|
|
220
|
+
try { text = readFileSync(join(dir, f), "utf8"); } catch { continue; }
|
|
221
|
+
for (const m of text.matchAll(/^\|\s*(TS-[A-Za-z0-9_-]+)\s*\|/gm)) if (!rows.includes(m[1])) rows.push(m[1]);
|
|
222
|
+
}
|
|
223
|
+
return rows;
|
|
224
|
+
}
|
|
225
|
+
|
|
226
|
+
/**
|
|
227
|
+
* Whether a PASS grades every Test Surface row by name.
|
|
228
|
+
*
|
|
229
|
+
* A PASS that grades the surface as a whole — "device rows", "local rows" — validated like any other
|
|
230
|
+
* and switched off everything that reads a verdict row by row: the requirements matrix read no
|
|
231
|
+
* evidence under every covered requirement, a failed row had no criterion to become a bug, and a row
|
|
232
|
+
* nobody looked at passed with the rest. A row counts as graded when its id appears in a criterion's
|
|
233
|
+
* text or its `traces_to`. Only a PASS is held to it: a FAIL sends the round back either way.
|
|
234
|
+
*
|
|
235
|
+
* @param {string} cwd - Project root.
|
|
236
|
+
* @param {string} slug - Feature slug.
|
|
237
|
+
* @param {object} verdict - The WorkResult's `verdict`.
|
|
238
|
+
* @returns {(string|null)} A reason naming the rows no criterion graded, or null.
|
|
239
|
+
*/
|
|
240
|
+
export function coverageProblem(cwd, slug, verdict) {
|
|
241
|
+
if (verdict?.overall !== "PASS") return null;
|
|
242
|
+
const rows = surfaceRows(cwd, slug);
|
|
243
|
+
if (!rows.length) return null;
|
|
244
|
+
const named = new Set();
|
|
245
|
+
for (const c of Array.isArray(verdict.criteria) ? verdict.criteria : []) {
|
|
246
|
+
const text = [c?.criterion, ...(Array.isArray(c?.traces_to) ? c.traces_to : [])].map((x) => String(x ?? "")).join(" ");
|
|
247
|
+
for (const m of text.matchAll(/\bTS-[A-Za-z0-9_]+(?:-[A-Za-z0-9_]+)*/g)) named.add(m[0]);
|
|
248
|
+
}
|
|
249
|
+
const missing = rows.filter((r) => !named.has(r));
|
|
250
|
+
if (!missing.length) return null;
|
|
251
|
+
const shown = missing.slice(0, 8).join(", ") + (missing.length > 8 ? ` and ${missing.length - 8} more` : "");
|
|
252
|
+
return `the PASS verdict grades ${rows.length - missing.length} of ${rows.length} Test Surface rows by name — ` +
|
|
253
|
+
`each row is its own criterion, its id in the criterion or its traces_to; ungraded: ${shown}`;
|
|
254
|
+
}
|
|
255
|
+
|
|
206
256
|
export function citationProblem(cwd, slug, verdict, { round = null } = {}) {
|
|
207
257
|
if (verdict?.overall !== "PASS" && verdict?.overall !== "FAIL") return null;
|
|
208
258
|
if (!isScoped(cwd, slug)) return null;
|
|
@@ -247,7 +297,7 @@ export function evalVerdict(cwd, slug, round) {
|
|
|
247
297
|
? `the evaluator returned ${status || "no status"}: ${first}`
|
|
248
298
|
: `status ${status || "unknown"} with no PASS/FAIL verdict`), status);
|
|
249
299
|
}
|
|
250
|
-
const problem = verdictProblem(v) || citationProblem(cwd, slug, v, { round });
|
|
300
|
+
const problem = verdictProblem(v) || coverageProblem(cwd, slug, v) || citationProblem(cwd, slug, v, { round });
|
|
251
301
|
if (problem) return unfit(problem, status, overall);
|
|
252
302
|
return {
|
|
253
303
|
found: true,
|
package/kernel/reduce/ingest.mjs
CHANGED
|
@@ -37,7 +37,7 @@ import { fileURLToPath } from "node:url";
|
|
|
37
37
|
import { validate } from "../verify/envelope.mjs";
|
|
38
38
|
import { runArgs } from "../lib/argv.mjs";
|
|
39
39
|
import { tasksDir, localRoot, dispatchReceipts, legLedger, readRunId } from "../lib/paths.mjs";
|
|
40
|
-
import { citationProblem, verdictProblem } from "../probe/eval.mjs";
|
|
40
|
+
import { citationProblem, coverageProblem, verdictProblem } from "../probe/eval.mjs";
|
|
41
41
|
|
|
42
42
|
const HERE = dirname(fileURLToPath(import.meta.url));
|
|
43
43
|
const RESULT_SCHEMA = JSON.parse(readFileSync(resolve(HERE, "../schemas/work-result.schema.json"), "utf8"));
|
|
@@ -692,17 +692,20 @@ export async function cli(rawArgv) {
|
|
|
692
692
|
process.exit(1);
|
|
693
693
|
}
|
|
694
694
|
|
|
695
|
-
// --- T0 citation
|
|
696
|
-
// A PASS or FAIL on a scoped spec that cites no T0 artifact
|
|
697
|
-
//
|
|
698
|
-
//
|
|
695
|
+
// --- Verdict gate: structure, row coverage, T0 citation ---------------------------------------
|
|
696
|
+
// A PASS or FAIL on a scoped spec that cites no T0 artifact, or a PASS that does not grade every
|
|
697
|
+
// Test Surface row by name, is not a judgement this run may act on (see `citationProblem`,
|
|
698
|
+
// `coverageProblem`). `probe eval` refuses it to the round loop; refusing it here as well keeps
|
|
699
|
+
// the verdict ledger from recording a verdict the loop will never branch on.
|
|
699
700
|
if (result.verdict) {
|
|
701
|
+
const evalSlug = String(result.order_id).split("/")[0];
|
|
700
702
|
const evalRound = Number((String(result.order_id).match(/-r(\d+)$/) || [])[1]) || null;
|
|
701
|
-
const problem = verdictProblem(result.verdict) ||
|
|
703
|
+
const problem = verdictProblem(result.verdict) || coverageProblem(cwd, evalSlug, result.verdict)
|
|
704
|
+
|| citationProblem(cwd, evalSlug, result.verdict, { round: evalRound });
|
|
702
705
|
if (problem) {
|
|
703
706
|
console.error(`ingest-result: result refused — ${problem}.`);
|
|
704
707
|
console.error(` The round stays open: re-dispatch the evaluator against its order, which lists`);
|
|
705
|
-
console.error(` the T0 artifacts to cite. Nothing was written.`);
|
|
708
|
+
console.error(` the T0 artifacts to cite and the spec whose rows it grades. Nothing was written.`);
|
|
706
709
|
process.exit(1);
|
|
707
710
|
}
|
|
708
711
|
}
|
package/kernel/reduce/ship.mjs
CHANGED
|
@@ -370,6 +370,21 @@ export function buildReport(facts) {
|
|
|
370
370
|
return deboard(L.join("\n"), board.anchors);
|
|
371
371
|
}
|
|
372
372
|
|
|
373
|
+
/**
|
|
374
|
+
* The QA line the report may print. The run says whether it dispatched the hunt; the hunt's own
|
|
375
|
+
* report says whether anything was hunted. A hunter that could not reach the app still returns, and
|
|
376
|
+
* a report opening `charters: 0/…` over a dispatched hunt read as "QA: run" with no findings — which
|
|
377
|
+
* a reader takes for an app with nothing wrong, not for an app nobody drove.
|
|
378
|
+
* @param {string|undefined} passed - What the caller recorded (`run`, `skipped`), if anything.
|
|
379
|
+
* @param {string|null} huntReport - The hunt report's text, or null when there is none.
|
|
380
|
+
* @returns {string} `skipped`, `not-hunted` (dispatched, zero charters run), or `run`.
|
|
381
|
+
*/
|
|
382
|
+
export function qaStatus(passed, huntReport) {
|
|
383
|
+
if (passed === "skipped" || (!passed && !huntReport)) return passed || "skipped";
|
|
384
|
+
if (huntReport && /^charters:\s*0\s*\//m.test(huntReport)) return "not-hunted";
|
|
385
|
+
return passed || "run";
|
|
386
|
+
}
|
|
387
|
+
|
|
373
388
|
/**
|
|
374
389
|
* Gather every fact from disk and render the report.
|
|
375
390
|
* @param {{cwd:string, slug:string, verdict?:string, qa?:string}} opts - Inputs.
|
|
@@ -394,7 +409,7 @@ export function generate({ cwd, slug, verdict, qa }) {
|
|
|
394
409
|
at: today(),
|
|
395
410
|
verdict: verdict || run.final_verdict || "not-evaluated",
|
|
396
411
|
census: (() => { try { return JSON.parse(readIf(hammerCensus(cwd, slug)) || "null"); } catch { return null; } })(),
|
|
397
|
-
qa: qa
|
|
412
|
+
qa: qaStatus(qa, huntReport),
|
|
398
413
|
rounds: derivedRounds.rounds_used,
|
|
399
414
|
roundsJudged: derivedRounds.rounds_judged,
|
|
400
415
|
intakeSha: receipt.intake_sha256,
|
package/package.json
CHANGED
|
@@ -70,9 +70,13 @@ Phase Q3 │ Report ───────► qa/hunt-report.md — no score, no
|
|
|
70
70
|
## GATE Q0 — Preflight
|
|
71
71
|
|
|
72
72
|
```
|
|
73
|
+
FIRST: read `payload.kb_rules_path` for the tool paths it names — a device tool that is not on PATH is
|
|
74
|
+
named there by its full path, and a bare-name lookup (`which`, a bare call) proves nothing.
|
|
73
75
|
HARD (any miss → STOP, report which):
|
|
74
76
|
✅ deliverable reachable: one real request at `app_url`, or `launch_cmd` run and exit 0, or one
|
|
75
|
-
real invocation of the entry point (not a ping, not a guess that a tool is missing)
|
|
77
|
+
real invocation of the entry point (not a ping, not a guess that a tool is missing). When the
|
|
78
|
+
order carries `launch_cmd`, run it before concluding anything; the report names the command
|
|
79
|
+
and its exit. A hunt that reached nothing returns `status: failed`, never `done`.
|
|
76
80
|
✅ EVAL-FEATURE-<slug>.md exists with verdict: PASS
|
|
77
81
|
✅ if discovery/ledger.md exists: ledger.feature == <feature> (read-only context check —
|
|
78
82
|
a missing ledger is fine; ingest creates it when your findings land)
|
|
@@ -175,6 +175,11 @@ round. Write it so someone without your context can answer it in one reply.
|
|
|
175
175
|
}
|
|
176
176
|
```
|
|
177
177
|
|
|
178
|
+
**Every Test Surface row is its own criterion.** Name the row's id (`TS-05-05`) in `criterion`, one
|
|
179
|
+
row per entry — never a range (`TS-01-04/05`) and never a group ("device rows"). Ingest refuses a
|
|
180
|
+
PASS that leaves any row of the spec unnamed, because the requirement matrix, the next round's bugs
|
|
181
|
+
and the rewritten-check list all read the verdict one row at a time.
|
|
182
|
+
|
|
178
183
|
**`traces_to` is copied, not invented.** Fill it from the `(covers: REQ-…)` clause of the
|
|
179
184
|
acceptance criteria your criterion grades: the AC already carries the link, written when the plan
|
|
180
185
|
was reviewed, and you record which requirement your criterion maps back to. An AC with no `covers:`
|
|
@@ -200,6 +205,7 @@ separation is the whole point of the architecture.
|
|
|
200
205
|
## Verification checklist
|
|
201
206
|
|
|
202
207
|
- [ ] Every criterion traces to committed spec text (UC/domain-model/contract/Done-when/Non-Go)
|
|
208
|
+
- [ ] Every Test Surface row is graded as its own criterion, by id
|
|
203
209
|
- [ ] `traces_to` copied from the graded ACs' `covers:` clauses — empty where they carry none
|
|
204
210
|
- [ ] Every PASS cites a confirming probe; every FAIL cites evidence or "NO EVIDENCE"
|
|
205
211
|
- [ ] Every FAIL was re-probed once; confidence assigned per the ledger rule
|
|
@@ -507,8 +507,9 @@ Effect: one feature-level pass over the running app against all AC + Done-when;
|
|
|
507
507
|
verdicts, refuted boxes, T0 citations). It touches NO task file and NO board.
|
|
508
508
|
ingest-result <results/evaluate-r<r>.json>: appends the .verdicts JSONL ledger, un-ticks the
|
|
509
509
|
refuted AC boxes, sets eval_verdict frontmatter — the judge returns data, ingest writes.
|
|
510
|
-
A verdict on a scoped spec that cites no T0 artifact
|
|
511
|
-
|
|
510
|
+
A verdict on a scoped spec that cites no T0 artifact, or a PASS that does not name
|
|
511
|
+
every Test Surface row as its own criterion, is refused and the round stays open:
|
|
512
|
+
re-dispatch the evaluator once with the refusal's reason, do not advance the round.
|
|
512
513
|
Read back: EVAL-FEATURE-<slug>.md → verdict (pass|fail) + the bug list (each bug has
|
|
513
514
|
task ref, severity, file:line, expected vs actual).
|
|
514
515
|
```
|
|
@@ -1868,8 +1868,8 @@ while (verdict !== "pass" && round <= maxRounds) {
|
|
|
1868
1868
|
// freezes at GATE L4 has to say that as plainly as the gate block already does.
|
|
1869
1869
|
verdict = "not-evaluated";
|
|
1870
1870
|
} else {
|
|
1871
|
-
const
|
|
1872
|
-
skill: "spec-evaluator", operation: "evaluate", schema: EVAL, phase: "Eval", label
|
|
1871
|
+
const evalOnce = (label, extraNote = "") => worker({
|
|
1872
|
+
skill: "spec-evaluator", operation: "evaluate", schema: EVAL, phase: "Eval", label,
|
|
1873
1873
|
model: evalModel, round,
|
|
1874
1874
|
// No `t0_artifacts` here, deliberately: `harness compile` derives them from the round's green
|
|
1875
1875
|
// T0 verdicts on disk, for every lane — this script could only name paths it was told about.
|
|
@@ -1877,12 +1877,22 @@ while (verdict !== "pass" && round <= maxRounds) {
|
|
|
1877
1877
|
// artifact and the project profile, so the judge gets the launch evidence without this script
|
|
1878
1878
|
// having to carry it.
|
|
1879
1879
|
payload: { dimensions: evalDims, run_cmd: rs.run_cmd, round },
|
|
1880
|
-
extra: "Evaluate the running feature against every acceptance criterion and Done-when. One feature-level pass; cite every artifact the order lists under t0_artifacts, re-hashing each yourself.",
|
|
1880
|
+
extra: "Evaluate the running feature against every acceptance criterion and Done-when. One feature-level pass; cite every artifact the order lists under t0_artifacts, re-hashing each yourself." + extraNote,
|
|
1881
1881
|
});
|
|
1882
|
+
let e = await evalOnce(`eval:r${round}`);
|
|
1882
1883
|
if (e.__failed) return await withWarnings(diedAt("L3", e));
|
|
1883
1884
|
// The pass/fail branch is decided from the WorkResult on disk, not from the dispatching
|
|
1884
1885
|
// agent's own summary of it (`e.overall`) — see EVAL_VERDICT's comment for why.
|
|
1885
|
-
|
|
1886
|
+
let ev = await query(`probe eval --slug ${slug} --round ${round}`, EVAL_VERDICT, "Eval", `verdict:r${round}`);
|
|
1887
|
+
// A verdict the kernel refused — a PASS that grades the surface as a group, a missing citation —
|
|
1888
|
+
// is a correctable answer, not a dead judge: the judge is sent back once with the refusal's own
|
|
1889
|
+
// words, and only a second refusal ends the round.
|
|
1890
|
+
if (ev && !ev.ok && ev.overall && ev.reason) {
|
|
1891
|
+
log(`EVAL r${round} — verdict refused (${ev.reason}); the judge is sent back once`);
|
|
1892
|
+
e = await evalOnce(`eval:r${round}:again`, ` Your previous verdict for this round was refused: ${ev.reason}. Grade again and correct that.`);
|
|
1893
|
+
if (e.__failed) return await withWarnings(diedAt("L3", e));
|
|
1894
|
+
ev = await query(`probe eval --slug ${slug} --round ${round}`, EVAL_VERDICT, "Eval", `verdict:r${round}:again`);
|
|
1895
|
+
}
|
|
1886
1896
|
if (!ev) return await withWarnings(diedAt("L3", nullFail(`verdict:r${round}`)));
|
|
1887
1897
|
// A round with no verdict to act on is NOT a dead worker. An evaluator that refused the round
|
|
1888
1898
|
// wrote a result saying why, and `probe eval` carries it as `reason`; reported as "died after
|