shapeup-sdlc 3.15.1 → 3.16.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude-plugin/plugin.json +1 -1
- package/AGENTS.md +1 -1
- package/commands/ship.md +4 -2
- package/kernel/gate.mjs +33 -1
- package/kernel/probe/eval.mjs +52 -2
- package/kernel/reduce/ingest.mjs +10 -7
- package/kernel/reduce/ship.mjs +18 -7
- package/package.json +1 -1
- package/skills/spec-evaluator/SKILL.md +6 -0
- package/skills/tech-lead/SKILL.md +4 -4
- package/skills/tech-lead/references/gates.md +3 -1
- package/skills/tech-lead/references/protocol.md +3 -2
- package/skills/tech-lead/workflows/shapeup-run.js +20 -6
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "shapeup-sdlc-plugin",
|
|
3
3
|
"displayName": "ShapeUp SDLC Plugin",
|
|
4
|
-
"version": "3.
|
|
4
|
+
"version": "3.16.1",
|
|
5
5
|
"description": "Shape Up SDLC harness for Claude Code: shaping, intake, orient, scope-mapping, building (T0-verified, sandboxed, scope-contracted), evaluation and QA skills orchestrated by a tech-lead.",
|
|
6
6
|
"author": {
|
|
7
7
|
"name": "Liberty Nguyen",
|
package/AGENTS.md
CHANGED
|
@@ -36,7 +36,7 @@ Betting Table: PO decides; rejected pitches loop back to raw idea.
|
|
|
36
36
|
| Wire | ⏸ **L1a.5** — Wiring Review ✚ | `/solution-architect` (`wire`): sole writer of committed `wiring-map.md` — per-UC engine → seam → entry-point call site → affordance, per `project-profile.md` |
|
|
37
37
|
| Map Scopes | ⏸ **L1b** — Board Review (+ substrate disjointness lint) | `/scope-architect` (scope contracts ✦ — sole writer); traceability oracle advisory ✚. A registered requirement that no acceptance criterion grades and no scope claims is **red** here, and L1b prints the `REQ → AC` table: the two ways out are an AC carrying `(covers: REQ-…)` or the PO marking the clause `CUT (PO-approved)`. Red only where the plan is still cheap to change — after L1b nobody re-reads the pitch |
|
|
38
38
|
| Build Vertically | ⏸ **L2** — Board 100% ✅ + T0-green ✦ | per dispatch: compile order → `/task-executor` (--order) → ingest result; T0-verified per attempt (fixtures + DB probe ✦), substrate-sandboxed ✦. Scopes build **concurrently** ✦ — `--parallel-scopes N` caps it (default 4), a scope is released the moment its own dependencies are green, and a scope green in this round is skipped rather than rebuilt. Then the **round build gate** ⚙: the ledger's run command, then the profile's `build_probe` and `launch_probe`, run once per round before EVAL — a red gate ends the round with no verdict and its failing step is compiled into the next round's orders as bugs; a `mobile` profile with no `launch_probe` is warned about every round, so the install/launch risk has an owner |
|
|
39
|
-
| EVAL (once per round) | ⏸ **L3** — Verdict | `/spec-evaluator` (--order), only over a round whose build gate ⚙ is not red: spec- + test-surface-conformance ★, T0 citation ✦; refuted boxes/verdict applied by ingest. The order names the round's build gate artifact and the profile's `launch_probe`, so a `[ui]` row is graded on the app the gate launched rather than recorded as no evidence — a project with no `launch_probe` still gets no on-device grade |
|
|
39
|
+
| EVAL (once per round) | ⏸ **L3** — Verdict | `/spec-evaluator` (--order), only over a round whose build gate ⚙ is not red: spec- + test-surface-conformance ★, T0 citation ✦; refuted boxes/verdict applied by ingest. The order names the round's build gate artifact and the profile's `launch_probe`, so a `[ui]` row is graded on the app the gate launched rather than recorded as no evidence — a project with no `launch_probe` still gets no on-device grade. A PASS names every Test Surface row as its own criterion: one that grades the surface as a group is refused, and the judge is sent back once with the reason |
|
|
40
40
|
| FAIL → round r+1 | — | regression rule ★: bugs + full Test Surface of touched UC. Every criterion the verdict graded FAIL reaches the round, whether or not the judge filed a bug for it, addressed to the scope that owns its use case — a round with a FAIL verdict and nothing to fix is how a run stalls. A row whose check the scope rewrote between a failing trial and a passing one is listed for the judge, who reads it against the row before the pass counts |
|
|
41
41
|
|
|
42
42
|
✦ = requires scope contracts (`shapeup/<slug>/scopes/*.md`); ✚ = requires the spine artifacts (`requirements.md`, `wiring-map.md`, `project-profile.md`). Traceability stays advisory until `covers:` is populated. Absent artifact ⇒ arm skipped (non-regression).
|
package/commands/ship.md
CHANGED
|
@@ -99,6 +99,8 @@ Additional flags, pass through to `tech-lead` only when the user names them:
|
|
|
99
99
|
- `--rounds N` → override the outer circuit breaker (build+eval cycles, default 3).
|
|
100
100
|
- `--attempts N` → override the inner circuit breaker (per-scope T0 attempts, default 5;
|
|
101
101
|
no-op on specs without scope contracts).
|
|
102
|
-
- `--
|
|
102
|
+
- `--exec-model / --eval-model / --qa-model <name>` → override GATE L0.8's
|
|
103
103
|
resolved model matrix for this run only (highest precedence over `.claude/settings.local.json`
|
|
104
|
-
/ `.claude/settings.json` / skill defaults).
|
|
104
|
+
/ `.claude/settings.json` / skill defaults). `--orch-model` is shown in the L0 block and changes
|
|
105
|
+
nothing: the orchestrator is the session already running this command, so its model is the one
|
|
106
|
+
that session was started with.
|
package/kernel/gate.mjs
CHANGED
|
@@ -54,7 +54,7 @@ import { readFileSync, writeFileSync, appendFileSync, existsSync, mkdirSync } fr
|
|
|
54
54
|
import { parseBoard } from "./reduce/board.mjs";
|
|
55
55
|
import { join, dirname } from "node:path";
|
|
56
56
|
import { runArgs } from "./lib/argv.mjs";
|
|
57
|
-
import { gateAnswerCandidates, gates as gatesPath, LOCAL, resultsDir, tasksDir, hammerCensus, readRunId } from "./lib/paths.mjs";
|
|
57
|
+
import { gateAnswerCandidates, gates as gatesPath, LOCAL, resultsDir, tasksDir, hammerCensus, readRunId, harnessRun } from "./lib/paths.mjs";
|
|
58
58
|
|
|
59
59
|
export const GATE_IDS = ["L0", "L1a", "L1a.5", "L1b", "L2", "L3", "QA", "H", "L4", "COACH-1"];
|
|
60
60
|
|
|
@@ -327,6 +327,34 @@ export function appendGateLedger(cwd, slug, row) {
|
|
|
327
327
|
} catch { return false; }
|
|
328
328
|
}
|
|
329
329
|
|
|
330
|
+
/**
|
|
331
|
+
* This run's earlier ship sign-off, if it already has one.
|
|
332
|
+
*
|
|
333
|
+
* A launched run crosses L4 itself and then closes, and the orchestrator's own closing step resolved
|
|
334
|
+
* it again afterwards — a second `ship` row, after the close, for a decision already taken. Once the
|
|
335
|
+
* run is closed, its decided L4 row is returned instead of resolving again. An open run still
|
|
336
|
+
* re-resolves (the census it reads may have changed), a reopened run has no close, and a paused or
|
|
337
|
+
* aborted row is not a sign-off.
|
|
338
|
+
*
|
|
339
|
+
* @param {string} cwd - Project root.
|
|
340
|
+
* @param {string} slug - Feature slug.
|
|
341
|
+
* @param {string|null} runId - The run the resolve belongs to.
|
|
342
|
+
* @returns {(object|null)} The earlier L4 row, or null.
|
|
343
|
+
*/
|
|
344
|
+
export function priorSignOff(cwd, slug, runId) {
|
|
345
|
+
if (!runId) return null;
|
|
346
|
+
let ledger = "";
|
|
347
|
+
try { ledger = readFileSync(harnessRun(cwd, slug), "utf8"); } catch { return null; }
|
|
348
|
+
if (!/^closed_status:[ \t]*[^\s~]/m.test(ledger)) return null;
|
|
349
|
+
let text = "";
|
|
350
|
+
try { text = readFileSync(gatesPath(cwd, slug), "utf8"); } catch { return null; }
|
|
351
|
+
for (const line of text.split("\n")) {
|
|
352
|
+
let row; try { row = JSON.parse(line); } catch { continue; }
|
|
353
|
+
if (row?.gate === "L4" && row.run_id === runId && row.status === "ok") return row;
|
|
354
|
+
}
|
|
355
|
+
return null;
|
|
356
|
+
}
|
|
357
|
+
|
|
330
358
|
/** Gates this lane will actually hit — used by --verify to catch a set that stalls halfway. */
|
|
331
359
|
export function requiredGates({ autoLevel = "unattended", tiny = false, qa = true } = {}) {
|
|
332
360
|
if (tiny) return ["L0", "L4"];
|
|
@@ -483,6 +511,10 @@ export function cli(rawArgv) {
|
|
|
483
511
|
const gate = args.resolve ?? null;
|
|
484
512
|
if (!gate) die("nothing to do — pass --init, --list, --verify, or --resolve <gate-id>");
|
|
485
513
|
|
|
514
|
+
if (gate === "L4" && args.slug) {
|
|
515
|
+
const prior = priorSignOff(cwd, args.slug, readRunId(cwd, args.slug));
|
|
516
|
+
if (prior) out({ ok: true, gate: "L4", status: prior.status, decision: prior.decision, source: prior.source, already_resolved_at: prior.at }, 0);
|
|
517
|
+
}
|
|
486
518
|
const r = narrowToEvidence(resolve(found.set, gate, found.source), cwd, args.slug ?? null);
|
|
487
519
|
if (r.status === "error") die(r.reason);
|
|
488
520
|
// A gate with no `--slug` (e.g. `--file` used ad hoc, outside any run) has nowhere to file a
|
package/kernel/probe/eval.mjs
CHANGED
|
@@ -35,7 +35,7 @@ import { existsSync, readFileSync, readdirSync, realpathSync } from "node:fs";
|
|
|
35
35
|
import { join, resolve, resolve as resolvePath, sep } from "node:path";
|
|
36
36
|
import { createHash } from "node:crypto";
|
|
37
37
|
import { runArgs } from "../lib/argv.mjs";
|
|
38
|
-
import { resultsDir, scopesDir, readRunId, verdictsDir } from "../lib/paths.mjs";
|
|
38
|
+
import { resultsDir, scopesDir, readRunId, verdictsDir, specDir } from "../lib/paths.mjs";
|
|
39
39
|
|
|
40
40
|
/** Longest `reason` reported. A deviation is prose written by a worker and can run to paragraphs. */
|
|
41
41
|
const REASON_MAX = 400;
|
|
@@ -203,6 +203,56 @@ export function verdictProblem(verdict) {
|
|
|
203
203
|
return null;
|
|
204
204
|
}
|
|
205
205
|
|
|
206
|
+
/**
|
|
207
|
+
* The Test Surface row ids the spec declares, in spec order.
|
|
208
|
+
*
|
|
209
|
+
* @param {string} cwd - Project root.
|
|
210
|
+
* @param {string} slug - Feature slug.
|
|
211
|
+
* @returns {string[]} Every `TS-…` id that opens a table row in a use-case file; [] with no spec tree.
|
|
212
|
+
*/
|
|
213
|
+
export function surfaceRows(cwd, slug) {
|
|
214
|
+
const dir = join(specDir(cwd, slug), "usecases");
|
|
215
|
+
let files = [];
|
|
216
|
+
try { files = readdirSync(dir).filter((f) => /^UC-.*\.md$/.test(f)).sort(); } catch { return []; }
|
|
217
|
+
const rows = [];
|
|
218
|
+
for (const f of files) {
|
|
219
|
+
let text = "";
|
|
220
|
+
try { text = readFileSync(join(dir, f), "utf8"); } catch { continue; }
|
|
221
|
+
for (const m of text.matchAll(/^\|\s*(TS-[A-Za-z0-9_-]+)\s*\|/gm)) if (!rows.includes(m[1])) rows.push(m[1]);
|
|
222
|
+
}
|
|
223
|
+
return rows;
|
|
224
|
+
}
|
|
225
|
+
|
|
226
|
+
/**
|
|
227
|
+
* Whether a PASS grades every Test Surface row by name.
|
|
228
|
+
*
|
|
229
|
+
* A PASS that grades the surface as a whole — "device rows", "local rows" — validated like any other
|
|
230
|
+
* and switched off everything that reads a verdict row by row: the requirements matrix read no
|
|
231
|
+
* evidence under every covered requirement, a failed row had no criterion to become a bug, and a row
|
|
232
|
+
* nobody looked at passed with the rest. A row counts as graded when its id appears in a criterion's
|
|
233
|
+
* text or its `traces_to`. Only a PASS is held to it: a FAIL sends the round back either way.
|
|
234
|
+
*
|
|
235
|
+
* @param {string} cwd - Project root.
|
|
236
|
+
* @param {string} slug - Feature slug.
|
|
237
|
+
* @param {object} verdict - The WorkResult's `verdict`.
|
|
238
|
+
* @returns {(string|null)} A reason naming the rows no criterion graded, or null.
|
|
239
|
+
*/
|
|
240
|
+
export function coverageProblem(cwd, slug, verdict) {
|
|
241
|
+
if (verdict?.overall !== "PASS") return null;
|
|
242
|
+
const rows = surfaceRows(cwd, slug);
|
|
243
|
+
if (!rows.length) return null;
|
|
244
|
+
const named = new Set();
|
|
245
|
+
for (const c of Array.isArray(verdict.criteria) ? verdict.criteria : []) {
|
|
246
|
+
const text = [c?.criterion, ...(Array.isArray(c?.traces_to) ? c.traces_to : [])].map((x) => String(x ?? "")).join(" ");
|
|
247
|
+
for (const m of text.matchAll(/\bTS-[A-Za-z0-9_]+(?:-[A-Za-z0-9_]+)*/g)) named.add(m[0]);
|
|
248
|
+
}
|
|
249
|
+
const missing = rows.filter((r) => !named.has(r));
|
|
250
|
+
if (!missing.length) return null;
|
|
251
|
+
const shown = missing.slice(0, 8).join(", ") + (missing.length > 8 ? ` and ${missing.length - 8} more` : "");
|
|
252
|
+
return `the PASS verdict grades ${rows.length - missing.length} of ${rows.length} Test Surface rows by name — ` +
|
|
253
|
+
`each row is its own criterion, its id in the criterion or its traces_to; ungraded: ${shown}`;
|
|
254
|
+
}
|
|
255
|
+
|
|
206
256
|
export function citationProblem(cwd, slug, verdict, { round = null } = {}) {
|
|
207
257
|
if (verdict?.overall !== "PASS" && verdict?.overall !== "FAIL") return null;
|
|
208
258
|
if (!isScoped(cwd, slug)) return null;
|
|
@@ -247,7 +297,7 @@ export function evalVerdict(cwd, slug, round) {
|
|
|
247
297
|
? `the evaluator returned ${status || "no status"}: ${first}`
|
|
248
298
|
: `status ${status || "unknown"} with no PASS/FAIL verdict`), status);
|
|
249
299
|
}
|
|
250
|
-
const problem = verdictProblem(v) || citationProblem(cwd, slug, v, { round });
|
|
300
|
+
const problem = verdictProblem(v) || coverageProblem(cwd, slug, v) || citationProblem(cwd, slug, v, { round });
|
|
251
301
|
if (problem) return unfit(problem, status, overall);
|
|
252
302
|
return {
|
|
253
303
|
found: true,
|
package/kernel/reduce/ingest.mjs
CHANGED
|
@@ -37,7 +37,7 @@ import { fileURLToPath } from "node:url";
|
|
|
37
37
|
import { validate } from "../verify/envelope.mjs";
|
|
38
38
|
import { runArgs } from "../lib/argv.mjs";
|
|
39
39
|
import { tasksDir, localRoot, dispatchReceipts, legLedger, readRunId } from "../lib/paths.mjs";
|
|
40
|
-
import { citationProblem, verdictProblem } from "../probe/eval.mjs";
|
|
40
|
+
import { citationProblem, coverageProblem, verdictProblem } from "../probe/eval.mjs";
|
|
41
41
|
|
|
42
42
|
const HERE = dirname(fileURLToPath(import.meta.url));
|
|
43
43
|
const RESULT_SCHEMA = JSON.parse(readFileSync(resolve(HERE, "../schemas/work-result.schema.json"), "utf8"));
|
|
@@ -692,17 +692,20 @@ export async function cli(rawArgv) {
|
|
|
692
692
|
process.exit(1);
|
|
693
693
|
}
|
|
694
694
|
|
|
695
|
-
// --- T0 citation
|
|
696
|
-
// A PASS or FAIL on a scoped spec that cites no T0 artifact
|
|
697
|
-
//
|
|
698
|
-
//
|
|
695
|
+
// --- Verdict gate: structure, row coverage, T0 citation ---------------------------------------
|
|
696
|
+
// A PASS or FAIL on a scoped spec that cites no T0 artifact, or a PASS that does not grade every
|
|
697
|
+
// Test Surface row by name, is not a judgement this run may act on (see `citationProblem`,
|
|
698
|
+
// `coverageProblem`). `probe eval` refuses it to the round loop; refusing it here as well keeps
|
|
699
|
+
// the verdict ledger from recording a verdict the loop will never branch on.
|
|
699
700
|
if (result.verdict) {
|
|
701
|
+
const evalSlug = String(result.order_id).split("/")[0];
|
|
700
702
|
const evalRound = Number((String(result.order_id).match(/-r(\d+)$/) || [])[1]) || null;
|
|
701
|
-
const problem = verdictProblem(result.verdict) ||
|
|
703
|
+
const problem = verdictProblem(result.verdict) || coverageProblem(cwd, evalSlug, result.verdict)
|
|
704
|
+
|| citationProblem(cwd, evalSlug, result.verdict, { round: evalRound });
|
|
702
705
|
if (problem) {
|
|
703
706
|
console.error(`ingest-result: result refused — ${problem}.`);
|
|
704
707
|
console.error(` The round stays open: re-dispatch the evaluator against its order, which lists`);
|
|
705
|
-
console.error(` the T0 artifacts to cite. Nothing was written.`);
|
|
708
|
+
console.error(` the T0 artifacts to cite and the spec whose rows it grades. Nothing was written.`);
|
|
706
709
|
process.exit(1);
|
|
707
710
|
}
|
|
708
711
|
}
|
package/kernel/reduce/ship.mjs
CHANGED
|
@@ -32,7 +32,7 @@ import { runArgs } from "../lib/argv.mjs";
|
|
|
32
32
|
import {
|
|
33
33
|
report as reportPath, tasksDir, verdictsDir, trials, evaluationDir, qaDir,
|
|
34
34
|
roundLedger, discoveryLedger, receipt as receiptPath, harnessRun, relShared,
|
|
35
|
-
activeOrder, runArgsPath, readReceipt, runIdFromReceipt, hammerCensus,
|
|
35
|
+
activeOrder, runArgsPath, readReceipt, runIdFromReceipt, hammerCensus, LOCAL,
|
|
36
36
|
} from "../lib/paths.mjs";
|
|
37
37
|
import { readTrials } from "../verify/t0.mjs";
|
|
38
38
|
import { ratchetReport } from "../probe/stats.mjs";
|
|
@@ -112,16 +112,27 @@ export function boardCensus(cwd, slug) {
|
|
|
112
112
|
* An id with no resolvable use case becomes a neutral phrase rather than the id: the report loses a
|
|
113
113
|
* pointer that never resolved off this machine anyway, and keeps the sentence around it.
|
|
114
114
|
*
|
|
115
|
+
* A path into the run trace is the same leak by another spelling: the QA findings section quotes
|
|
116
|
+
* the hunt report, which points at the discovery ledger by its local path, and the next run's
|
|
117
|
+
* spec-lint refused the report for it. The path becomes "the run trace", and the sentence stays.
|
|
118
|
+
*
|
|
115
119
|
* @param {*} text - Any value destined for the committed report.
|
|
116
120
|
* @param {Record<string, string[]>} anchors - Board id → its `use_case_refs`.
|
|
117
|
-
* @returns {string} The text with every `TASK-…` replaced by a stable anchor
|
|
121
|
+
* @returns {string} The text with every `TASK-…` replaced by a stable anchor and every run-trace
|
|
122
|
+
* path by a phrase.
|
|
118
123
|
*/
|
|
119
124
|
export function deboard(text, anchors = {}) {
|
|
120
|
-
|
|
121
|
-
|
|
122
|
-
|
|
123
|
-
|
|
124
|
-
|
|
125
|
+
const esc = LOCAL.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
|
|
126
|
+
return String(text ?? "")
|
|
127
|
+
.replace(/\bTASK-[A-Za-z0-9][\w.-]*/g, (id) => {
|
|
128
|
+
const ucs = anchors[id];
|
|
129
|
+
if (ucs && ucs.length) return ucs.join("/");
|
|
130
|
+
return "a board task";
|
|
131
|
+
})
|
|
132
|
+
.replace(new RegExp(`\`?${esc}/[^\\s\`]+\`?`, "g"), (m) => {
|
|
133
|
+
const tail = (m.replace(/`/g, "").match(/[)\]},.;:'"]+$/) || [""])[0];
|
|
134
|
+
return `the run trace${tail}`;
|
|
135
|
+
});
|
|
125
136
|
}
|
|
126
137
|
|
|
127
138
|
/**
|
package/package.json
CHANGED
|
@@ -175,6 +175,11 @@ round. Write it so someone without your context can answer it in one reply.
|
|
|
175
175
|
}
|
|
176
176
|
```
|
|
177
177
|
|
|
178
|
+
**Every Test Surface row is its own criterion.** Name the row's id (`TS-05-05`) in `criterion`, one
|
|
179
|
+
row per entry — never a range (`TS-01-04/05`) and never a group ("device rows"). Ingest refuses a
|
|
180
|
+
PASS that leaves any row of the spec unnamed, because the requirement matrix, the next round's bugs
|
|
181
|
+
and the rewritten-check list all read the verdict one row at a time.
|
|
182
|
+
|
|
178
183
|
**`traces_to` is copied, not invented.** Fill it from the `(covers: REQ-…)` clause of the
|
|
179
184
|
acceptance criteria your criterion grades: the AC already carries the link, written when the plan
|
|
180
185
|
was reviewed, and you record which requirement your criterion maps back to. An AC with no `covers:`
|
|
@@ -200,6 +205,7 @@ separation is the whole point of the architecture.
|
|
|
200
205
|
## Verification checklist
|
|
201
206
|
|
|
202
207
|
- [ ] Every criterion traces to committed spec text (UC/domain-model/contract/Done-when/Non-Go)
|
|
208
|
+
- [ ] Every Test Surface row is graded as its own criterion, by id
|
|
203
209
|
- [ ] `traces_to` copied from the graded ACs' `covers:` clauses — empty where they carry none
|
|
204
210
|
- [ ] Every PASS cites a confirming probe; every FAIL cites evidence or "NO EVIDENCE"
|
|
205
211
|
- [ ] Every FAIL was re-probed once; confidence assigned per the ledger rule
|
|
@@ -110,11 +110,11 @@ failed, launch from the install path and have the operator `/add-dir` the plugin
|
|
|
110
110
|
did not actually receive from the PO — an unattended lane with no answer for a gate is meant to
|
|
111
111
|
`abort` (see `harness gate`'s `on_missing`), not silently proceed.
|
|
112
112
|
|
|
113
|
-
## Step 4 — GATE L4 — Ship Sign-Off
|
|
113
|
+
## Step 4 — GATE L4 — Ship Sign-Off
|
|
114
114
|
|
|
115
|
-
|
|
116
|
-
|
|
117
|
-
|
|
115
|
+
A `shipped`/`escalated` return already crossed L4 in the run and froze `REPORT.md`: never resolve it
|
|
116
|
+
again (a row after the close is a second sign-off). A return that did not reach L4 freezes the report
|
|
117
|
+
now and RESOLVES the gate (`references/gates.md` GATE L4). Either way, emit:
|
|
118
118
|
|
|
119
119
|
```
|
|
120
120
|
⏸ GATE L4 — Ship Sign-Off
|
|
@@ -571,7 +571,9 @@ On confirm:
|
|
|
571
571
|
- If the PO provides substantive feedback (not just 'y' or empty) → automatically delegate via Agent (model: exec — see references/protocol.md "Invocation mechanism"): Skill(shapeup-sdlc-plugin:coach) with the provided feedback for RLHF. The coach runs its own GATE COACH-1 to have the PO categorize each rule, then files it under the responsible skill in `shapeup/knowledge-base/<skill>.md` (committed → team-shared). Coachable: `task-executor`, `ba-pitch-analyzer`, `qa-edge-hunter`, `orient`, `scope-architect`, `solution-architect` (each reads its own file at the top of its next run) and `tech-lead` (workflow guidance, read at the next GATE L0). Guidance never decides a gate: a filed rule may add a question or a check to a gate block, never an answer. The tech lead does not categorize the feedback itself — that is the coach's gate, by design (no assumptions).
|
|
572
572
|
- Then output → `✅ [slug] [shipped & deployed | built & verified, deploy pending] — [r] rounds, verdict PASS.`
|
|
573
573
|
|
|
574
|
-
**
|
|
574
|
+
**A launched run already crossed L4 — never resolve it again after a `shipped` or `escalated`
|
|
575
|
+
return; a second row after the close is a second sign-off.** In the prose lane, **resolve the gate
|
|
576
|
+
itself before any of the above** — this is the decision that shipped the run,
|
|
575
577
|
and without it the trace holds no record of that decision at all: `node
|
|
576
578
|
"${CLAUDE_PLUGIN_ROOT}/kernel/harness.mjs" gate --resolve L4 --slug <slug>
|
|
577
579
|
[--file <path>|--preset <name>]`. Exit 0 (`decision=ship|hold`) — render the block above and close
|
|
@@ -507,8 +507,9 @@ Effect: one feature-level pass over the running app against all AC + Done-when;
|
|
|
507
507
|
verdicts, refuted boxes, T0 citations). It touches NO task file and NO board.
|
|
508
508
|
ingest-result <results/evaluate-r<r>.json>: appends the .verdicts JSONL ledger, un-ticks the
|
|
509
509
|
refuted AC boxes, sets eval_verdict frontmatter — the judge returns data, ingest writes.
|
|
510
|
-
A verdict on a scoped spec that cites no T0 artifact
|
|
511
|
-
|
|
510
|
+
A verdict on a scoped spec that cites no T0 artifact, or a PASS that does not name
|
|
511
|
+
every Test Surface row as its own criterion, is refused and the round stays open:
|
|
512
|
+
re-dispatch the evaluator once with the refusal's reason, do not advance the round.
|
|
512
513
|
Read back: EVAL-FEATURE-<slug>.md → verdict (pass|fail) + the bug list (each bug has
|
|
513
514
|
task ref, severity, file:line, expected vs actual).
|
|
514
515
|
```
|
|
@@ -112,6 +112,10 @@ const argProblems = validateArgs(args);
|
|
|
112
112
|
if (argProblems.length) return { status: "aborted", aborted_at: "args", reason: argProblems.join("; ") };
|
|
113
113
|
|
|
114
114
|
const slug = args.slug;
|
|
115
|
+
// The report's path, never the ship command's one-line `detail`: that is a sub-agent's sentence, and
|
|
116
|
+
// it reached the RunReturn's `report` field in place of the path the orchestrator opens. The
|
|
117
|
+
// committed root is a fixed name, so the path is known without reading anything.
|
|
118
|
+
const REPORT_PATH = `shapeup/${slug}/REPORT.md`;
|
|
115
119
|
const KERNEL = `${args.pluginRoot}/kernel/harness.mjs`;
|
|
116
120
|
const execModel = args.models.exec;
|
|
117
121
|
const evalModel = args.models.eval;
|
|
@@ -697,7 +701,7 @@ async function settleAtGateH(ret) {
|
|
|
697
701
|
status: "shipped", verdict, rounds_used: round, after: "gate_h", breaker: ret.breaker ?? null,
|
|
698
702
|
census: h.verdict, cut_list: h.cut_list, green_scopes: ret.green_scopes,
|
|
699
703
|
unapplied_results: ret.unapplied_results || [], qa_findings: 0,
|
|
700
|
-
report:
|
|
704
|
+
report: REPORT_PATH,
|
|
701
705
|
});
|
|
702
706
|
}
|
|
703
707
|
|
|
@@ -1868,8 +1872,8 @@ while (verdict !== "pass" && round <= maxRounds) {
|
|
|
1868
1872
|
// freezes at GATE L4 has to say that as plainly as the gate block already does.
|
|
1869
1873
|
verdict = "not-evaluated";
|
|
1870
1874
|
} else {
|
|
1871
|
-
const
|
|
1872
|
-
skill: "spec-evaluator", operation: "evaluate", schema: EVAL, phase: "Eval", label
|
|
1875
|
+
const evalOnce = (label, extraNote = "") => worker({
|
|
1876
|
+
skill: "spec-evaluator", operation: "evaluate", schema: EVAL, phase: "Eval", label,
|
|
1873
1877
|
model: evalModel, round,
|
|
1874
1878
|
// No `t0_artifacts` here, deliberately: `harness compile` derives them from the round's green
|
|
1875
1879
|
// T0 verdicts on disk, for every lane — this script could only name paths it was told about.
|
|
@@ -1877,12 +1881,22 @@ while (verdict !== "pass" && round <= maxRounds) {
|
|
|
1877
1881
|
// artifact and the project profile, so the judge gets the launch evidence without this script
|
|
1878
1882
|
// having to carry it.
|
|
1879
1883
|
payload: { dimensions: evalDims, run_cmd: rs.run_cmd, round },
|
|
1880
|
-
extra: "Evaluate the running feature against every acceptance criterion and Done-when. One feature-level pass; cite every artifact the order lists under t0_artifacts, re-hashing each yourself.",
|
|
1884
|
+
extra: "Evaluate the running feature against every acceptance criterion and Done-when. One feature-level pass; cite every artifact the order lists under t0_artifacts, re-hashing each yourself." + extraNote,
|
|
1881
1885
|
});
|
|
1886
|
+
let e = await evalOnce(`eval:r${round}`);
|
|
1882
1887
|
if (e.__failed) return await withWarnings(diedAt("L3", e));
|
|
1883
1888
|
// The pass/fail branch is decided from the WorkResult on disk, not from the dispatching
|
|
1884
1889
|
// agent's own summary of it (`e.overall`) — see EVAL_VERDICT's comment for why.
|
|
1885
|
-
|
|
1890
|
+
let ev = await query(`probe eval --slug ${slug} --round ${round}`, EVAL_VERDICT, "Eval", `verdict:r${round}`);
|
|
1891
|
+
// A verdict the kernel refused — a PASS that grades the surface as a group, a missing citation —
|
|
1892
|
+
// is a correctable answer, not a dead judge: the judge is sent back once with the refusal's own
|
|
1893
|
+
// words, and only a second refusal ends the round.
|
|
1894
|
+
if (ev && !ev.ok && ev.overall && ev.reason) {
|
|
1895
|
+
log(`EVAL r${round} — verdict refused (${ev.reason}); the judge is sent back once`);
|
|
1896
|
+
e = await evalOnce(`eval:r${round}:again`, ` Your previous verdict for this round was refused: ${ev.reason}. Grade again and correct that.`);
|
|
1897
|
+
if (e.__failed) return await withWarnings(diedAt("L3", e));
|
|
1898
|
+
ev = await query(`probe eval --slug ${slug} --round ${round}`, EVAL_VERDICT, "Eval", `verdict:r${round}:again`);
|
|
1899
|
+
}
|
|
1886
1900
|
if (!ev) return await withWarnings(diedAt("L3", nullFail(`verdict:r${round}`)));
|
|
1887
1901
|
// A round with no verdict to act on is NOT a dead worker. An evaluator that refused the round
|
|
1888
1902
|
// wrote a result saying why, and `probe eval` carries it as `reason`; reported as "died after
|
|
@@ -2002,7 +2016,7 @@ return await withWarnings({
|
|
|
2002
2016
|
rounds_used: round,
|
|
2003
2017
|
dims_not_evaluated: ALL_DIMS.filter((d) => !evalDims.includes(d)),
|
|
2004
2018
|
qa_findings: qaFindings,
|
|
2005
|
-
report:
|
|
2019
|
+
report: REPORT_PATH,
|
|
2006
2020
|
});
|
|
2007
2021
|
|
|
2008
2022
|
// =============================================================================================
|