shapeup-sdlc 3.14.0 → 3.15.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "name": "shapeup-sdlc-plugin",
3
3
  "displayName": "ShapeUp SDLC Plugin",
4
- "version": "3.14.0",
4
+ "version": "3.15.0",
5
5
  "description": "Shape Up SDLC harness for Claude Code: shaping, intake, orient, scope-mapping, building (T0-verified, sandboxed, scope-contracted), evaluation and QA skills orchestrated by a tech-lead.",
6
6
  "author": {
7
7
  "name": "Liberty Nguyen",
@@ -527,6 +527,44 @@ export function verdictBugs(cwd, slug, round) {
527
527
  return v.bugs.filter((b) => !refuted.has(String(b?.id)) && !refuted.has(String(b?.criterion)));
528
528
  }
529
529
 
530
+ /**
531
+ * Who owns the row a bug or a failed criterion is about — by use case, not by file.
532
+ *
533
+ * A bug names a code location, and electing its owner from that location is right when one scope
534
+ * writes the file. Screens are shared: on a live run two bugs about rows of UC-05 and UC-07 both
535
+ * named screen files that three scopes could write, both were elected to a third scope that owned
536
+ * neither row, and that scope could not write the row's check and escalated. The criterion says
537
+ * which row it is; the row's use case says which scope owns it. Read from the name directly
538
+ * (`UC-05 …`), else from the spec's use-case file whose Test Surface lists the row id.
539
+ *
540
+ * @param {string} cwd - Project root.
541
+ * @param {string} slug - Feature slug.
542
+ * @returns {function(string): (string|undefined)} Criterion text → owning scope_id, or undefined
543
+ * when no use case can be read off it or no scope lists that use case.
544
+ */
545
+ export function rowOwner(cwd, slug) {
546
+ const owners = new Map();
547
+ for (const { contract, id } of readAllContracts(scopesDir(cwd, slug))) {
548
+ const sid = contract?.scope_id || id;
549
+ for (const uc of Array.isArray(contract?.use_cases) ? contract.use_cases : []) if (!owners.has(uc)) owners.set(uc, sid);
550
+ }
551
+ const rowUc = new Map();
552
+ const ucDir = join(defaultSpecDir(cwd, slug), "usecases");
553
+ let files = [];
554
+ try { files = readdirSync(ucDir).filter((f) => /^UC-.*\.md$/.test(f)); } catch { /* no spec tree */ }
555
+ for (const f of files) {
556
+ const uc = f.replace(/\.md$/, "");
557
+ let text = "";
558
+ try { text = readFileSync(join(ucDir, f), "utf8"); } catch { continue; }
559
+ for (const m of text.matchAll(/^\|\s*(TS-[A-Za-z0-9_-]+)\s*\|/gm)) if (!rowUc.has(m[1])) rowUc.set(m[1], uc);
560
+ }
561
+ return (criterion) => {
562
+ const s = String(criterion ?? "");
563
+ const uc = (s.match(/\bUC-[A-Za-z0-9_-]+/) || [])[0] || rowUc.get((s.match(/\bTS-[A-Za-z0-9_-]+/) || [])[0]);
564
+ return uc ? owners.get(uc) : undefined;
565
+ };
566
+ }
567
+
530
568
  /**
531
569
  * The previous round's FAILED CRITERIA that the judge filed no bug for, as bug entries.
532
570
  *
@@ -578,6 +616,22 @@ export function criteriaBugs(cwd, slug, round) {
578
616
  return out;
579
617
  }
580
618
 
619
+ /**
620
+ * Stamp each verdict bug with the scope that owns the row it names, where one can be read. A bug
621
+ * that already carries a scope, or names no row a scope owns, is left to the file election.
622
+ *
623
+ * @param {Array<object>} bugs - Verdict and criterion bugs.
624
+ * @param {function(string): (string|undefined)} owner - From {@link rowOwner}.
625
+ * @returns {Array<object>} The same bugs, `scope_id` added where the row's owner is known.
626
+ */
627
+ export function byRow(bugs, owner) {
628
+ return (bugs || []).map((b) => {
629
+ if (b?.scope_id) return b;
630
+ const sid = owner(b?.criterion);
631
+ return sid ? { ...b, scope_id: sid } : b;
632
+ });
633
+ }
634
+
581
635
  /**
582
636
  * The previous round's RED BUILD GATE, as bug entries the fix round can act on.
583
637
  *
@@ -1099,7 +1153,7 @@ export async function cli(rawArgv) {
1099
1153
  // this line, and none of them can pass a payload to a build order (see the banner above).
1100
1154
  // Two sources, one channel: the judge's cited defects and the build gate's failing steps.
1101
1155
  const bugs = scope
1102
- ? bugsForScope([...verdictBugs(cwd, slug, round), ...criteriaBugs(cwd, slug, round), ...buildBugs(cwd, slug, round)], scope.scope_id, scopeSubstrates(cwd, slug))
1156
+ ? bugsForScope(byRow([...verdictBugs(cwd, slug, round), ...criteriaBugs(cwd, slug, round)], rowOwner(cwd, slug)).concat(buildBugs(cwd, slug, round)), scope.scope_id, scopeSubstrates(cwd, slug))
1103
1157
  : [];
1104
1158
 
1105
1159
  // A ROUND CARRYING CITED DEFECTS IS A `fix`, AND THE ORDER HAS TO SAY SO.
@@ -1225,7 +1279,10 @@ export async function cli(rawArgv) {
1225
1279
  const revised = revisedChecksFor(cwd, slug, round);
1226
1280
  if (revised.length) payloadExtra.revised_checks = revised;
1227
1281
  }
1228
- if (operation === "evaluate") {
1282
+ // The QA hunt drives the same running app the judge grades, so it needs the same way in. A mobile
1283
+ // deliverable has no URL, and a hunt order carrying only `app_url: null` reported "no reachable
1284
+ // deliverable" and hunted nothing over a build the round gate had just launched.
1285
+ if (operation === "evaluate" || operation === "hunt") {
1229
1286
  const ev = launchEvidenceFor(cwd, slug, round);
1230
1287
  if (ev.build_gate !== undefined && payloadExtra.build_gate === undefined) payloadExtra.build_gate = ev.build_gate;
1231
1288
  if (ev.launch_cmd !== undefined && payloadExtra.launch_cmd === undefined) payloadExtra.launch_cmd = ev.launch_cmd;
@@ -633,10 +633,21 @@ export function deriveLedgerFacts(cwd, slug) {
633
633
  const readJson = (p) => { try { return JSON.parse(readFileSync(p, "utf8")); } catch { return null; } };
634
634
  const evalRows = [];
635
635
  const rDir = resultsDir(cwd, slug);
636
+ // THIS RUN'S VERDICTS ONLY. A run opened afresh over a slug that already ran keeps the earlier
637
+ // run's `evaluate-r<N>.json` beside its own, and a later-numbered one from the earlier run was read
638
+ // as this run's final verdict: measured, a run whose last round PASSed closed as `final_verdict:
639
+ // FAIL` from a previous run's round 3. A result carries no run key; its order does.
640
+ const runId = readRunId(cwd, slug);
641
+ const thisRun = (n) => {
642
+ if (!runId) return true;
643
+ const o = readJson(join(ordersDir(cwd, slug), `evaluate-r${n}.json`));
644
+ return !o?.run_id || o.run_id === runId;
645
+ };
636
646
  if (existsSync(rDir)) {
637
647
  for (const f of readdirSync(rDir)) {
638
648
  const m = f.match(/^evaluate-r(\d+)\.json$/);
639
649
  if (!m) continue;
650
+ if (!thisRun(m[1])) continue;
640
651
  const r = readJson(join(rDir, f));
641
652
  const overall = r?.verdict?.overall;
642
653
  if (overall) evalRows.push({ round: Number(m[1]), overall: String(overall), criteria: Array.isArray(r.verdict.criteria) ? r.verdict.criteria : [] });
@@ -649,7 +660,11 @@ export function deriveLedgerFacts(cwd, slug) {
649
660
  if (existsSync(gp)) {
650
661
  for (const line of readFileSync(gp, "utf8").split("\n")) {
651
662
  if (!line.trim()) continue;
652
- try { decisions.push(JSON.parse(line)); } catch { /* a torn line proves nothing */ }
663
+ try {
664
+ const g = JSON.parse(line);
665
+ // Gate rows carry the run key; a row from an earlier run over the same slug is its history.
666
+ if (!runId || !g?.run_id || g.run_id === runId) decisions.push(g);
667
+ } catch { /* a torn line proves nothing */ }
653
668
  }
654
669
  }
655
670
  const t0ByRound = {};
@@ -658,6 +673,7 @@ export function deriveLedgerFacts(cwd, slug) {
658
673
  for (const f of readdirSync(vDir).filter((x) => x.endsWith(".json")).sort()) {
659
674
  const v = readJson(join(vDir, f));
660
675
  if (typeof v?.round !== "number") continue;
676
+ if (runId && v.run_id && v.run_id !== runId) continue;
661
677
  const t = (t0ByRound[v.round] ||= { green: 0, red: 0 });
662
678
  if (v.overall === "green") t.green++; else t.red++;
663
679
  }
@@ -667,7 +683,10 @@ export function deriveLedgerFacts(cwd, slug) {
667
683
  if (existsSync(bDir)) {
668
684
  for (const f of readdirSync(bDir).sort()) {
669
685
  const m = f.match(/^r(\d+)-t\d+\.json$/);
670
- if (m) gateByRound[Number(m[1])] = readJson(join(bDir, f))?.overall ?? "?";
686
+ if (!m) continue;
687
+ const g = readJson(join(bDir, f));
688
+ if (runId && g?.run_id && g.run_id !== runId) continue;
689
+ gateByRound[Number(m[1])] = g?.overall ?? "?";
671
690
  }
672
691
  }
673
692
  const roundNums = [...new Set([...Object.keys(t0ByRound), ...Object.keys(gateByRound), ...evalRows.map((e) => e.round)].map(Number))].sort((a, b) => a - b);
@@ -380,6 +380,8 @@
380
380
  "spec_folder",
381
381
  "eval_report",
382
382
  "app_url",
383
+ "launch_cmd",
384
+ "build_gate",
383
385
  "ledger",
384
386
  "kb_rules_path"
385
387
  ],
@@ -2428,7 +2430,7 @@
2428
2430
  },
2429
2431
  "launch_cmd": {
2430
2432
  "type": "string",
2431
- "description": "spec-evaluator: the project profile's launch probe — installs the built artifact, starts it and asserts the first screen. Derived by `harness compile` from project-profile.md, absent when the profile declares none. Where `run_cmd` is only a build, this is how the app is brought up; a non-zero exit is a finding, not a reason to guess another way."
2433
+ "description": "spec-evaluator, qa-edge-hunter: the project profile's launch probe — installs the built artifact, starts it and asserts the first screen. Derived by `harness compile` from project-profile.md, absent when the profile declares none. Where `run_cmd` is only a build, this is how the app is brought up; a non-zero exit is a finding, not a reason to guess another way."
2432
2434
  },
2433
2435
  "revised_checks": {
2434
2436
  "type": "array",
@@ -2445,7 +2447,7 @@
2445
2447
  },
2446
2448
  "build_gate": {
2447
2449
  "type": "string",
2448
- "description": "spec-evaluator: this run's newest round build gate artifact (build/r<N>-t<T>.json) — each step's exit code and output tail, the record that the build ran and the app launched. Derived by `harness compile`, absent when the gate never ran. Read it before grading a [ui] row NO EVIDENCE."
2450
+ "description": "spec-evaluator, qa-edge-hunter: this run's newest round build gate artifact (build/r<N>-t<T>.json) — each step's exit code and output tail, the record that the build ran and the app launched. Derived by `harness compile`, absent when the gate never ran. Read it before grading a [ui] row NO EVIDENCE."
2449
2451
  },
2450
2452
  "t0_artifacts": {
2451
2453
  "type": "array",
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "shapeup-sdlc",
3
- "version": "3.14.0",
3
+ "version": "3.15.0",
4
4
  "description": "Shape Up for coding agents \u2014 with gates the agent can't talk its way past. Harness for Claude Code.",
5
5
  "bin": {
6
6
  "shapeup-sdlc": "bin/init.mjs"
@@ -39,10 +39,13 @@ tech-lead: ... GATE L2 → EVAL → GATE L3 PASS ──► QA EDGE HUNT (you)
39
39
 
40
40
  Pure worker (harness rule: stateless workers, one stateful orchestrator). Its WorkOrder
41
41
  carries `payload.feature`, `payload.spec_folder`, `payload.eval_report`, `payload.app_url`,
42
- `payload.kb_rules_path`, and `payload.ledger` (the discovery ledger, READ-ONLY — covered-territory
42
+ `payload.launch_cmd`, `payload.build_gate`, `payload.kb_rules_path`, and `payload.ledger` (the discovery ledger, READ-ONLY — covered-territory
43
43
  context so a hunt does not re-report what is already known). **`app_url` is null when the
44
- deliverable is not served over HTTP** — a CLI, a library, a batch job. That is a normal order, not a
45
- malformed one: drive the built entry point instead, exactly as the Test Surface's process rows do.
44
+ deliverable is not served over HTTP** — a CLI, a library, a batch job, a mobile app. That is a normal
45
+ order, not a malformed one: drive the built entry point instead, exactly as the Test Surface's process
46
+ rows do. When the order carries `launch_cmd`, that is how the app is brought up — run it, and treat a
47
+ zero exit as the deliverable reached; `build_gate` records that the round's gate already launched it.
48
+ Drive it with the tools the project's knowledge base names, by the path it gives.
46
49
  Do not refuse the hunt, and do not invent a URL. Its write surface is
47
50
  `.shapeup/<feature>/qa/**` only. The Hunter never touches the discovery ledger itself —
48
51
  ingest appends its `discoveries[]` under a `## Discovered` section, preserving single-writer
@@ -68,7 +71,8 @@ Phase Q3 │ Report ───────► qa/hunt-report.md — no score, no
68
71
 
69
72
  ```
70
73
  HARD (any miss → STOP, report which):
71
- ✅ app reachable at the given URL (one real request, not a ping)
74
+ ✅ deliverable reachable: one real request at `app_url`, or `launch_cmd` run and exit 0, or one
75
+ real invocation of the entry point (not a ping, not a guess that a tool is missing)
72
76
  ✅ EVAL-FEATURE-<slug>.md exists with verdict: PASS
73
77
  ✅ if discovery/ledger.md exists: ledger.feature == <feature> (read-only context check —
74
78
  a missing ledger is fine; ingest creates it when your findings land)