shapeup-sdlc 3.14.1 → 3.15.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "name": "shapeup-sdlc-plugin",
3
3
  "displayName": "ShapeUp SDLC Plugin",
4
- "version": "3.14.1",
4
+ "version": "3.15.1",
5
5
  "description": "Shape Up SDLC harness for Claude Code: shaping, intake, orient, scope-mapping, building (T0-verified, sandboxed, scope-contracted), evaluation and QA skills orchestrated by a tech-lead.",
6
6
  "author": {
7
7
  "name": "Liberty Nguyen",
@@ -1279,7 +1279,10 @@ export async function cli(rawArgv) {
1279
1279
  const revised = revisedChecksFor(cwd, slug, round);
1280
1280
  if (revised.length) payloadExtra.revised_checks = revised;
1281
1281
  }
1282
- if (operation === "evaluate") {
1282
+ // The QA hunt drives the same running app the judge grades, so it needs the same way in. A mobile
1283
+ // deliverable has no URL, and a hunt order carrying only `app_url: null` reported "no reachable
1284
+ // deliverable" and hunted nothing over a build the round gate had just launched.
1285
+ if (operation === "evaluate" || operation === "hunt") {
1283
1286
  const ev = launchEvidenceFor(cwd, slug, round);
1284
1287
  if (ev.build_gate !== undefined && payloadExtra.build_gate === undefined) payloadExtra.build_gate = ev.build_gate;
1285
1288
  if (ev.launch_cmd !== undefined && payloadExtra.launch_cmd === undefined) payloadExtra.launch_cmd = ev.launch_cmd;
@@ -633,10 +633,21 @@ export function deriveLedgerFacts(cwd, slug) {
633
633
  const readJson = (p) => { try { return JSON.parse(readFileSync(p, "utf8")); } catch { return null; } };
634
634
  const evalRows = [];
635
635
  const rDir = resultsDir(cwd, slug);
636
+ // THIS RUN'S VERDICTS ONLY. A run opened afresh over a slug that already ran keeps the earlier
637
+ // run's `evaluate-r<N>.json` beside its own, and a later-numbered one from the earlier run was read
638
+ // as this run's final verdict: measured, a run whose last round PASSed closed as `final_verdict:
639
+ // FAIL` from a previous run's round 3. A result carries no run key; its order does.
640
+ const runId = readRunId(cwd, slug);
641
+ const thisRun = (n) => {
642
+ if (!runId) return true;
643
+ const o = readJson(join(ordersDir(cwd, slug), `evaluate-r${n}.json`));
644
+ return !o?.run_id || o.run_id === runId;
645
+ };
636
646
  if (existsSync(rDir)) {
637
647
  for (const f of readdirSync(rDir)) {
638
648
  const m = f.match(/^evaluate-r(\d+)\.json$/);
639
649
  if (!m) continue;
650
+ if (!thisRun(m[1])) continue;
640
651
  const r = readJson(join(rDir, f));
641
652
  const overall = r?.verdict?.overall;
642
653
  if (overall) evalRows.push({ round: Number(m[1]), overall: String(overall), criteria: Array.isArray(r.verdict.criteria) ? r.verdict.criteria : [] });
@@ -649,7 +660,11 @@ export function deriveLedgerFacts(cwd, slug) {
649
660
  if (existsSync(gp)) {
650
661
  for (const line of readFileSync(gp, "utf8").split("\n")) {
651
662
  if (!line.trim()) continue;
652
- try { decisions.push(JSON.parse(line)); } catch { /* a torn line proves nothing */ }
663
+ try {
664
+ const g = JSON.parse(line);
665
+ // Gate rows carry the run key; a row from an earlier run over the same slug is its history.
666
+ if (!runId || !g?.run_id || g.run_id === runId) decisions.push(g);
667
+ } catch { /* a torn line proves nothing */ }
653
668
  }
654
669
  }
655
670
  const t0ByRound = {};
@@ -658,6 +673,7 @@ export function deriveLedgerFacts(cwd, slug) {
658
673
  for (const f of readdirSync(vDir).filter((x) => x.endsWith(".json")).sort()) {
659
674
  const v = readJson(join(vDir, f));
660
675
  if (typeof v?.round !== "number") continue;
676
+ if (runId && v.run_id && v.run_id !== runId) continue;
661
677
  const t = (t0ByRound[v.round] ||= { green: 0, red: 0 });
662
678
  if (v.overall === "green") t.green++; else t.red++;
663
679
  }
@@ -667,7 +683,10 @@ export function deriveLedgerFacts(cwd, slug) {
667
683
  if (existsSync(bDir)) {
668
684
  for (const f of readdirSync(bDir).sort()) {
669
685
  const m = f.match(/^r(\d+)-t\d+\.json$/);
670
- if (m) gateByRound[Number(m[1])] = readJson(join(bDir, f))?.overall ?? "?";
686
+ if (!m) continue;
687
+ const g = readJson(join(bDir, f));
688
+ if (runId && g?.run_id && g.run_id !== runId) continue;
689
+ gateByRound[Number(m[1])] = g?.overall ?? "?";
671
690
  }
672
691
  }
673
692
  const roundNums = [...new Set([...Object.keys(t0ByRound), ...Object.keys(gateByRound), ...evalRows.map((e) => e.round)].map(Number))].sort((a, b) => a - b);
@@ -370,6 +370,21 @@ export function buildReport(facts) {
370
370
  return deboard(L.join("\n"), board.anchors);
371
371
  }
372
372
 
373
+ /**
374
+ * The QA line the report may print. The run says whether it dispatched the hunt; the hunt's own
375
+ * report says whether anything was hunted. A hunter that could not reach the app still returns, and
376
+ * a report opening `charters: 0/…` over a dispatched hunt read as "QA: run" with no findings — which
377
+ * a reader takes for an app with nothing wrong, not for an app nobody drove.
378
+ * @param {string|undefined} passed - What the caller recorded (`run`, `skipped`), if anything.
379
+ * @param {string|null} huntReport - The hunt report's text, or null when there is none.
380
+ * @returns {string} `skipped`, `not-hunted` (dispatched, zero charters run), or `run`.
381
+ */
382
+ export function qaStatus(passed, huntReport) {
383
+ if (passed === "skipped" || (!passed && !huntReport)) return passed || "skipped";
384
+ if (huntReport && /^charters:\s*0\s*\//m.test(huntReport)) return "not-hunted";
385
+ return passed || "run";
386
+ }
387
+
373
388
  /**
374
389
  * Gather every fact from disk and render the report.
375
390
  * @param {{cwd:string, slug:string, verdict?:string, qa?:string}} opts - Inputs.
@@ -394,7 +409,7 @@ export function generate({ cwd, slug, verdict, qa }) {
394
409
  at: today(),
395
410
  verdict: verdict || run.final_verdict || "not-evaluated",
396
411
  census: (() => { try { return JSON.parse(readIf(hammerCensus(cwd, slug)) || "null"); } catch { return null; } })(),
397
- qa: qa || (huntReport ? "run" : "skipped"),
412
+ qa: qaStatus(qa, huntReport),
398
413
  rounds: derivedRounds.rounds_used,
399
414
  roundsJudged: derivedRounds.rounds_judged,
400
415
  intakeSha: receipt.intake_sha256,
@@ -380,6 +380,8 @@
380
380
  "spec_folder",
381
381
  "eval_report",
382
382
  "app_url",
383
+ "launch_cmd",
384
+ "build_gate",
383
385
  "ledger",
384
386
  "kb_rules_path"
385
387
  ],
@@ -2428,7 +2430,7 @@
2428
2430
  },
2429
2431
  "launch_cmd": {
2430
2432
  "type": "string",
2431
- "description": "spec-evaluator: the project profile's launch probe — installs the built artifact, starts it and asserts the first screen. Derived by `harness compile` from project-profile.md, absent when the profile declares none. Where `run_cmd` is only a build, this is how the app is brought up; a non-zero exit is a finding, not a reason to guess another way."
2433
+ "description": "spec-evaluator, qa-edge-hunter: the project profile's launch probe — installs the built artifact, starts it and asserts the first screen. Derived by `harness compile` from project-profile.md, absent when the profile declares none. Where `run_cmd` is only a build, this is how the app is brought up; a non-zero exit is a finding, not a reason to guess another way."
2432
2434
  },
2433
2435
  "revised_checks": {
2434
2436
  "type": "array",
@@ -2445,7 +2447,7 @@
2445
2447
  },
2446
2448
  "build_gate": {
2447
2449
  "type": "string",
2448
- "description": "spec-evaluator: this run's newest round build gate artifact (build/r<N>-t<T>.json) — each step's exit code and output tail, the record that the build ran and the app launched. Derived by `harness compile`, absent when the gate never ran. Read it before grading a [ui] row NO EVIDENCE."
2450
+ "description": "spec-evaluator, qa-edge-hunter: this run's newest round build gate artifact (build/r<N>-t<T>.json) — each step's exit code and output tail, the record that the build ran and the app launched. Derived by `harness compile`, absent when the gate never ran. Read it before grading a [ui] row NO EVIDENCE."
2449
2451
  },
2450
2452
  "t0_artifacts": {
2451
2453
  "type": "array",
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "shapeup-sdlc",
3
- "version": "3.14.1",
3
+ "version": "3.15.1",
4
4
  "description": "Shape Up for coding agents \u2014 with gates the agent can't talk its way past. Harness for Claude Code.",
5
5
  "bin": {
6
6
  "shapeup-sdlc": "bin/init.mjs"
@@ -39,10 +39,13 @@ tech-lead: ... GATE L2 → EVAL → GATE L3 PASS ──► QA EDGE HUNT (you)
39
39
 
40
40
  Pure worker (harness rule: stateless workers, one stateful orchestrator). Its WorkOrder
41
41
  carries `payload.feature`, `payload.spec_folder`, `payload.eval_report`, `payload.app_url`,
42
- `payload.kb_rules_path`, and `payload.ledger` (the discovery ledger, READ-ONLY — covered-territory
42
+ `payload.launch_cmd`, `payload.build_gate`, `payload.kb_rules_path`, and `payload.ledger` (the discovery ledger, READ-ONLY — covered-territory
43
43
  context so a hunt does not re-report what is already known). **`app_url` is null when the
44
- deliverable is not served over HTTP** — a CLI, a library, a batch job. That is a normal order, not a
45
- malformed one: drive the built entry point instead, exactly as the Test Surface's process rows do.
44
+ deliverable is not served over HTTP** — a CLI, a library, a batch job, a mobile app. That is a normal
45
+ order, not a malformed one: drive the built entry point instead, exactly as the Test Surface's process
46
+ rows do. When the order carries `launch_cmd`, that is how the app is brought up — run it, and treat a
47
+ zero exit as the deliverable reached; `build_gate` records that the round's gate already launched it.
48
+ Drive it with the tools the project's knowledge base names, by the path it gives.
46
49
  Do not refuse the hunt, and do not invent a URL. Its write surface is
47
50
  `.shapeup/<feature>/qa/**` only. The Hunter never touches the discovery ledger itself —
48
51
  ingest appends its `discoveries[]` under a `## Discovered` section, preserving single-writer
@@ -67,8 +70,13 @@ Phase Q3 │ Report ───────► qa/hunt-report.md — no score, no
67
70
  ## GATE Q0 — Preflight
68
71
 
69
72
  ```
73
+ FIRST: read `payload.kb_rules_path` for the tool paths it names — a device tool that is not on PATH is
74
+ named there by its full path, and a bare-name lookup (`which`, a bare call) proves nothing.
70
75
  HARD (any miss → STOP, report which):
71
- ✅ app reachable at the given URL (one real request, not a ping)
76
+ ✅ deliverable reachable: one real request at `app_url`, or `launch_cmd` run and exit 0, or one
77
+ real invocation of the entry point (not a ping, not a guess that a tool is missing). When the
78
+ order carries `launch_cmd`, run it before concluding anything; the report names the command
79
+ and its exit. A hunt that reached nothing returns `status: failed`, never `done`.
72
80
  ✅ EVAL-FEATURE-<slug>.md exists with verdict: PASS
73
81
  ✅ if discovery/ledger.md exists: ledger.feature == <feature> (read-only context check —
74
82
  a missing ledger is fine; ingest creates it when your findings land)