shapeup-sdlc 3.7.6 → 3.7.7

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "name": "shapeup-sdlc-plugin",
3
3
  "displayName": "ShapeUp SDLC Plugin",
4
- "version": "3.7.6",
4
+ "version": "3.7.7",
5
5
  "description": "Shape Up SDLC harness for Claude Code: shaping, intake, orient, scope-mapping, building (T0-verified, sandboxed, scope-contracted), evaluation and QA skills orchestrated by a tech-lead.",
6
6
  "author": {
7
7
  "name": "Liberty Nguyen",
package/kernel/gate.mjs CHANGED
@@ -54,7 +54,7 @@ import { readFileSync, writeFileSync, appendFileSync, existsSync, mkdirSync } fr
54
54
  import { parseBoard } from "./reduce/board.mjs";
55
55
  import { join, dirname } from "node:path";
56
56
  import { runArgs } from "./lib/argv.mjs";
57
- import { gateAnswerCandidates, gates as gatesPath, LOCAL, resultsDir, tasksDir, hammerCensus } from "./lib/paths.mjs";
57
+ import { gateAnswerCandidates, gates as gatesPath, LOCAL, resultsDir, tasksDir, hammerCensus, readRunId } from "./lib/paths.mjs";
58
58
 
59
59
  export const GATE_IDS = ["L0", "L1a", "L1a.5", "L1b", "L2", "L3", "QA", "H", "L4", "COACH-1"];
60
60
 
@@ -465,7 +465,11 @@ export function cli(rawArgv) {
465
465
  // per-run ledger row — the same reasoning `resolveRunId` uses for "no run is active": absence is
466
466
  // the correct answer, not an error, so the write is skipped rather than guessing a location.
467
467
  if (args.slug) {
468
+ // THE ROW CARRIES THE RUN KEY. It did not, and the export stamped the current run's key onto
469
+ // every row it found — a prior run's sign-off became this run's in the one table that answers
470
+ // "was this ship signed off". Driven on a two-run fixture before it was fixed.
468
471
  appendGateLedger(cwd, args.slug, {
472
+ at: new Date().toISOString(), run_id: readRunId(cwd, args.slug),
469
473
  gate: r.gate, status: r.status, decision: r.decision ?? null,
470
474
  source: r.source ?? found.source, note: r.note ?? r.reason ?? null,
471
475
  round: args.round ?? null,
@@ -35,7 +35,7 @@ import { existsSync, readFileSync, readdirSync } from "node:fs";
35
35
  import { join, resolve } from "node:path";
36
36
  import { createHash } from "node:crypto";
37
37
  import { runArgs } from "../lib/argv.mjs";
38
- import { resultsDir, scopesDir } from "../lib/paths.mjs";
38
+ import { resultsDir, scopesDir, readRunId } from "../lib/paths.mjs";
39
39
 
40
40
  /** Longest `reason` reported. A deviation is prose written by a worker and can run to paragraphs. */
41
41
  const REASON_MAX = 400;
@@ -89,7 +89,7 @@ export function isScoped(cwd, slug) {
89
89
  * @returns {(string|null)} A reason phrased for an operator, or null when the file at `path` exists,
90
90
  * hashes to the cited `sha256`, and its own `overall` reads "green".
91
91
  */
92
- function unresolvedCitation(cwd, citation) {
92
+ function unresolvedCitation(cwd, citation, { round = null, runId = null } = {}) {
93
93
  const rel = typeof citation?.path === "string" ? citation.path : "";
94
94
  if (!rel) return "names no artifact path";
95
95
  let text;
@@ -109,6 +109,20 @@ function unresolvedCitation(cwd, citation) {
109
109
  try { body = JSON.parse(text); }
110
110
  catch { return `cites ${rel}, whose bytes match the hash but do not read as a T0 verdict`; }
111
111
  if (body?.overall !== "green") return `cites ${rel}, whose own verdict is "${body?.overall ?? "unknown"}", not green`;
112
+ // THE ARTIFACT HAS TO BE THE ONE THE CITATION SAYS IT IS. A re-hash proves the bytes are the
113
+ // file's; it says nothing about whose verdict the file holds. A PASS citing scope alpha's green
114
+ // artifact while declaring scope beta, or a prior round's, or a prior run's over the same slug,
115
+ // passed the digest check unremarked. The citation's own required `scope_id`, the round being
116
+ // judged and the run's key are compared to what the artifact records about itself.
117
+ if (typeof citation.scope_id === "string" && body?.scope_id && body.scope_id !== citation.scope_id) {
118
+ return `cites ${rel} for scope "${citation.scope_id}", but the artifact records scope "${body.scope_id}"`;
119
+ }
120
+ if (round != null && typeof body?.round === "number" && body.round !== round) {
121
+ return `cites ${rel}, a round ${body.round} artifact, as evidence for round ${round}`;
122
+ }
123
+ if (runId && body?.run_id && body.run_id !== runId) {
124
+ return `cites ${rel}, an artifact of run ${body.run_id}, as evidence for run ${runId}`;
125
+ }
112
126
  return null;
113
127
  }
114
128
 
@@ -143,15 +157,49 @@ function unresolvedCitation(cwd, citation) {
143
157
  * citation resolves, an unscoped spec, or a block with no PASS/FAIL in it (there is no judgement
144
158
  * to invalidate).
145
159
  */
146
- export function citationProblem(cwd, slug, verdict) {
160
+ /**
161
+ * Why a verdict cannot stand on its own criteria, or null when it can.
162
+ *
163
+ * `overall` is the judge's field, and nothing recomputed it from the criteria the judge graded: a
164
+ * PASS over a failing criterion, or over no criterion at all, validated and ingested, and the
165
+ * round loop branched on it. The evaluator's own first rule is that absence of evidence is a FAIL,
166
+ * so a PASS criterion with no evidence is no evidence either. Recomputed here, on ingest and on
167
+ * read alike: PASS means every graded criterion passed with evidence and at least one was graded;
168
+ * FAIL means at least one graded criterion failed.
169
+ *
170
+ * @param {object} verdict - The WorkResult's `verdict`.
171
+ * @returns {(string|null)} A reason phrased for an operator, or null.
172
+ */
173
+ export function verdictProblem(verdict) {
174
+ const overall = verdict?.overall;
175
+ if (overall !== "PASS" && overall !== "FAIL") return null;
176
+ const criteria = Array.isArray(verdict.criteria) ? verdict.criteria : [];
177
+ const fails = criteria.filter((c) => c?.verdict === "FAIL");
178
+ const passes = criteria.filter((c) => c?.verdict === "PASS");
179
+ const other = criteria.length - fails.length - passes.length;
180
+ if (overall === "PASS") {
181
+ if (criteria.length === 0) return "the PASS verdict grades no criterion at all — a PASS with no evidence is a claim";
182
+ if (fails.length) return `the verdict says PASS while ${fails.length} of its ${criteria.length} criteria read FAIL — overall is derived from the criteria, never declared over them`;
183
+ if (other) return `the verdict says PASS while ${other} of its criteria carry no PASS/FAIL verdict`;
184
+ const bare = passes.filter((c) => !(typeof c?.evidence === "string" && c.evidence.trim()));
185
+ if (bare.length) return `the PASS verdict has ${bare.length} criterion(s) marked PASS with no evidence — absence of evidence is a FAIL by the evaluator's own first rule`;
186
+ return null;
187
+ }
188
+ if (criteria.length === 0) return "the FAIL verdict grades no criterion at all — a FAIL must name what failed";
189
+ if (!fails.length) return `the verdict says FAIL while every one of its ${criteria.length} graded criteria reads PASS — a FAIL must cite the criterion it failed`;
190
+ return null;
191
+ }
192
+
193
+ export function citationProblem(cwd, slug, verdict, { round = null } = {}) {
147
194
  if (verdict?.overall !== "PASS" && verdict?.overall !== "FAIL") return null;
148
195
  if (!isScoped(cwd, slug)) return null;
149
196
  if (!Array.isArray(verdict.t0_citations) || !verdict.t0_citations.length) {
150
197
  return `the ${verdict.overall} verdict cites no T0 artifact, and a verdict on a scoped spec must ` +
151
198
  "cite the T0 verdict it re-hashed (the order lists them under payload.t0_artifacts)";
152
199
  }
200
+ const runId = readRunId(cwd, slug);
153
201
  for (const citation of verdict.t0_citations) {
154
- const reason = unresolvedCitation(cwd, citation);
202
+ const reason = unresolvedCitation(cwd, citation, { round, runId });
155
203
  if (reason) return `the ${verdict.overall} verdict ${reason} — a T0 citation is re-hashed from disk, never taken on the handed word`;
156
204
  }
157
205
  return null;
@@ -185,7 +233,7 @@ export function evalVerdict(cwd, slug, round) {
185
233
  ? `the evaluator returned ${status || "no status"}: ${first}`
186
234
  : `status ${status || "unknown"} with no PASS/FAIL verdict`), status);
187
235
  }
188
- const problem = citationProblem(cwd, slug, v);
236
+ const problem = verdictProblem(v) || citationProblem(cwd, slug, v, { round });
189
237
  if (problem) return unfit(problem, status, overall);
190
238
  return {
191
239
  found: true,
@@ -22,7 +22,7 @@
22
22
 
23
23
  import { existsSync, readdirSync, readFileSync } from "node:fs";
24
24
  import { join } from "node:path";
25
- import { ordersDir, verdictsDir, roundBuildDir, resultsDir } from "../lib/paths.mjs";
25
+ import { ordersDir, verdictsDir, roundBuildDir, resultsDir, readRunId } from "../lib/paths.mjs";
26
26
 
27
27
  /** Parse a JSON file, returning null rather than throwing — every reader here is best-effort. */
28
28
  function readJson(p) {
@@ -59,12 +59,22 @@ const maxOf = (nums) => (nums.length ? Math.max(...nums) : null);
59
59
  * @returns {{rounds_used:*, rounds_judged:(number|null)}} Both counts.
60
60
  */
61
61
  export function deriveRounds(cwd, slug, fallback) {
62
+ // THIS RUN'S RECORDS ONLY. Orders, verdicts and build gates over one slug accumulate across runs
63
+ // and carry their run key; walking the directories unfiltered handed a run that had dispatched
64
+ // nothing a prior run's round count — into the committed report. A record with no key at all was
65
+ // written before the key existed and is kept; one with a different key is another run's.
66
+ const runId = readRunId(cwd, slug);
67
+ const mine = (rec) => !runId || !rec?.run_id || rec.run_id === runId;
62
68
  const orderRounds = [];
63
69
  const oDir = ordersDir(cwd, slug);
70
+ const orderOf = {};
64
71
  if (existsSync(oDir)) {
65
72
  for (const f of readdirSync(oDir)) {
66
73
  if (!f.endsWith(".json")) continue;
67
- const r = orderRound(readJson(join(oDir, f))?.order_id);
74
+ const o = readJson(join(oDir, f));
75
+ orderOf[f] = o;
76
+ if (!mine(o)) continue;
77
+ const r = orderRound(o?.order_id);
68
78
  if (r !== null) orderRounds.push(r);
69
79
  }
70
80
  }
@@ -74,7 +84,7 @@ export function deriveRounds(cwd, slug, fallback) {
74
84
  if (existsSync(vDir)) {
75
85
  for (const f of readdirSync(vDir).filter((x) => x.endsWith(".json"))) {
76
86
  const v = readJson(join(vDir, f));
77
- if (typeof v?.round === "number") verdictRounds.push(v.round);
87
+ if (mine(v) && typeof v?.round === "number") verdictRounds.push(v.round);
78
88
  }
79
89
  }
80
90
 
@@ -83,7 +93,7 @@ export function deriveRounds(cwd, slug, fallback) {
83
93
  if (existsSync(bDir)) {
84
94
  for (const f of readdirSync(bDir)) {
85
95
  const m = f.match(/^r(\d+)-t\d+\.json$/);
86
- if (m) buildGateRounds.push(Number(m[1]));
96
+ if (m && mine(readJson(join(bDir, f)))) buildGateRounds.push(Number(m[1]));
87
97
  }
88
98
  }
89
99
 
@@ -92,7 +102,8 @@ export function deriveRounds(cwd, slug, fallback) {
92
102
  if (existsSync(rDir)) {
93
103
  for (const f of readdirSync(rDir)) {
94
104
  const m = f.match(/^evaluate-r(\d+)\.json$/);
95
- if (m) evalRounds.push(Number(m[1]));
105
+ // A WorkResult carries no run key; it reaches one through its order of the same name.
106
+ if (m && mine(orderOf[f] ?? readJson(join(oDir, f)))) evalRounds.push(Number(m[1]));
96
107
  }
97
108
  }
98
109
 
@@ -37,7 +37,7 @@ import { fileURLToPath } from "node:url";
37
37
  import { validate } from "../verify/envelope.mjs";
38
38
  import { runArgs } from "../lib/argv.mjs";
39
39
  import { tasksDir, localRoot, dispatchReceipts, legLedger, readRunId } from "../lib/paths.mjs";
40
- import { citationProblem } from "../probe/eval.mjs";
40
+ import { citationProblem, verdictProblem } from "../probe/eval.mjs";
41
41
 
42
42
  const HERE = dirname(fileURLToPath(import.meta.url));
43
43
  const RESULT_SCHEMA = JSON.parse(readFileSync(resolve(HERE, "../schemas/work-result.schema.json"), "utf8"));
@@ -697,7 +697,8 @@ export async function cli(rawArgv) {
697
697
  // on (see `citationProblem`). `probe eval` refuses it to the round loop; refusing it here as well
698
698
  // keeps the verdict ledger from recording a verdict the loop will never branch on.
699
699
  if (result.verdict) {
700
- const problem = citationProblem(cwd, String(result.order_id).split("/")[0], result.verdict);
700
+ const evalRound = Number((String(result.order_id).match(/-r(\d+)$/) || [])[1]) || null;
701
+ const problem = verdictProblem(result.verdict) || citationProblem(cwd, String(result.order_id).split("/")[0], result.verdict, { round: evalRound });
701
702
  if (problem) {
702
703
  console.error(`ingest-result: result refused — ${problem}.`);
703
704
  console.error(` The round stays open: re-dispatch the evaluator against its order, which lists`);
@@ -329,7 +329,13 @@ export function buildReport(facts) {
329
329
  "*Run state (board, orders, results, T0 artifacts, evaluation and QA reports) stays in the",
330
330
  "gitignored local tier (ADR-0001). This report",
331
331
  "is the frozen conclusion of it.*", "");
332
- return L.join("\n");
332
+ // THE WHOLE REPORT, ONCE. Board ids were anchored in three places and leaked through a fourth: a
333
+ // discovery-ledger entry copied verbatim into "Discovered, not built" carried `[TASK-005]`, this
334
+ // kernel wrote it straight to the committed tier past the hook that guards the model's edits, and
335
+ // the next run's L1b lint red'd the file the previous run had frozen. A committed report may not
336
+ // name a board id anywhere, so the rule is applied to the finished text rather than section by
337
+ // section.
338
+ return deboard(L.join("\n"), board.anchors);
333
339
  }
334
340
 
335
341
  /**
@@ -173,12 +173,14 @@ function criterionRows(dir, runId, t) {
173
173
  * (see `appendGateLedger`), so this is a pass-through with a stamped `run_id` fallback rather than
174
174
  * a re-derivation: two readers of "what did this gate decide" must not compute the answer twice.
175
175
  * @param {object} g - One parsed line of `gates.jsonl`.
176
- * @param {(string|null)} runId - Run key for a row written before it carried its own.
176
+ * @param {(string|null)} runId - The run being exported (unused for attribution — the row's own key is the only one exported).
177
177
  * @returns {object} A flat `gate_decision` row.
178
178
  */
179
179
  function gateDecisionRow(g, runId) {
180
180
  return {
181
- run_id: g?.run_id ?? runId ?? null,
181
+ // The row's own key, never the current run's stamped on: a row that carries no key was written
182
+ // before the ledger did, and is exported as unattributed rather than claimed.
183
+ run_id: g?.run_id ?? null,
182
184
  gate: g?.gate ?? null,
183
185
  decision: g?.decision ?? null,
184
186
  status: g?.status ?? null,
@@ -260,7 +262,9 @@ export function collectRun(cwd, slug) {
260
262
  criterion_verdict: criterionRows(evaluationDir(cwd, slug), runId, t),
261
263
  hook_decision,
262
264
  // The decision that crossed each gate, and the round build gate's own artifact.
263
- gate_decision: readJsonl(gatesPath(cwd, slug), t).map((g) => gateDecisionRow(g, runId)),
265
+ // Scoped to the run, like hook_decision one line up: gate rows over one slug accumulate across
266
+ // runs, and a prior run's L4 exported under this run's key is a fabricated sign-off.
267
+ gate_decision: readJsonl(gatesPath(cwd, slug), t).filter((g) => runId && g?.run_id === runId).map((g) => gateDecisionRow(g, runId)),
264
268
  build_gate: readJsonDir(roundBuildDir(cwd, slug), t).map((a) => buildGateRow(a, runId)),
265
269
  leg: readJsonl(legLedger(cwd, slug), t)
266
270
  .filter((r) => !runId || !r?.run_id || r.run_id === runId)
@@ -40,7 +40,7 @@ import { resolve, join, dirname, relative, isAbsolute } from "node:path";
40
40
  import { readBoard } from "../compile.mjs";
41
41
  import { runArgs } from "../lib/argv.mjs";
42
42
  import { sharedRoot, traceDir, relLocal } from "../lib/paths.mjs";
43
- import { readContract, unreadableReason, LEGACY_LAYOUT, WIRING_MAP, PROJECT_PROFILE } from "../lib/contract.mjs";
43
+ import { readContract, unreadableReason, LEGACY_LAYOUT, WIRING_MAP, PROJECT_PROFILE, reqId } from "../lib/contract.mjs";
44
44
 
45
45
  // --- requirements.md registry parser -----------------------------------------
46
46
  // A committed markdown table: | REQ-id | clause (verbatim) | source | status | note |
@@ -87,7 +87,10 @@ export function coveredReqIds(board) {
87
87
  for (const task of board) {
88
88
  for (const ac of task.acceptance_criteria || []) {
89
89
  const covers = typeof ac === "object" && Array.isArray(ac.covers) ? ac.covers : [];
90
- for (const id of covers) if (/^REQ-\d+$/.test(id)) covered.add(id);
90
+ // ONE KEY SPACE. `R-2`, `[[REQ-5]]` and `req-4` are the same clause spelled three ways, and the
91
+ // sibling rule accepts all of them; testing the raw string here counted every one as nothing, so
92
+ // an author told at L1b to cover a requirement with an AC — which they had — stayed red.
93
+ for (const raw of covers) { const id = reqId(raw); if (/^REQ-\d+$/.test(id)) covered.add(id); }
91
94
  }
92
95
  }
93
96
  return covered;
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "shapeup-sdlc",
3
- "version": "3.7.6",
3
+ "version": "3.7.7",
4
4
  "description": "Shape Up for coding agents \u2014 with gates the agent can't talk its way past. Harness for Claude Code.",
5
5
  "bin": {
6
6
  "shapeup-sdlc": "bin/init.mjs"