shapeup-sdlc 3.4.0 → 3.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (43) hide show
  1. package/.claude-plugin/plugin.json +1 -1
  2. package/AGENTS.md +16 -4
  3. package/README.md +7 -3
  4. package/SECURITY.md +1 -1
  5. package/hooks/sandbox-guard.mjs +69 -6
  6. package/kernel/compile.mjs +33 -12
  7. package/kernel/harness.mjs +10 -4
  8. package/kernel/init/run-args.mjs +206 -0
  9. package/kernel/init/run.mjs +10 -0
  10. package/kernel/lib/contract.mjs +68 -1
  11. package/kernel/lib/paths.mjs +10 -0
  12. package/kernel/probe/concurrency.mjs +31 -6
  13. package/kernel/probe/digest.mjs +15 -1
  14. package/kernel/probe/owner.mjs +4 -1
  15. package/kernel/probe/requirements.mjs +296 -0
  16. package/kernel/probe/resume.mjs +195 -6
  17. package/kernel/probe/rounds.mjs +104 -0
  18. package/kernel/reduce/graph.mjs +5 -2
  19. package/kernel/reduce/ingest.mjs +69 -15
  20. package/kernel/reduce/ship.mjs +52 -31
  21. package/kernel/reduce/snapshot.mjs +23 -2
  22. package/kernel/report/export.mjs +54 -2
  23. package/kernel/report/facts.mjs +24 -2
  24. package/{skills/tech-lead → kernel}/schemas/domain.schema.json +20 -12
  25. package/kernel/verify/envelope.mjs +2 -2
  26. package/kernel/verify/skills.mjs +1 -1
  27. package/kernel/verify/spec.mjs +130 -4
  28. package/kernel/verify/trace.mjs +16 -7
  29. package/package.json +1 -1
  30. package/skills/ba-pitch-analyzer/SKILL.md +16 -1
  31. package/skills/coach/SKILL.md +8 -2
  32. package/skills/hill-chart/SKILL.md +3 -4
  33. package/skills/scope-architect/SKILL.md +16 -1
  34. package/skills/scope-hammer/SKILL.md +11 -2
  35. package/skills/spec-evaluator/SKILL.md +12 -1
  36. package/skills/tech-lead/SKILL.md +10 -10
  37. package/skills/tech-lead/references/gates.md +70 -12
  38. package/skills/tech-lead/references/protocol.md +4 -2
  39. package/skills/tech-lead/workflows/shapeup-run.js +176 -38
  40. package/skills/translator/SKILL.md +1 -1
  41. /package/{skills/tech-lead → kernel}/schemas/gate-answers.schema.json +0 -0
  42. /package/{skills/tech-lead → kernel}/schemas/work-order.schema.json +0 -0
  43. /package/{skills/tech-lead → kernel}/schemas/work-result.schema.json +0 -0
@@ -31,11 +31,13 @@ import { join, dirname } from "node:path";
31
31
  import { runArgs } from "../lib/argv.mjs";
32
32
  import {
33
33
  report as reportPath, tasksDir, verdictsDir, trials, evaluationDir, qaDir,
34
- roundLedger, discoveryLedger, receipt as receiptPath, harnessRun, relShared, resultsDir,
34
+ roundLedger, discoveryLedger, receipt as receiptPath, harnessRun, relShared,
35
35
  activeOrder,
36
36
  } from "../lib/paths.mjs";
37
37
  import { readTrials } from "../verify/t0.mjs";
38
38
  import { ratchetReport } from "../probe/stats.mjs";
39
+ import { projectRequirements, summaryLine } from "../probe/requirements.mjs";
40
+ import { deriveRounds } from "../probe/rounds.mjs";
39
41
  import { collectDiff, scanDiff, summarize } from "./leftovers.mjs";
40
42
 
41
43
  /** @returns {string} Today as `YYYY-MM-DD` (UTC). */
@@ -166,13 +168,13 @@ export function section(md, heading) {
166
168
  */
167
169
  export function buildReport(facts) {
168
170
  const {
169
- slug, at, verdict, qa, rounds, board, t0, artifacts, ratchet,
170
- evalCriteria, evalBugs, qaFindings, decisions, discovered, intakeSha, leftovers,
171
+ slug, at, verdict, qa, rounds, roundsJudged, board, t0, artifacts, ratchet,
172
+ evalCriteria, evalBugs, qaFindings, decisions, discovered, intakeSha, leftovers, requirements,
171
173
  } = facts;
172
174
 
173
175
  const L = [];
174
176
  L.push("---", "type: ship-report", `feature: ${slug}`, `date: ${at}`,
175
- `verdict: ${verdict}`, `rounds_used: ${rounds ?? "~"}`, `qa: ${qa}`,
177
+ `verdict: ${verdict}`, `rounds_used: ${rounds ?? "~"}`, `rounds_judged: ${roundsJudged ?? "~"}`, `qa: ${qa}`,
176
178
  `intake_sha256: ${intakeSha ?? "~"}`, "---", "");
177
179
  L.push(`# ${slug} — ship report`, "");
178
180
  L.push("Frozen at GATE L4. Every figure below is derived from run artifacts on disk — the trial",
@@ -182,6 +184,10 @@ export function buildReport(facts) {
182
184
  L.push("| | |", "|---|---|");
183
185
  L.push(`| Verdict | **${verdict}** |`);
184
186
  L.push(`| Rounds used | ${rounds ?? "—"} |`);
187
+ // A round built and a round judged are different facts — a round can die before EVAL ever sees
188
+ // it, so this row is its own line rather than folded into "Rounds used" above. Omitted when EVAL
189
+ // never ran at all, the same way the sections below it are.
190
+ if (roundsJudged != null) L.push(`| Rounds judged | ${roundsJudged} |`);
185
191
  L.push(`| Board | ${board.done}/${board.total} tasks done |`);
186
192
  L.push(`| T0 artifacts | ${artifacts} |`);
187
193
  L.push(`| QA | ${qa} |`);
@@ -214,6 +220,40 @@ export function buildReport(facts) {
214
220
  L.push("");
215
221
  }
216
222
 
223
+ // The requirement matrix — the way back from a verdict to the clause the pitch asked for, frozen
224
+ // at the one moment the run's local evidence still exists. Omitted entirely when the run has no
225
+ // registry: a table of nothing reads as "no requirements", which is a different claim from "this
226
+ // run predates the registry". Derived like every other figure here, by the same probe the L4 line
227
+ // and GATE H's census read, so the three cannot disagree.
228
+ if (requirements?.registry && requirements.rows.length) {
229
+ L.push("## Requirements", "");
230
+ L.push("One row per registered clause. A requirement has evidence when an acceptance criterion",
231
+ "covers it AND a criterion grading it passed — `covers:` is the join, the judge's anchor is the",
232
+ "path back. This is a projection, never a verdict: it never blocked this ship.", "");
233
+ L.push(`**${summaryLine(requirements)}** · run \`${requirements.run_id ?? "unknown"}\``, "");
234
+ // A clause, an AC and a criterion are all free prose, and a literal pipe in any of them breaks
235
+ // the row into columns nobody wrote — a frozen report that misrenders its own evidence.
236
+ /**
237
+ * Escape a free-prose value for a Markdown table cell.
238
+ * @param {string} s - The value.
239
+ * @returns {string} The value with every literal pipe escaped.
240
+ */
241
+ const cell = (s) => String(s).replace(/\|/g, "\\|");
242
+ L.push("| REQ | source | evidence | covering AC | criterion | T0 |", "|---|---|---|---|---|---|");
243
+ for (const r of requirements.rows) {
244
+ const ac = r.covering_acs.length ? `${r.covering_acs[0].task_id}: ${r.covering_acs[0].ac}${r.covering_acs.length > 1 ? ` (+${r.covering_acs.length - 1})` : ""}` : "—";
245
+ const crit = r.criteria.length ? `${r.criteria[0].criterion}${r.criteria.length > 1 ? ` (+${r.criteria.length - 1})` : ""} → ${r.criteria.map((c) => c.verdict).join(",")}` : "—";
246
+ const t0h = r.t0.length ? r.t0.map((h) => String(h).slice(0, 12)).join(", ") : "—";
247
+ L.push(`| ${r.id} | ${cell(r.source || "—")} | ${r.evidence} | ${cell(ac)} | ${cell(crit)} | ${t0h} |`);
248
+ }
249
+ L.push("");
250
+ if (requirements.inconsistencies.length) {
251
+ L.push("Anchored to a requirement no acceptance criterion covers — reconcile, do not count as evidence:", "");
252
+ for (const i of requirements.inconsistencies) L.push(`- ${i.requirement} ← "${i.criterion}" (${i.verdict})`);
253
+ L.push("");
254
+ }
255
+ }
256
+
217
257
  // The ratchet aggregate is derived, ~10 scalars that do not grow with the run, which is why it
218
258
  // can live in the committed tier while `metrics/` correctly stays gitignored (ADR-0001: a
219
259
  // committed shard keyed on $HOSTNAME only grows). Without this the instrument existed and was
@@ -256,32 +296,6 @@ export function buildReport(facts) {
256
296
  return L.join("\n");
257
297
  }
258
298
 
259
- /**
260
- * How many BUILD/EVAL rounds this run actually completed.
261
- *
262
- * `harness-run.md`'s `rounds_used` frontmatter field is written ONCE, at GATE L0.1 (`init run`),
263
- * as `0` — nothing in the round loop ever rewrites it as rounds complete. The orchestrator's own
264
- * `RunReturn` carries the real count (`shapeup-run.js`'s `rounds_used: round`), but that value
265
- * never reaches `reduce ship`, so every real run's report printed "Rounds used | 0" beside its own
266
- * `results/evaluate-r1.json` — a claim the frontmatter makes about the run, contradicted by the
267
- * artifact sitting next to it. Derived instead, the same way `probe resume`'s `eval_rounds_done`
268
- * already does: the highest `evaluate-r<N>.json` result on disk. Falls back to the frontmatter
269
- * value only when no EVAL round ever ran (the `--tiny` lane has no round concept at all), so a
270
- * bare or tiny run's reporting is unchanged.
271
- * @param {string} cwd - Project root.
272
- * @param {string} slug - Feature slug.
273
- * @param {(string|undefined)} fallback - `run.rounds_used` from the frontmatter.
274
- * @returns {(number|string|undefined)} The derived round count, or the fallback.
275
- */
276
- function roundsUsed(cwd, slug, fallback) {
277
- const dir = resultsDir(cwd, slug);
278
- const done = (existsSync(dir) ? readdirSync(dir) : [])
279
- .map((f) => f.match(/^evaluate-r(\d+)\.json$/))
280
- .filter(Boolean)
281
- .map((m) => Number(m[1]));
282
- return done.length ? Math.max(...done) : fallback;
283
- }
284
-
285
299
  /**
286
300
  * Gather every fact from disk and render the report.
287
301
  * @param {{cwd:string, slug:string, verdict?:string, qa?:string}} opts - Inputs.
@@ -296,16 +310,23 @@ export function generate({ cwd, slug, verdict, qa }) {
296
310
  const ledger = readIf(roundLedger(cwd, slug));
297
311
  const discovery = readIf(discoveryLedger(cwd, slug));
298
312
 
313
+ // Two numbers, not one: `rounds` (built — an order, a T0 verdict, a build-gate artifact or an
314
+ // EVAL result) and `roundsJudged` (EVAL actually returned a verdict for), kept separate so a run
315
+ // whose later rounds never reached EVAL still reports the rounds it built.
316
+ const derivedRounds = deriveRounds(cwd, slug, run.rounds_used);
317
+
299
318
  const facts = {
300
319
  slug,
301
320
  at: today(),
302
321
  verdict: verdict || run.final_verdict || "not-evaluated",
303
322
  qa: qa || (huntReport ? "run" : "skipped"),
304
- rounds: roundsUsed(cwd, slug, run.rounds_used),
323
+ rounds: derivedRounds.rounds_used,
324
+ roundsJudged: derivedRounds.rounds_judged,
305
325
  intakeSha: receipt.intake_sha256,
306
326
  board: boardCensus(cwd, slug),
307
327
  t0: t0Summary(cwd, slug),
308
328
  ratchet: ratchetReport(readTrials(trials(cwd, slug))),
329
+ requirements: projectRequirements({ cwd, slug }),
309
330
  artifacts: verdictArtifactCount(cwd, slug),
310
331
  evalCriteria: section(evalReport, /^#+\s.*criteria/i) || section(evalReport, /^#+\s*spec-conformance/i),
311
332
  evalBugs: section(evalReport, /^#+\s*Bugs?\b/i),
@@ -25,6 +25,7 @@ import { resolve, join } from "node:path";
25
25
  import { validate } from "../verify/envelope.mjs";
26
26
  import { runArgs } from "../lib/argv.mjs";
27
27
  import { localDir, localRoot, relLocal, globLocal, runSnapshot as runSnapshotPath } from "../lib/paths.mjs";
28
+ import { deriveRounds } from "../probe/rounds.mjs";
28
29
 
29
30
  /**
30
31
  * Read a JSON file, tolerating absence/parse errors.
@@ -126,8 +127,28 @@ export function deriveSnapshot(cwd) {
126
127
  if (existsSync(runPath)) {
127
128
  try {
128
129
  const fm = frontmatter(readFileSync(runPath, "utf8"));
129
- if (MID_RUN.has(fm.status) || ["shipped", "escalated"].includes(fm.status)) snapshot.status = fm.status;
130
- if (/^\d+$/.test(fm.rounds_used || "")) snapshot.rounds_used = Number(fm.rounds_used);
130
+ // The terminal allowlist mirrors kernel/probe/resume.mjs's TERMINAL_STATUSES, not just the two
131
+ // members it used to carry — a schema/allowlist that disagrees with RUN_STATUSES is silent on
132
+ // both sides here (findRun only ever surfaces a MID_RUN run today, so this whole branch is
133
+ // unreachable in practice), but is exactly the divergence class this file's own imports exist
134
+ // to close elsewhere (deriveRounds, above). "aborted" is a real RUN_STATUSES member.
135
+ if (MID_RUN.has(fm.status) || ["shipped", "escalated", "aborted"].includes(fm.status)) snapshot.status = fm.status;
136
+ // MIGRATED to the same mechanical derivation `reduce ship` and `report export` already use
137
+ // (kernel/probe/rounds.mjs), rather than reading `harness-run.md`'s `rounds_used` literally.
138
+ // That frontmatter line is written ONCE, as 0, by `init run`, and nothing in the round loop
139
+ // ever rewrites it — so the third mechanical reader of "how many rounds did this run build"
140
+ // was the one still reporting the pre-fix number. Measured on a two-round, no-EVAL fixture:
141
+ // this line alone reported `rounds_used: 0` beside its own `round: 2` a few lines below,
142
+ // the exact disagreement `deriveRounds` exists to close. `fm.rounds_used` still travels in as
143
+ // the fallback for a run neither this fix nor deriveRounds can see evidence for (a `--tiny`
144
+ // lane, or a run from before any of these artifacts existed).
145
+ const derivedRounds = deriveRounds(cwd, run.slug, fm.rounds_used);
146
+ if (Number.isFinite(Number(derivedRounds.rounds_used))) snapshot.rounds_used = Number(derivedRounds.rounds_used);
147
+ // Kept as its OWN field, never folded into rounds_used — a round built is not a round judged,
148
+ // and collapsing the two into one number is exactly the ambiguity this migration removes.
149
+ // null (no EVAL result on disk yet) is a real answer and is not written at all, the same
150
+ // optional-field discipline every other snapshot field here follows.
151
+ if (Number.isFinite(derivedRounds.rounds_judged)) snapshot.rounds_judged = derivedRounds.rounds_judged;
131
152
  if (/^\d+$/.test(fm.max_rounds || "")) snapshot.max_rounds = Number(fm.max_rounds);
132
153
  if (fm.auto_level) snapshot.auto_level = fm.auto_level;
133
154
  if (fm.spec_folder) snapshot.spec_folder = fm.spec_folder;
@@ -47,10 +47,11 @@ import { runArgs } from "../lib/argv.mjs";
47
47
  import { splitFrontmatter } from "../lib/contract.mjs";
48
48
  import { runIdFromReceipt, readReceipt } from "../lib/paths.mjs";
49
49
  import { TABLES, runRow, dispatchFacts } from "./facts.mjs";
50
+ import { deriveRounds } from "../probe/rounds.mjs";
50
51
  import {
51
52
  localDir, activeScope, receipt as receiptPath, harnessRun, ordersDir, resultsDir,
52
53
  trials as trialsPath, verdictsDir, evaluationDir, decisions as decisionsPath,
53
- exportsDir, exportRunDir,
54
+ gates as gatesPath, roundBuildDir, exportsDir, exportRunDir,
54
55
  } from "../lib/paths.mjs";
55
56
 
56
57
  export const EXPORT_SCHEMA_VERSION = 1;
@@ -164,6 +165,51 @@ function criterionRows(dir, runId, t) {
164
165
  return out;
165
166
  }
166
167
 
168
+ /**
169
+ * Flatten one gate-crossing ledger row (`gates.jsonl`, `kernel/gate.mjs`'s sole writer) into a
170
+ * flat `gate_decision` fact row. The row on disk already carries exactly these fields
171
+ * (see `appendGateLedger`), so this is a pass-through with a stamped `run_id` fallback rather than
172
+ * a re-derivation: two readers of "what did this gate decide" must not compute the answer twice.
173
+ * @param {object} g - One parsed line of `gates.jsonl`.
174
+ * @param {(string|null)} runId - Run key for a row written before it carried its own.
175
+ * @returns {object} A flat `gate_decision` row.
176
+ */
177
+ function gateDecisionRow(g, runId) {
178
+ return {
179
+ run_id: g?.run_id ?? runId ?? null,
180
+ gate: g?.gate ?? null,
181
+ decision: g?.decision ?? null,
182
+ status: g?.status ?? null,
183
+ source: g?.source ?? null,
184
+ round: g?.round ?? null,
185
+ has_note: !!(g?.note && String(g.note).trim()),
186
+ };
187
+ }
188
+
189
+ /**
190
+ * Flatten one round build-gate artifact (`kernel/verify/build.mjs`'s `writeRoundBuild`) into a flat
191
+ * `build_gate` fact row. The gate ends a round exactly as EVAL does (AGENTS.md's round
192
+ * build gate ⚙), and until now had no fact table at all.
193
+ * @param {object} a - A parsed round-build artifact.
194
+ * @param {(string|null)} runId - Run key for an artifact written before it carried its own.
195
+ * @returns {object} A flat `build_gate` row.
196
+ */
197
+ function buildGateRow(a, runId) {
198
+ const steps = Array.isArray(a?.steps) ? a.steps : [];
199
+ return {
200
+ run_id: a?.run_id ?? runId ?? null,
201
+ round: a?.round ?? null,
202
+ trial: a?.trial ?? null,
203
+ at: a?.at ?? null,
204
+ overall: a?.overall ?? null,
205
+ archetype: a?.archetype ?? null,
206
+ steps_total: steps.length,
207
+ steps_failed: steps.filter((s) => !s?.skipped && s?.pass === false).length,
208
+ warnings: Array.isArray(a?.warnings) ? a.warnings.length : 0,
209
+ discovered_tasks: Array.isArray(a?.discovered_tasks) ? a.discovered_tasks.length : 0,
210
+ };
211
+ }
212
+
167
213
  // ---------------------------------------------------------------------------
168
214
  // The export itself
169
215
  // ---------------------------------------------------------------------------
@@ -190,7 +236,10 @@ export function collectRun(cwd, slug) {
190
236
  const results = readJsonDir(resultsDir(cwd, slug), t);
191
237
 
192
238
  const { dispatch, ac_result, discovery, file_touched } = dispatchFacts({ orders, results, runId });
193
- const run = runRow({ receipt: rec, ledger, runId });
239
+ // Computed here, once, from the same trace this whole function reads, and handed to
240
+ // runRow rather than re-derived by it: runRow stays pure (no I/O), this function already has cwd.
241
+ const rounds = deriveRounds(cwd, slug, ledger.rounds_used);
242
+ const run = runRow({ receipt: rec, ledger, runId, rounds });
194
243
 
195
244
  // Hook decisions are checkout-wide, so they are FILTERED to this run rather than read from a
196
245
  // per-run file. Rows with a null key belong to no run (a hook that fired outside one) and are
@@ -208,6 +257,9 @@ export function collectRun(cwd, slug) {
208
257
  t0_verdict: readJsonDir(verdictsDir(cwd, slug), t).map((a) => t0Row(a, runId)),
209
258
  criterion_verdict: criterionRows(evaluationDir(cwd, slug), runId, t),
210
259
  hook_decision,
260
+ // The decision that crossed each gate, and the round build gate's own artifact.
261
+ gate_decision: readJsonl(gatesPath(cwd, slug), t).map((g) => gateDecisionRow(g, runId)),
262
+ build_gate: readJsonDir(roundBuildDir(cwd, slug), t).map((a) => buildGateRow(a, runId)),
211
263
  },
212
264
  defects: { records_skipped: t.skipped },
213
265
  };
@@ -27,6 +27,11 @@
27
27
  export const TABLES = [
28
28
  "run", "dispatch", "ac_result", "discovery", "file_touched",
29
29
  "trial", "t0_verdict", "criterion_verdict", "hook_decision",
30
+ // The decision that shipped a run (or any other gate) had a ledger row (`gates.jsonl`)
31
+ // and no table — a reader had to open the LOCAL trace itself, which the export exists so nobody
32
+ // has to. `build_gate` is the round build gate's own artifact (kernel/verify/build.mjs), on the
33
+ // same terms: it ends a round exactly as EVAL does, and had no table either.
34
+ "gate_decision", "build_gate",
30
35
  ];
31
36
 
32
37
  /** Coerce anything to a finite number, or null. Keeps `0` and rejects `NaN`/`""`/undefined. */
@@ -76,9 +81,14 @@ export function parseOrderStem(orderId) {
76
81
  * @param {(object|null)} o.receipt - Parsed `receipt.json`.
77
82
  * @param {(object|null)} [o.ledger] - Parsed `harness-run.md` frontmatter (a flat scalar map).
78
83
  * @param {(string|null)} [o.runId] - The run key, when already resolved.
84
+ * @param {({rounds_used:*, rounds_judged:(number|null)}|null)} [o.rounds] - The two-number
85
+ * derivation (`probe/rounds.mjs`'s `deriveRounds`), computed by the caller because it needs the
86
+ * filesystem and this function stays pure. Falls back to the ledger's own (unreliable — see
87
+ * `deriveRounds`) `rounds_used` line when the caller has not derived one, so an existing caller
88
+ * is unaffected rather than broken.
79
89
  * @returns {(object|null)} The run row, or null when there is no receipt to describe.
80
90
  */
81
- export function runRow({ receipt, ledger = null, runId = null }) {
91
+ export function runRow({ receipt, ledger = null, runId = null, rounds = null }) {
82
92
  if (!receipt) return null;
83
93
  const c = receipt.config || {};
84
94
  const fm = ledger || {};
@@ -87,6 +97,9 @@ export function runRow({ receipt, ledger = null, runId = null }) {
87
97
  slug: receipt.slug ?? null,
88
98
  started_at: receipt.started_at ?? null,
89
99
  closed_at: fm.closed_at && fm.closed_at !== "~" ? fm.closed_at : null,
100
+ // Why the run ended at a terminal status, written by `probe resume --close` alongside
101
+ // `closed_at` — the two facts a trace needs to tell a live run from a dead one apart.
102
+ close_cause: fm.close_cause && fm.close_cause !== "~" ? fm.close_cause : null,
90
103
  intake_sha256: receipt.intake_sha256 ?? null,
91
104
  intake_chars: num(receipt.intake_chars),
92
105
  intake_lines: num(receipt.intake_lines),
@@ -101,8 +114,17 @@ export function runRow({ receipt, ledger = null, runId = null }) {
101
114
  // Copied from the ledger, never re-derived: the run's own status line is the harness's answer,
102
115
  // and a read plane that recomputed it would be asserting a second one.
103
116
  status: fm.status ?? null,
117
+ // The terminal status `closeRun` alone writes, carried BESIDE `status` rather than instead of
118
+ // it. `status:` is ordinary phase traffic and a later phase may move it, so an export that
119
+ // carried only that line could show a run wearing another close's cause. These two disagreeing
120
+ // is itself the fact worth exporting: it says the ledger was written after the close.
121
+ closed_status: fm.closed_status && fm.closed_status !== "~" ? fm.closed_status : null,
104
122
  final_verdict: fm.final_verdict && fm.final_verdict !== "~" ? fm.final_verdict : null,
105
- rounds_used: num(Number(fm.rounds_used)),
123
+ // Two fields, not one: `rounds_used` is the highest round carrying ANY build evidence,
124
+ // `rounds_judged` the highest round EVAL actually returned a verdict for. A caller that has not
125
+ // derived `rounds` falls back to the ledger's own (pre-fix, unreliable) line, non-regression.
126
+ rounds_used: rounds ? num(Number(rounds.rounds_used)) : num(Number(fm.rounds_used)),
127
+ rounds_judged: rounds ? num(rounds.rounds_judged) : null,
106
128
  };
107
129
  }
108
130
 
@@ -341,6 +341,7 @@
341
341
  "spec_folder",
342
342
  "feature",
343
343
  "discovered_ledger",
344
+ "requirements",
344
345
  "kb_rules_path"
345
346
  ],
346
347
  "scope-architect": [
@@ -667,14 +668,14 @@
667
668
  "string",
668
669
  "null"
669
670
  ],
670
- "description": "Source file of the failure; null/absent when the log line carried no location (raw triple — never invented)."
671
+ "description": "Source file of the failure; null/absent when the log line named no file at all (raw triple — never invented)."
671
672
  },
672
673
  "line": {
673
674
  "type": [
674
675
  "integer",
675
676
  "null"
676
677
  ],
677
- "description": "Line of the failure; null/absent when the log line carried no location. Paired with `file` — a triple has both or neither, and neither is ever invented to satisfy a shape."
678
+ "description": "Line of the failure; null/absent when the diagnostic carried no line number — independently of `file`, since a diagnostic can name a file with no line (a resource-compiler error, for example). Never invented to satisfy a shape."
678
679
  },
679
680
  "core_message": {
680
681
  "type": "string",
@@ -1890,9 +1891,10 @@
1890
1891
  "building",
1891
1892
  "evaluating",
1892
1893
  "shipped",
1893
- "escalated"
1894
+ "escalated",
1895
+ "aborted"
1894
1896
  ],
1895
- "description": "Mirrors harness-run.md frontmatter status."
1897
+ "description": "Mirrors harness-run.md frontmatter status — kernel/probe/resume.mjs's RUN_STATUSES is the source enum this one must not drift from."
1896
1898
  },
1897
1899
  "round": {
1898
1900
  "type": "integer",
@@ -1903,7 +1905,12 @@
1903
1905
  "description": "From the latest t0/verdicts/r<N>-a<M>.json filename."
1904
1906
  },
1905
1907
  "rounds_used": {
1906
- "type": "integer"
1908
+ "type": "integer",
1909
+ "description": "Highest round carrying any build evidence (an order, a T0 verdict, a round build-gate artifact, or an EVAL result) — kernel/probe/rounds.mjs's deriveRounds(), the same derivation the ship report and the export use. Falls back to harness-run.md's literal frontmatter value only when no such evidence exists on disk."
1910
+ },
1911
+ "rounds_judged": {
1912
+ "type": "integer",
1913
+ "description": "Highest round EVAL actually returned a verdict for (an evaluate-r<N>.json result on disk) — its own field, never folded into rounds_used: a round built is not a round judged. Omitted when no round has been judged yet."
1907
1914
  },
1908
1915
  "max_rounds": {
1909
1916
  "type": "integer"
@@ -2611,6 +2618,10 @@
2611
2618
  "type": "boolean",
2612
2619
  "description": "ANALYZE finished: the spec folder's usecases/ carries at least one use case that is not _index.md. WIRE reads these — one wiring-map entry per use case — which is why ANALYZE precedes WIRE in the phase chain: dispatched against an empty spec folder, WIRE escalates on every launch."
2613
2620
  },
2621
+ "has_requirements": {
2622
+ "type": "boolean",
2623
+ "description": "The requirements registry is on disk: shapeup/<slug>/requirements.md exists. A PLAIN FACT, not a phase — the orchestrator guards its single `coverage` dispatch on this boolean, and it is deliberately absent from kernel/probe/resume.mjs's PHASE_ARTIFACT map, which doubles as nextPhase()'s ordered list: an entry there would fast-forward every run recorded before the registry existed to the registry instead of to build."
2624
+ },
2614
2625
  "has_wiring_map": {
2615
2626
  "type": "boolean",
2616
2627
  "description": "WIRE finished: shapeup/<slug>/wiring-map.md exists."
@@ -2699,10 +2710,10 @@
2699
2710
  }
2700
2711
  },
2701
2712
  "RunArgs": {
2702
- "description": "C1 — the launch half of the workflow's only conversation. Compiled ONCE by tech-lead at GATE L0 from harness init run output + the L0.8 model matrix + budgets, written to .shapeup/<slug>/run-args.json and handed to the harness run launch as one JSON literal — the workflow cannot ask follow-ups and cannot read config files itself, so everything a run will ever need travels in this one record. A workflow script validates its own subset of this shape in code (no runtime schema check at the C1 boundary itself); this entry is the central-registry definition the workflow script and the tech-lead skill both read as the one true shape.",
2713
+ "description": "C1 — the launch half of the workflow's only conversation. Resolved ONCE at GATE L0 from harness init run output + the L0.8 model matrix + budgets, then built by `harness init run-args`, which writes it to .shapeup/<slug>/run-args.json AND prints it — tech-lead passes that printed value to the harness run launch as one JSON literal, never a second, hand-typed copy of it. The workflow cannot ask follow-ups and cannot read config files itself, so everything a run will ever need travels in this one record. A workflow script validates its own subset of this shape in code (no runtime schema check at the C1 boundary itself); this entry is the central-registry definition the workflow script, `harness init run-args` and the tech-lead skill all read as the one true shape.",
2703
2714
  "x-tier": "EMBEDDED",
2704
- "x-location": ".shapeup/<slug>/run-args.json — written fresh by tech-lead on every launch and relaunch; the workflow receives it as its args and never reads other config",
2705
- "x-writer": "tech-lead (GATE L0, on every launch AND every relaunch after a paused gate)",
2715
+ "x-location": ".shapeup/<slug>/run-args.json — written fresh on every launch and relaunch by `harness init run-args`; the workflow receives it as its args and never reads other config",
2716
+ "x-writer": "harness init run-args (kernel), invoked by tech-lead at GATE L0 on every launch AND every relaunch after a paused gate",
2706
2717
  "x-readers": "the Workflow runtime (shapeup-run, and shapeup-run's own inner round dispatch)",
2707
2718
  "x-not-here": "Run config the LEDGER already carries does NOT get a second home in RunArgs — eval_dimensions, lens, spec_folder, stack, run_cmd, app_url are read off harness-run.md frontmatter by resume-state on every launch AND every relaunch, so a copy here would be a second source that can disagree with the first. RunArgs carries what a workflow cannot derive from disk (identity, budgets, the model matrix, pluginRoot, startedAt) plus noEval, which no frontmatter line holds.",
2708
2719
  "type": "object",
@@ -2754,16 +2765,13 @@
2754
2765
  },
2755
2766
  "budgets": {
2756
2767
  "type": "object",
2757
- "description": "The three-level circuit breaker's own limits (AGENTS.md) — outer round_budget, inner attempt_budget, and the opt-in wall-clock breaker.",
2768
+ "description": "The two RunArgs-level circuit breakers (AGENTS.md) — outer round_budget, inner attempt_budget. The third, opt-in DEADLINE breaker (the wall-clock budget) is NOT a RunArgs field: it is typed once, at `harness init run --wall-clock-budget`, and lands in the run receipt's `wall_clock_budget_s` — `harness verify budget` reads that receipt field directly and never sees this launch's RunArgs at all, so it has no member here to declare.",
2758
2769
  "properties": {
2759
2770
  "maxRounds": {
2760
2771
  "type": "integer"
2761
2772
  },
2762
2773
  "attemptBudget": {
2763
2774
  "type": "integer"
2764
- },
2765
- "wallClockS": {
2766
- "type": "integer"
2767
2775
  }
2768
2776
  }
2769
2777
  },
@@ -12,7 +12,7 @@
12
12
  // #/$defs/Name — a definition in the SAME schema document
13
13
  // domain.schema.json#/$defs/Name — a definition in a SIBLING file (the central domain
14
14
  // registry; resolved against the schema's own dir,
15
- // falling back to skills/tech-lead/schemas/)
15
+ // falling back to kernel/schemas/)
16
16
  //
17
17
  // Usage (CLI): node kernel/harness.mjs verify envelope <envelope.json> <schema.json>
18
18
  // exit 0 = valid, 1 = invalid (errors printed one per line)
@@ -28,7 +28,7 @@ import { runArgs } from "../lib/argv.mjs";
28
28
  import { runHook, readStdin, settle } from "../../hooks/lib/decision.mjs";
29
29
 
30
30
  const HERE = dirname(fileURLToPath(import.meta.url));
31
- export const SCHEMAS_DIR = resolve(HERE, "../../skills/tech-lead/schemas");
31
+ export const SCHEMAS_DIR = resolve(HERE, "../schemas");
32
32
 
33
33
  /**
34
34
  * Validate a value against the JSON-Schema subset the envelope schemas use (type, required,
@@ -45,7 +45,7 @@ export const PLUGIN_ROOT = resolve(HERE, "../..");
45
45
  * its own domain registry has a broken installation, which is the very thing being checked.
46
46
  */
47
47
  export function roster(root = PLUGIN_ROOT) {
48
- const schemaPath = join(root, "skills/tech-lead/schemas/domain.schema.json");
48
+ const schemaPath = join(root, "kernel/schemas/domain.schema.json");
49
49
  const schema = JSON.parse(readFileSync(schemaPath, "utf8"));
50
50
  const names = schema?.$defs?.WorkerName?.enum;
51
51
  if (!Array.isArray(names) || !names.length) {
@@ -34,6 +34,11 @@
34
34
  // SCOPE-COVERS a contract's covers entry that is not a REQ-id (warn), or names a REQ that
35
35
  // is not in requirements.md (red, when a registry exists) — shape alone let a scope
36
36
  // claim coverage of a requirement that does not exist
37
+ // REQ-UNCOVERED the other direction of the same edge: a registered requirement still marked
38
+ // covered that NO acceptance criterion grades and NO scope claims. SCOPE-COVERS asks
39
+ // whether a link resolves; this asks whether a requirement has one at all. Red here
40
+ // and only advisory in trace-lint, because a requirement nothing reaches is a plan
41
+ // defect the PO can still answer at L1b — cover it, or cut it on the record
37
42
  // SCOPE-PARTITION a task claimed by more than one scope. The UC anchor is a SPEC link, not an
38
43
  // assignment: one use case is routinely implemented by several scopes, so on a
39
44
  // four-scope/one-UC cut every scope claimed every task and would build all of them.
@@ -65,9 +70,18 @@ import { parseBoard, deriveUnlocks } from "../reduce/board.mjs";
65
70
  import { runArgs } from "../lib/argv.mjs";
66
71
  import { LOCAL } from "../lib/paths.mjs";
67
72
  import { specDir, scopesDir, tasksDir, intake, sharedRoot, requirements } from "../lib/paths.mjs";
68
- import { readAllContracts, unreadableReason, ucId, scopePartitionConflicts, SCOPE_CONTRACT } from "../lib/contract.mjs";
73
+ import { readAllContracts, unreadableReason, ucId, reqId, scopePartitionConflicts, SCOPE_CONTRACT } from "../lib/contract.mjs";
74
+ import { UNREADABLE, LEGACY_LAYOUT } from "../lib/contract.mjs";
75
+ import { validate as validateAgainstSchema, SCHEMAS_DIR } from "./envelope.mjs";
69
76
  import { breadboard as stagedBreadboard } from "../lib/paths.mjs";
70
77
  import { parseBreadboard, hasBreadboardTables, idCounts } from "../lib/breadboard.mjs";
78
+ // ONE implementation of covers-closure, two reporters: trace-lint narrates it, spec-lint gates it.
79
+ // Re-deriving either here is how the advisory report and the gate start disagreeing about which
80
+ // requirement is covered. This closes the import ring spec → trace → compile → probe/resume → spec,
81
+ // which holds only while no module in it dereferences an imported binding at module-evaluation
82
+ // time — do NOT add a top-level `const x = someImportedFn()` to any of the four.
83
+ import { parseRequirements, coveredReqIds } from "./trace.mjs";
84
+ import { readBoard } from "../compile.mjs";
71
85
 
72
86
  // Inlined from hooks/sandbox-guard.mjs so this skill ships self-contained (a skill's scripts
73
87
  // must not reach outside its own folder — channels that copy only skills/ would dangle).
@@ -347,7 +361,11 @@ export function lintScopeAnchors({ scopes, specDir: specRoot, reqIds = null, tas
347
361
  else if (id && !ids.has(id)) findings.push({ rule: "SCOPE-DEPS", level: "red", scope: where, detail: `depends_on "${id}" is not a scope in this run — the scheduler drops the edge, so this scope may build before its dependency` });
348
362
  }
349
363
  for (const r of s.covers || []) {
350
- const req = String(r).trim();
364
+ // ONE KEY SPACE. A pitch numbers its requirements `R<n>` and the registry keys off
365
+ // `REQ-<n>`; `reqId` maps the first onto the second BEFORE the pattern below, so a link the
366
+ // planner actually wrote resolves instead of reading as a shape warning nobody can act on.
367
+ // A reference neither space recognises comes back verbatim and still fails the pattern.
368
+ const req = reqId(r);
351
369
  if (!/^REQ-[A-Z0-9-]+$/i.test(req)) {
352
370
  findings.push({ rule: "SCOPE-COVERS", level: "warn", scope: where, detail: `covers "${r}" is not a REQ-id — the requirement edge will not resolve` });
353
371
  continue;
@@ -372,6 +390,55 @@ export function lintScopeAnchors({ scopes, specDir: specRoot, reqIds = null, tas
372
390
  return findings;
373
391
  }
374
392
 
393
+ /**
394
+ * REQ-UNCOVERED — a live requirement that nothing in the plan reaches.
395
+ *
396
+ * THE OTHER DIRECTION OF THE COVERS EDGE. `SCOPE-COVERS` walks the links that exist and asks
397
+ * whether each one resolves; a requirement with no link at all satisfies it perfectly. Measured on
398
+ * a full run of one pitch: twenty-one requirements, every one of them with an acceptance criterion
399
+ * somewhere, and only eleven reaching a criterion the judge grades — the board is the last place a
400
+ * requirement can be dropped without anything going red, because after L1b nobody re-reads the
401
+ * pitch.
402
+ *
403
+ * WHY THE BOARD HERE IS `readBoard`, NOT `lint()`'s `tasks`. `parseBoard` (`kernel/reduce/board.mjs`)
404
+ * builds the scheduling view and its records carry no `acceptance_criteria` field at all, while
405
+ * `coveredReqIds` reads exactly that field — feed it the wrong board and the covered set is empty
406
+ * and EVERY requirement reds on EVERY run. `readBoard` (`kernel/compile.mjs`) is the parser that
407
+ * carries the criteria, and it is the only other one there may be: a second parser of the task file
408
+ * is explicitly ruled out where the first one lives.
409
+ *
410
+ * A SCOPE'S CLAIM COUNTS. The arm is about requirements nothing reaches, not about which layer
411
+ * reaches them: a clause claimed by a contract's `covers:` has an owner who answers for it at L1b,
412
+ * even before the criterion that grades it is written. `CUT (PO-approved)` is likewise an answer
413
+ * already given, not a defect — which is why `status` is read rather than assumed.
414
+ *
415
+ * @param {{clauses:Array<{id:string, clause:string, source:string, status:string}>,
416
+ * board:Array<object>, scopes:Array<{covers?:string[]}>}} input - The registry clauses
417
+ * (`parseRequirements`), the board `readBoard` parsed, and the scope contracts. An empty
418
+ * `clauses` (no registry on disk) yields no findings — absent artifact ⇒ arm skipped.
419
+ * @returns {Array<{rule:string, level:("red"|"warn"), scope:string, detail:string}>} One red per
420
+ * uncovered live requirement; [] when every one is graded, claimed or cut.
421
+ */
422
+ export function lintRequirementCoverage({ clauses = [], board = [], scopes = [] }) {
423
+ const findings = [];
424
+ const graded = coveredReqIds(board);
425
+ // The contracts speak the pitch's numbering as readily as the registry's; `reqId` lands both in
426
+ // the one key space before the comparison, exactly as SCOPE-COVERS does above.
427
+ const claimed = new Set();
428
+ for (const s of scopes) for (const r of s.covers || []) claimed.add(reqId(r).toUpperCase());
429
+ for (const c of clauses) {
430
+ if (c.status !== "covered") continue; // CUT (PO-approved) — an answer on the record, not a gap
431
+ const id = c.id.toUpperCase();
432
+ if (graded.has(c.id) || claimed.has(id)) continue;
433
+ const from = c.source ? ` ← ${c.source}` : "";
434
+ findings.push({ rule: "REQ-UNCOVERED", level: "red", scope: c.id, detail:
435
+ `${c.id}${from} is graded by no acceptance criterion and claimed by no scope — "${(c.clause || "").slice(0, 60)}" ` +
436
+ "would ship unverified and nothing downstream would say so. Cover it with an AC carrying " +
437
+ `(covers: ${c.id}), or mark it CUT (PO-approved) in requirements.md.` });
438
+ }
439
+ return findings;
440
+ }
441
+
375
442
  /**
376
443
  * Every dependency cycle among the scopes, each reported once from its lowest-sorting member.
377
444
  * @param {Array<{scope_id:string, depends_on?:string[]}>} scopes - The contracts.
@@ -663,6 +730,51 @@ export function runBreadboard(cwd, slug, intakeContent) {
663
730
  return hasBreadboardTables(intakeContent) ? intakeContent : null;
664
731
  }
665
732
 
733
+ /**
734
+ * Every scope contract whose PARSED shape fails `$defs/ScopeContract`.
735
+ *
736
+ * `kernel/lib/contract.mjs`'s own banner promised this check — "spec-lint re-validates every parsed
737
+ * contract against domain.schema.json, so a hand-edit that breaks the shape fails loudly instead of
738
+ * silently widening a sandbox" — and it did not exist. `compile` validated, spec-lint did not, so a
739
+ * contract could pass GATE L1b green and then be refused at dispatch by the one reader that checked.
740
+ *
741
+ * Measured 2026-09-19 on a real run: a planner wrote every `required_states` table cell bare
742
+ * (`loading, error, ready`) where the dialect wants `[loading, error, ready]`, so all 32 manifest
743
+ * rows across the six UI scopes parsed as strings. `verify spec` reported `red=0`; `compile` then
744
+ * refused all six with `expected array, got string`, and those scopes were never dispatched — no
745
+ * order, no leg, no T0 trial. The round reached EVAL with six of eighteen scopes missing and the
746
+ * evaluator escalated rather than grading. This arm turns that into a red at the gate, naming the
747
+ * scope and the field, with the message the compiler would otherwise produce an hour later.
748
+ *
749
+ * The validator is the one `compile` already uses; there is no second implementation here.
750
+ *
751
+ * @param {Array<{contract:object, path:string}>} contracts - Parsed contracts with their paths.
752
+ * @param {object} domainSchema - The parsed `domain.schema.json`.
753
+ * @returns {Array<{rule:string, level:string, scope:string, detail:string}>} One red per invalid
754
+ * contract; [] when the schema cannot be read (absent artifact ⇒ arm skipped).
755
+ */
756
+ export function lintContractSchema(contracts, domainSchema) {
757
+ const def = domainSchema?.$defs?.ScopeContract;
758
+ if (!def) return [];
759
+ const schema = { ...def, $defs: domainSchema.$defs };
760
+ const out = [];
761
+ for (const { contract, path } of contracts) {
762
+ const c = { ...contract };
763
+ delete c[UNREADABLE];
764
+ delete c[LEGACY_LAYOUT];
765
+ let res;
766
+ try { res = validateAgainstSchema(c, schema); } catch { continue; } // fail open, never closed
767
+ if (res?.valid) continue;
768
+ out.push({
769
+ rule: "CONTRACT-SCHEMA", level: "red", scope: contract.scope_id || path,
770
+ detail: `the contract parses, but not into the shape a WorkOrder carries — ${(res.errors || [])[0] || "schema validation failed"}. ` +
771
+ `compile refuses an order that fails its own schema, so as written this scope would be silently undispatched. ` +
772
+ `A list in a table cell is written [a, b], brackets and all.`,
773
+ });
774
+ }
775
+ return out;
776
+ }
777
+
666
778
  /**
667
779
  * Run the full spec lint (scopes + structure) for a slug.
668
780
  * @param {{cwd:string, slug:string}} opts - Working root and feature slug.
@@ -679,10 +791,21 @@ export function lint({ cwd, slug }) {
679
791
  const intakeContent = existsSync(intakePath) ? readFileSync(intakePath, "utf8") : "";
680
792
  // The REQ registry, when the tree has one — absent means covers-closure simply cannot apply.
681
793
  const reqFile = requirements(cwd, slug);
682
- const reqIds = existsSync(reqFile)
683
- ? new Set([...readFileSync(reqFile, "utf8").matchAll(/\bREQ-[A-Z0-9-]+/gi)].map((m) => m[0].toUpperCase()))
794
+ const reqText = existsSync(reqFile) ? readFileSync(reqFile, "utf8") : null;
795
+ const reqIds = reqText !== null
796
+ ? new Set([...reqText.matchAll(/\bREQ-[A-Z0-9-]+/gi)].map((m) => m[0].toUpperCase()))
684
797
  : null;
798
+ // Table rows only, and with the status/source cells REQ-UNCOVERED reports from — the id set
799
+ // above is deliberately looser (it also sees ids named in the registry's prose) and stays that
800
+ // way, because the two arms ask different questions of the same file.
801
+ const reqClauses = reqText !== null ? parseRequirements(reqText) : [];
685
802
  const repoFiles = walkFiles(cwd);
803
+ // Loaded HERE, not at module scope. `spec → trace → compile → probe/resume → spec` is a live
804
+ // import ring, and a top-level dereference of an imported binding is what would break it.
805
+ // Unreadable schema ⇒ the arm skips itself, like every other absent-artifact arm.
806
+ let domainSchema = null;
807
+ try { domainSchema = JSON.parse(readFileSync(join(SCHEMAS_DIR, "domain.schema.json"), "utf8")); } catch { /* arm skipped */ }
808
+
686
809
  const findings = [
687
810
  // A contract whose table this parser cannot see reads as a contract that declared no
688
811
  // table, and every rule below then passes for the part it could not read. Loud, not empty.
@@ -690,8 +813,11 @@ export function lint({ cwd, slug }) {
690
813
  .map(({ contract, path }) => ({ reason: unreadableReason(contract), scope: contract.scope_id || path }))
691
814
  .filter((x) => x.reason)
692
815
  .map((x) => ({ rule: "CONTRACT-UNREADABLE", level: "red", scope: x.scope, detail: `${x.reason} — the rules below could not check what they could not read` })),
816
+ ...lintContractSchema(contracts, domainSchema),
693
817
  ...lintScopes(scopes, repoFiles),
694
818
  ...lintScopeAnchors({ scopes, specDir: specRoot, reqIds, tasks }),
819
+ // `readBoard`, not the `tasks` above: only the compile-order parser carries acceptance_criteria.
820
+ ...lintRequirementCoverage({ clauses: reqClauses, board: readBoard(cwd, slug), scopes }),
695
821
  ...lintCommittedTier({ cwd, slug }),
696
822
  ...lintStructure({ specDir: specRoot, tasks, intakeContent }),
697
823
  ...(() => {