shapeup-sdlc 3.4.0 → 3.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (43) hide show
  1. package/.claude-plugin/plugin.json +1 -1
  2. package/AGENTS.md +16 -4
  3. package/README.md +7 -3
  4. package/SECURITY.md +1 -1
  5. package/hooks/sandbox-guard.mjs +69 -6
  6. package/kernel/compile.mjs +33 -12
  7. package/kernel/harness.mjs +10 -4
  8. package/kernel/init/run-args.mjs +206 -0
  9. package/kernel/init/run.mjs +10 -0
  10. package/kernel/lib/contract.mjs +68 -1
  11. package/kernel/lib/paths.mjs +10 -0
  12. package/kernel/probe/concurrency.mjs +31 -6
  13. package/kernel/probe/digest.mjs +15 -1
  14. package/kernel/probe/owner.mjs +4 -1
  15. package/kernel/probe/requirements.mjs +296 -0
  16. package/kernel/probe/resume.mjs +195 -6
  17. package/kernel/probe/rounds.mjs +104 -0
  18. package/kernel/reduce/graph.mjs +5 -2
  19. package/kernel/reduce/ingest.mjs +69 -15
  20. package/kernel/reduce/ship.mjs +52 -31
  21. package/kernel/reduce/snapshot.mjs +23 -2
  22. package/kernel/report/export.mjs +54 -2
  23. package/kernel/report/facts.mjs +24 -2
  24. package/{skills/tech-lead → kernel}/schemas/domain.schema.json +20 -12
  25. package/kernel/verify/envelope.mjs +2 -2
  26. package/kernel/verify/skills.mjs +1 -1
  27. package/kernel/verify/spec.mjs +130 -4
  28. package/kernel/verify/trace.mjs +16 -7
  29. package/package.json +1 -1
  30. package/skills/ba-pitch-analyzer/SKILL.md +16 -1
  31. package/skills/coach/SKILL.md +8 -2
  32. package/skills/hill-chart/SKILL.md +3 -4
  33. package/skills/scope-architect/SKILL.md +16 -1
  34. package/skills/scope-hammer/SKILL.md +11 -2
  35. package/skills/spec-evaluator/SKILL.md +12 -1
  36. package/skills/tech-lead/SKILL.md +10 -10
  37. package/skills/tech-lead/references/gates.md +70 -12
  38. package/skills/tech-lead/references/protocol.md +4 -2
  39. package/skills/tech-lead/workflows/shapeup-run.js +176 -38
  40. package/skills/translator/SKILL.md +1 -1
  41. /package/{skills/tech-lead → kernel}/schemas/gate-answers.schema.json +0 -0
  42. /package/{skills/tech-lead → kernel}/schemas/work-order.schema.json +0 -0
  43. /package/{skills/tech-lead → kernel}/schemas/work-result.schema.json +0 -0
@@ -94,6 +94,9 @@ export const PROJECT_PROFILE = { tables: {} };
94
94
  */
95
95
  export const UNREADABLE = "$unreadable_tables";
96
96
 
97
+ /** `found_under` marker: a table field that was also declared in the frontmatter, where nothing reads it. */
98
+ export const FRONTMATTER_COPY = "the frontmatter, where it is silently discarded";
99
+
97
100
  /**
98
101
  * Marks a contract that parsed only via a MIGRATION reader — readable, but not in the canonical
99
102
  * shape. Without it the fallback is permanent by silence: the file works, nothing says it is the
@@ -440,7 +443,34 @@ export function parseContract(md, spec = SCOPE_CONTRACT) {
440
443
  const unreadable = [...(meta[UNREADABLE] || [])];
441
444
  delete out[UNREADABLE];
442
445
  for (const [field, heading] of Object.entries(spec.tables || {})) {
443
- if (tables[heading]) { out[field] = tables[heading]; continue; }
446
+ // A TABLE FIELD DECLARED IN THE FRONTMATTER TOO IS A DISCARDED DECLARATION, and it has to be
447
+ // loud. `renderContract` writes table fields ONLY as a table — it filters them out of the
448
+ // frontmatter it emits — so the table is the source by construction and the line below is right
449
+ // to prefer it. What the old code did not say is that the frontmatter copy was then dropped
450
+ // without a word, and `splitFrontmatter` cannot even represent this shape: a block sequence of
451
+ // MAPPINGS flattens to a list of strings, so the copy is usually garbage as well as ignored.
452
+ //
453
+ // Measured: a planner wrote `affordance_manifest` in both places, the frontmatter copy carrying
454
+ // the correct `required_states: [idle]` and the table cell a bare `idle`. The parser took the
455
+ // table, the value was a string where the schema wants an array, `compile` refused the order,
456
+ // and the scope was NEVER DISPATCHED — spec-lint passing the contract, the build leg reporting
457
+ // `state: "done", error: null`, four of nine scopes gone every round, until EVAL refused to
458
+ // grade a round whose scopes had never run. The contract even captioned its own table "rendered
459
+ // here for reviewers": the author's model was the exact inverse of the parser's. That
460
+ // disagreement is the defect; which side wins is not.
461
+ //
462
+ // ONLY WHEN A TABLE ACTUALLY WINS. With no rows under the heading the frontmatter value is what
463
+ // the field becomes, and `affordance_manifest: []` meaning "none" is legal and current — five of
464
+ // those nine contracts do exactly that and are correct. An earlier cut of this rule fired on all
465
+ // nine and turned a working run red, which is the opposite of the fix.
466
+ if (tables[heading]) {
467
+ if (field in meta) {
468
+ unreadable.push({ field, expected_heading: heading, found_under: FRONTMATTER_COPY,
469
+ rows: Array.isArray(meta[field]) ? meta[field].length : 1 });
470
+ }
471
+ out[field] = tables[heading];
472
+ continue;
473
+ }
444
474
  // The field is absent — but is it absent because nobody declared it, or because the
445
475
  // author declared it somewhere this parser does not look? Those are opposite facts and the
446
476
  // old code returned the same thing for both. A table carrying this field's signature columns,
@@ -534,6 +564,14 @@ export function unreadableReason(contract) {
534
564
  return `\`${x.field}\` was written as a \`## ${x.field}\` markdown section, which this dialect reads as prose — ` +
535
565
  `it must be a FRONTMATTER key (a \`- \` block list or an inline [a, b] list), so as written the field parsed as ABSENT`;
536
566
  }
567
+ // A table field ALSO declared in the frontmatter. Its own case because the fix is "delete the
568
+ // frontmatter copy", not "move the table" — and because the copy is not merely redundant: the
569
+ // parser takes the table, so whatever the copy said was discarded without a word.
570
+ if (x.found_under === FRONTMATTER_COPY) {
571
+ return `\`${x.field}\` is a TABLE field and was also declared in the frontmatter, ` +
572
+ `where it is read by nothing and silently discarded — the \`## ${x.expected_heading}\` table is the only source. ` +
573
+ `Delete the frontmatter copy, and check the table says what it said: a list cell is written \`[a, b]\``;
574
+ }
537
575
  return `\`${x.field}\` must be a table under a \`## ${x.expected_heading}\` heading; found ${x.rows} matching row(s) under "${x.found_under}" instead, so the field parsed as ABSENT`;
538
576
  })
539
577
  .join("; ");
@@ -652,6 +690,35 @@ export function ucId(ref) {
652
690
  return String(ref ?? "").trim().replace(/^\[\[|\]\]$/g, "").replace(/^usecases\//, "").replace(/\.md$/, "").trim();
653
691
  }
654
692
 
693
+ /**
694
+ * Normalise one requirement reference to the registry's key space (`REQ-<n>`).
695
+ *
696
+ * TWO KEY SPACES FOR ONE THING, and the measurement is what decided this. A pitch numbers its
697
+ * requirements `R1…R21`; the registry, every AC's `covers:` clause and every `traces_to[]` key off
698
+ * `REQ-<n>`. Measured across two runs of one pitch, the scope contracts cited `R<n>` — 20 of 21
699
+ * requirements had a scope claiming them, and every one of those links resolved to nothing,
700
+ * reported 23 times a run as a shape warning nobody could act on. So the edge WAS produced; it was
701
+ * severed by spelling alone.
702
+ *
703
+ * Normalising here rather than teaching the planner to emit `REQ-<n>` is deliberate: `covers[]` is
704
+ * an OPTIONAL contract field, so a fix that depends on a planner choosing to comply converts
705
+ * whatever links the next run happens to write, while this converts the links on contracts already
706
+ * committed, with no worker behaviour change. The craft still asks for `REQ-<n>` going forward.
707
+ *
708
+ * THIS IS A MAPPING PERFORMED BEFORE THE PATTERN, NOT A LOOSENING OF IT. `^REQ-[0-9]+$` is
709
+ * unchanged everywhere it appears; a reference this function does not recognise is returned
710
+ * verbatim, so it still fails that pattern and is still reported.
711
+ *
712
+ * @param {string} ref - A requirement reference (`REQ-12`, `R12`, `R-12`, `[[REQ-12]]`, any case).
713
+ * @returns {string} The canonical `REQ-<n>` id, or the trimmed input unchanged when it is not a
714
+ * numbered requirement reference at all.
715
+ */
716
+ export function reqId(ref) {
717
+ const s = String(ref ?? "").trim().replace(/^\[\[|\]\]$/g, "").trim();
718
+ const m = s.match(/^(?:REQ|R)-?([0-9]+)$/i);
719
+ return m ? `REQ-${m[1]}` : s;
720
+ }
721
+
655
722
  /**
656
723
  * The tasks on a LOCAL board that belong to a scope, joined through the COMMITTED spec.
657
724
  *
@@ -140,6 +140,16 @@ export const receipt = (cwd, slug) => join(localRoot(cwd, slug), RECEIPT_FILE);
140
140
  export const intake = (cwd, slug) => join(localRoot(cwd, slug), "intake.md");
141
141
  /** The breadboard the pitch was shaped with, verbatim, next to its digest in the receipt. */
142
142
  export const breadboard = (cwd, slug) => join(localRoot(cwd, slug), "breadboard.md");
143
+ /**
144
+ * The launch record's filename, as a constant rather than a literal at each call site — the
145
+ * writer (`harness init run-args`) and every reader (`probe concurrency`'s `dialFrom()`) resolve
146
+ * the same path through this constant, the same way {@link runIdFromRoot} resolves `RECEIPT_FILE`
147
+ * against a bare run root rather than through {@link receipt}'s `(cwd, slug)` form: a caller that
148
+ * already holds the run root (an archived trace, `--run-root <dir>`) has no slug to reconstruct.
149
+ */
150
+ export const RUN_ARGS_FILE = "run-args.json";
151
+ /** The launch record — the RunArgs object a run was launched with (GATE L0.9b), kernel-written. */
152
+ export const runArgsPath = (cwd, slug) => join(localRoot(cwd, slug), RUN_ARGS_FILE);
143
153
  /** The run ledger — rounds, decisions, status frontmatter. */
144
154
  export const harnessRun = (cwd, slug) => join(localRoot(cwd, slug), "harness-run.md");
145
155
  /** File-derived mid-run digest, frozen by `reduce snapshot --write` as an audit anchor. */
@@ -30,13 +30,16 @@
30
30
  //
31
31
  // Usage: node kernel/harness.mjs probe concurrency --slug <slug> [--cwd <dir>] [--run-root <dir>]
32
32
  // [--round N] [--gap-s N] [--format json|table]
33
+ // [--require-run-args]
33
34
  // Exit: 0 = a report was produced with at least one usable leg · 1 = ran, and no leg in scope had
34
35
  // a usable interval (the report still prints, and says why) · 2 = malformed argv.
36
+ // With --require-run-args the report is skipped: 0 = run-args.json exists at this run root ·
37
+ // 6 = it does not (mirrors `probe resume --require`'s "artifact absent" convention).
35
38
 
36
39
  import { existsSync, readFileSync } from "node:fs";
37
40
  import { join, resolve } from "node:path";
38
41
  import { runArgs } from "../lib/argv.mjs";
39
- import { localRoot, runIdFromRoot, RECEIPT_FILE } from "../lib/paths.mjs";
42
+ import { localRoot, runIdFromRoot, RECEIPT_FILE, RUN_ARGS_FILE } from "../lib/paths.mjs";
40
43
 
41
44
  /** The round-addressed order id forms `<scope>-r<N>-a<M>` and `<phase>-r<N>`. */
42
45
  const ROUND_SUFFIX = /^(.*?)-r(\d+)(?:-a(\d+))?$/;
@@ -301,16 +304,18 @@ export function summarise(index, legs) {
301
304
  /**
302
305
  * Which fan-out width the run was launched with — read, never assumed.
303
306
  *
304
- * The dial is written into `run-args.json` by the launcher. It is absent from every run recorded so
305
- * far, so the honest answer is the effective default WITH the fact that it is a default: reporting
306
- * `4` unqualified would assert an operator choice nobody made.
307
+ * The dial is written into `run-args.json` by `harness init run-args` (GATE L0.9b), the kernel
308
+ * writer tech-lead invokes right before the launch. A run opened before that writer existed, or one
309
+ * whose launcher never passed `--parallel-scopes`, still has no file or no key — so the honest
310
+ * answer stays the effective default WITH the fact that it is a default: reporting `4` unqualified
311
+ * would assert an operator choice nobody made.
307
312
  *
308
313
  * @param {string} runRoot - The run's LOCAL root.
309
314
  * @returns {{max_parallel_scopes:number, source:string}} The value and where it came from.
310
315
  */
311
316
  export function dialFrom(runRoot) {
312
317
  try {
313
- const a = JSON.parse(readFileSync(join(runRoot, "run-args.json"), "utf8"));
318
+ const a = JSON.parse(readFileSync(join(runRoot, RUN_ARGS_FILE), "utf8"));
314
319
  const n = Number(a?.maxParallelScopes);
315
320
  if (Number.isFinite(n) && n >= 1) return { max_parallel_scopes: n, source: "run-args" };
316
321
  return { max_parallel_scopes: DEFAULT_MAX_PARALLEL_SCOPES, source: "default (run-args.json declares none)" };
@@ -474,7 +479,7 @@ export function table(r) {
474
479
  /** The typed argv contract (see `./lib/argv.mjs`). */
475
480
  export const ARGV_SPEC = {
476
481
  usage: "harness.mjs probe concurrency (--slug <slug> | --run-root <dir>) [--cwd <dir>] [--round N] " +
477
- "[--gap-s N] [--format json|table]",
482
+ "[--gap-s N] [--format json|table] [--require-run-args]",
478
483
  _: { arity: 0, max: 0, name: "(no positional operands)" },
479
484
  slug: { type: "str" },
480
485
  // The archived-trace and post-export cases, and the same escape `verify t0 --out` has: a caller
@@ -484,6 +489,10 @@ export const ARGV_SPEC = {
484
489
  round: { type: "int", min: 1 },
485
490
  "gap-s": { type: "int", min: 1, default: DEFAULT_GAP_S },
486
491
  format: { type: "enum", values: ["json", "table"], default: "json" },
492
+ // The positive enforcer for the launch record (see the banner below). Skips the concurrency
493
+ // report entirely — this is a presence check, not a measurement, and the two must not be
494
+ // confused by sharing an exit code.
495
+ "require-run-args": { type: "flag" },
487
496
  };
488
497
 
489
498
  /**
@@ -492,6 +501,14 @@ export const ARGV_SPEC = {
492
501
  * @param {string[]} rawArgv - The subcommand's own arguments (harness.mjs strips the verb words).
493
502
  * @returns {void} Exits 0 when at least one leg had a usable interval, 1 when none did — and the
494
503
  * report prints either way, because "nothing was measurable" is the answer, not an error.
504
+ *
505
+ * `--require-run-args` is a different question with its own exit convention, mirroring
506
+ * `probe resume --require`'s 0 (satisfied) / 6 (artifact absent): it never reaches the report at
507
+ * all. The launch record had exactly one reader (`dialFrom()`, below) and zero enforcers — a
508
+ * run missing it proceeded green with a silently substituted default, "indistinguishable from an
509
+ * operator choice" (`gates.md` L0.9b). This flag is what `shapeup-run.js` calls at Preflight, before
510
+ * ORIENT, so that state stops being invisible: this module already owns run-root resolution and the
511
+ * file's path, so the check is a few lines here rather than a new kernel entry point.
495
512
  */
496
513
  export function cli(rawArgv) {
497
514
  const args = runArgs(ARGV_SPEC, rawArgv);
@@ -504,6 +521,14 @@ export function cli(rawArgv) {
504
521
  ? resolve(args.runRoot)
505
522
  : localRoot(resolve(args.cwd || process.cwd()), args.slug);
506
523
 
524
+ if (args.requireRunArgs) {
525
+ const path = join(runRoot, RUN_ARGS_FILE);
526
+ const present = existsSync(path);
527
+ console.log(JSON.stringify({ ok: present, run_root: runRoot, run_args_path: path,
528
+ reason: present ? null : "no run-args.json at this run root — GATE L0.9b's launch record was never written before this launch" }));
529
+ process.exit(present ? 0 : 6);
530
+ }
531
+
507
532
  const r = report(runRoot, { round: args.round ?? null, gapS: args.gapS });
508
533
  console.log(args.format === "table" ? table(r) : JSON.stringify(r));
509
534
  process.exit(r.launches.length ? 0 : 1);
@@ -6,7 +6,9 @@
6
6
  // Script-first by design: regex over known log formats is free (no model tokens); an
7
7
  // unrecognized line becomes a "raw" triple (file/line unknown) rather than being silently
8
8
  // dropped, so a Sonnet fallback (or a human) still has something to look at — this module never
9
- // invents a file:line it didn't find in the text.
9
+ // invents a file:line it didn't find in the text. `file` and `line` are independent: a
10
+ // diagnostic that names a file but no line number (a resource-compiler error, for example)
11
+ // still yields its file — `line` stays null rather than being guessed at.
10
12
  //
11
13
  // Zero dependencies, zero network — same discipline as oracles/*.
12
14
 
@@ -19,6 +21,18 @@ const PATTERNS = [
19
21
  { re: /^(?:✗|not ok\b.*?)[^()]*\((.+?):(\d+)\)\s*$/, kind: "test-failure" },
20
22
  // ESLint/tsc style: "path/to/file.ts:12:34 - error TS2345: message"
21
23
  { re: /^(.+?):(\d+):\d+\s*[-–]\s*(?:error|warning)\b.*$/, kind: "compiler-diagnostic" },
24
+ // File-level diagnostic with NO line number: "resource.xml: error: message" or
25
+ // "resource.xml - fatal error: message" (resource compilers and linkers report this way —
26
+ // the failure is the whole file, so there is no line to cite). The file must look like a
27
+ // path (ends in a dotted extension, no embedded whitespace/colon) so this stays anchored to
28
+ // real diagnostics rather than matching arbitrary prose that happens to contain "error:".
29
+ { re: /^([^\s:]+?\.[A-Za-z0-9]{1,10})\s*[:\-–—]\s*(?:fatal\s+error|error|warning)\b.*$/i, kind: "compiler-diagnostic" },
30
+ // Bundler style: "ERROR in ./src/components/Foo.tsx" (webpack et al.) — file, no line. The
31
+ // captured token must look like a path — leads with "./"/"../", or ends in a dotted extension
32
+ // of 1-10 alnum chars (same anchor the sibling pattern above uses) — so prose after "ERROR in"
33
+ // ("ERROR in the build pipeline", "ERROR in test suite failed to run") is left unmatched
34
+ // instead of handing back a fabricated file.
35
+ { re: /^(?:ERROR|WARNING)\s+in\s+(\.{1,2}\/[^\s:]*|[^\s:]+\.[A-Za-z0-9]{1,10})\b/i, kind: "compiler-diagnostic" },
22
36
  // Generic "Error: message" line followed later by a stack — capture the message alone.
23
37
  { re: /^\s*(?:Error|TypeError|ReferenceError|AssertionError)\s*:\s*(.+)$/, kind: "error-message" },
24
38
  ];
@@ -46,8 +46,11 @@ import { matchesAny } from "../../hooks/sandbox-guard.mjs";
46
46
  */
47
47
  export function ownership(path, scopes, cwd = null) {
48
48
  const rel = String(path).replace(/^\.\//, "");
49
- const writers = (scopes || []).filter((s) => matchesAny(rel, s.allowed)).map((s) => s.scope_id).sort();
50
49
  const shared = (scopes || []).filter((s) => matchesAny(rel, s.shared || [])).map((s) => s.scope_id).sort();
50
+ // `writers` is admits-the-path, the same union the sandbox fence composes (allowed ++ shared) —
51
+ // a path declared only in a contract's `shared` list is still a scope this path may write, and
52
+ // must read as owned rather than UNOWNED. `shared_with` (below) stays the narrower subset.
53
+ const writers = (scopes || []).filter((s) => matchesAny(rel, s.allowed) || matchesAny(rel, s.shared || [])).map((s) => s.scope_id).sort();
51
54
  return { path: rel, owner: electOwner(rel, scopes), writers, shared_with: shared, exists: cwd ? existsSync(join(cwd, rel)) : null };
52
55
  }
53
56
 
@@ -0,0 +1,296 @@
1
+ #!/usr/bin/env node
2
+ // probe requirements — the way back: a pitch clause, the criterion that graded it, the verdict.
3
+ //
4
+ // WHY THIS IS A QUERY AND NOT A SENTENCE, which is the same reason `probe owner` is one. The
5
+ // requirement matrix is read at GATE L4 and cited by GATE H's census, and both are places where a
6
+ // narrated figure is indistinguishable from a measured one. Measured on the first verdict any run
7
+ // of the spine produced: 85 of 97 criterion rows carried a `traces_to` anchor in the WorkResult and
8
+ // 0 of 97 survived into the projection ingest wrote — so a matrix assembled from memory would have
9
+ // been assembled from an edge that no longer existed on disk. Every figure below is derived from
10
+ // files: the committed registry, the board's `covers:` clauses, the verdict ledger, and the EVAL
11
+ // result's own T0 citations. Nothing is passed in and nothing is written.
12
+ //
13
+ // THE JOIN, and which half is authoritative. A requirement has EVIDENCE when an acceptance
14
+ // criterion covers it (`(covers: REQ-…)` on the AC line — the planning-time edge, reviewed at L1b)
15
+ // AND a criterion that names it passed. `traces_to[]` is the navigation path from the judge's
16
+ // criterion back to the requirement, exactly what the schema already calls it: an anchor, never a
17
+ // grading input. A criterion whose `traces_to` names a REQ that no AC covers is printed as an
18
+ // INCONSISTENCY row and counted as nothing — folding it in would derive one L4 line from two
19
+ // unreconciled sources, which is the failure `probe owner` exists to prevent.
20
+ //
21
+ // IT PROJECTS ONE RUN. `order_id`, round and attempt all repeat across runs of one feature; the
22
+ // run key is the only thing that separates them, so a projection that ignored it would silently mix
23
+ // two runs of one slug. Rows written before the key reached the ledger carry no `run_id`; they are
24
+ // reported as unknown rather than folded into the run being projected.
25
+ //
26
+ // AN EMPTY JOIN READS AS CLEARLY AS A FULL ONE. A tree with no registry, a board with no `covers:`
27
+ // and a run with no verdict are all legitimate states, and each answers "no evidence" rather than
28
+ // failing — that answer is the point of the query, not an error in it.
29
+ //
30
+ // Usage:
31
+ // node "${CLAUDE_PLUGIN_ROOT}/kernel/harness.mjs" probe requirements --slug <slug> [--run-id <id>] [--format json|table] [--cwd <dir>]
32
+ //
33
+ // Exit code: 0 = answered (an empty projection is an answer), 2 = bad argv.
34
+
35
+ import { existsSync, readFileSync, readdirSync } from "node:fs";
36
+ import { resolve, join } from "node:path";
37
+ import { runArgs, isMain } from "../lib/argv.mjs";
38
+ import { requirements as requirementsFile, evaluationDir, resultsDir, readRunId } from "../lib/paths.mjs";
39
+ import { parseRequirements, coveredReqIds } from "../verify/trace.mjs";
40
+ import { readBoard } from "../compile.mjs";
41
+ import { reqId } from "../lib/contract.mjs";
42
+
43
+ /**
44
+ * Read a file, tolerating absence — every input to this projection is optional by design.
45
+ * @param {string} p - Absolute path.
46
+ * @returns {(string|null)} The contents, or null when the file is missing or unreadable.
47
+ */
48
+ function readIf(p) {
49
+ try { return existsSync(p) ? readFileSync(p, "utf8") : null; } catch { return null; }
50
+ }
51
+
52
+ /**
53
+ * The acceptance criteria on the LOCAL board that cover each requirement.
54
+ *
55
+ * The planning-time half of the join, and the authoritative one: a `covers:` clause is written when
56
+ * the plan is still cheap to change and is reviewed at L1b, whereas `traces_to` is written by the
57
+ * judge after the fact.
58
+ *
59
+ * NOT A SECOND COVERS-CLOSURE. Whether a requirement is covered is still decided by
60
+ * `coveredReqIds` in the oracle — there is one implementation of that question and this is not it.
61
+ * This walk exists only to keep the AC the closure discards, so the matrix can print WHICH
62
+ * criterion covers the clause instead of only that one does.
63
+ *
64
+ * @param {Array<object>} board - Task entries from `readBoard` (the only parser that carries
65
+ * `acceptance_criteria`; the scheduling view does not).
66
+ * @returns {Map<string, Array<{task_id:string, ac:string}>>} REQ-id → the ACs naming it, in board
67
+ * order. A requirement no AC covers is simply absent.
68
+ */
69
+ export function coveringAcs(board) {
70
+ const out = new Map();
71
+ for (const task of board || []) {
72
+ for (const ac of task.acceptance_criteria || []) {
73
+ if (typeof ac !== "object" || !Array.isArray(ac.covers)) continue;
74
+ for (const raw of ac.covers) {
75
+ const id = reqId(raw).toUpperCase();
76
+ if (!out.has(id)) out.set(id, []);
77
+ out.get(id).push({ task_id: task.id, ac: ac.text });
78
+ }
79
+ }
80
+ }
81
+ return out;
82
+ }
83
+
84
+ /**
85
+ * Every criterion row ingest projected, tagged with the EVAL target its ledger belongs to.
86
+ *
87
+ * The target is read off the FILE NAME rather than the row: `.verdicts-<target>.jsonl` is written
88
+ * per order, so the file is what says which dispatch produced the rows, and the same file is what
89
+ * locates the WorkResult carrying that dispatch's T0 citations.
90
+ *
91
+ * @param {string} cwd - Project root.
92
+ * @param {string} slug - Feature slug.
93
+ * @returns {Array<object>} One entry per ledger line, each the stored row plus `target`. Malformed
94
+ * lines are skipped; a missing evaluation directory yields [].
95
+ */
96
+ export function readVerdictRows(cwd, slug) {
97
+ const dir = evaluationDir(cwd, slug);
98
+ let files = [];
99
+ try { files = readdirSync(dir).filter((f) => /^\.verdicts-.+\.jsonl$/.test(f)).sort(); } catch { return []; }
100
+ const rows = [];
101
+ for (const f of files) {
102
+ const target = f.replace(/^\.verdicts-/, "").replace(/\.jsonl$/, "");
103
+ for (const line of (readIf(join(dir, f)) || "").split(/\r?\n/)) {
104
+ if (!line.trim()) continue;
105
+ try { rows.push({ ...JSON.parse(line), target }); } catch { /* a truncated line is not a verdict */ }
106
+ }
107
+ }
108
+ return rows;
109
+ }
110
+
111
+ /**
112
+ * The T0 artifacts an EVAL dispatch re-hashed, read from the WorkResult it wrote.
113
+ *
114
+ * The verdict ledger records criteria, not citations, so the machine fact a generator cannot
115
+ * fabricate lives one file over — in `results/<target>.json`. Read here rather than recomputed: the
116
+ * evaluator's own citation is what the round was accepted on.
117
+ *
118
+ * @param {string} cwd - Project root.
119
+ * @param {string} slug - Feature slug.
120
+ * @param {string} target - The EVAL order suffix (`evaluate-r1`), from the ledger's file name.
121
+ * @returns {Array<{scope_id:(string|null), path:(string|null), sha256:(string|null)}>} The cited
122
+ * artifacts; [] when the result is absent, unreadable or cites none.
123
+ */
124
+ export function t0Citations(cwd, slug, target) {
125
+ const body = readIf(join(resultsDir(cwd, slug), `${target}.json`));
126
+ if (!body) return [];
127
+ try {
128
+ const cites = JSON.parse(body)?.verdict?.t0_citations;
129
+ if (!Array.isArray(cites)) return [];
130
+ return cites.map((c) => ({ scope_id: c?.scope_id ?? null, path: c?.path ?? null, sha256: c?.sha256 ?? null }));
131
+ } catch { return []; }
132
+ }
133
+
134
+ /**
135
+ * Project one run's requirement matrix from the artifacts on disk.
136
+ *
137
+ * @param {{cwd:string, slug:string, runId?:(string|null)}} opts - Project root, feature slug, and
138
+ * the run to project; omitted, it is resolved from the run's own receipt.
139
+ * @returns {{slug:string, run_id:(string|null), registry:boolean, totals:object,
140
+ * rows:Array<object>, inconsistencies:Array<object>, ledger:object}} `rows` is one entry per
141
+ * registered clause — its source, status, covering ACs, the criteria that named it and their
142
+ * verdicts, and each criterion's T0 citations — with `evidence` one of `PASS`, `no evidence` or
143
+ * `cut`. `inconsistencies` holds criteria tracing to a REQ no AC covers. `ledger` says how many
144
+ * rows were read, projected, skipped as another run's, and left unkeyed. No registry ⇒ `rows` is
145
+ * empty and `registry` is false: an empty projection, not an error.
146
+ */
147
+ export function projectRequirements({ cwd, slug, runId = undefined }) {
148
+ const regText = readIf(requirementsFile(cwd, slug));
149
+ const clauses = regText === null ? [] : parseRequirements(regText);
150
+ const board = readBoard(cwd, slug);
151
+ // `covered` decides; `acs` only says which criterion did the covering.
152
+ const covered = coveredReqIds(board);
153
+ const acs = coveringAcs(board);
154
+ const run = runId === undefined ? readRunId(cwd, slug) : runId;
155
+
156
+ const all = readVerdictRows(cwd, slug);
157
+ const ledger = { rows_read: all.length, rows_projected: 0, rows_other_run: 0, rows_unknown_run: 0 };
158
+ const mine = [];
159
+ for (const r of all) {
160
+ if (r.run_id === undefined || r.run_id === null || r.run_id === "") { ledger.rows_unknown_run++; continue; }
161
+ if (run !== null && r.run_id === run) { ledger.rows_projected++; mine.push(r); continue; }
162
+ ledger.rows_other_run++;
163
+ }
164
+
165
+ // The criteria that name each requirement, and the T0 artifacts the round that graded them cited.
166
+ const t0Cache = new Map();
167
+ /**
168
+ * The cited T0 artifacts for one EVAL target, read once per target.
169
+ * @param {string} target - The EVAL order suffix.
170
+ * @returns {Array<object>} The citation rows.
171
+ */
172
+ const t0For = (target) => {
173
+ if (!t0Cache.has(target)) t0Cache.set(target, t0Citations(cwd, slug, target));
174
+ return t0Cache.get(target);
175
+ };
176
+
177
+ const byReq = new Map();
178
+ const inconsistencies = [];
179
+ for (const r of mine) {
180
+ const anchors = Array.isArray(r.traces_to) ? r.traces_to : [];
181
+ for (const raw of anchors) {
182
+ const id = reqId(raw).toUpperCase();
183
+ const entry = {
184
+ criterion: r.criterion ?? "", dimension: r.dimension ?? "", verdict: r.verdict ?? "",
185
+ confidence: r.confidence ?? null, evidence: r.evidence ?? "", target: r.target,
186
+ run: r.run ?? null, run_id: r.run_id ?? null, t0: t0For(r.target),
187
+ };
188
+ // THE AUTHORITATIVE HALF DECIDES. An anchor pointing at a requirement no acceptance criterion
189
+ // covers is a claim the plan never made; it is printed so somebody can reconcile it, and
190
+ // counted as nothing.
191
+ if (!covered.has(id)) { inconsistencies.push({ requirement: id, ...entry, why: "no acceptance criterion covers this requirement — the anchor resolves to nothing the plan claimed" }); continue; }
192
+ if (!byReq.has(id)) byReq.set(id, []);
193
+ byReq.get(id).push(entry);
194
+ }
195
+ }
196
+
197
+ const rows = clauses.map((c) => {
198
+ const id = c.id.toUpperCase();
199
+ const covering = acs.get(id) || [];
200
+ const criteria = byReq.get(id) || [];
201
+ const cut = c.status !== "covered";
202
+ const passed = criteria.some((x) => String(x.verdict).toUpperCase() === "PASS");
203
+ return {
204
+ id: c.id,
205
+ source: c.source || "",
206
+ clause: c.clause || "",
207
+ status: c.status,
208
+ covering_acs: covering,
209
+ criteria,
210
+ t0: [...new Set(criteria.flatMap((x) => x.t0.map((t) => t.sha256).filter(Boolean)))],
211
+ evidence: cut ? "cut" : passed ? "PASS" : "no evidence",
212
+ };
213
+ });
214
+
215
+ const totals = {
216
+ total: rows.length,
217
+ pass: rows.filter((r) => r.evidence === "PASS").length,
218
+ cut: rows.filter((r) => r.evidence === "cut").length,
219
+ no_evidence: rows.filter((r) => r.evidence === "no evidence").length,
220
+ inconsistencies: inconsistencies.length,
221
+ };
222
+ return { slug, run_id: run, registry: regText !== null, totals, rows, inconsistencies, ledger };
223
+ }
224
+
225
+ /**
226
+ * The one-line summary GATE L4 prints and GATE H's census cites.
227
+ *
228
+ * @param {object} r - A report from {@link projectRequirements}.
229
+ * @returns {string} `15/17 PASS · 1 CUT (PO) · 1 no evidence (REQ-12 ← R12)`, or the empty-run
230
+ * phrasing when there is no registry to project.
231
+ */
232
+ export function summaryLine(r) {
233
+ if (!r.registry) return "n/a (no registry)";
234
+ if (!r.totals.total) return "0 requirements registered";
235
+ const gaps = r.rows.filter((x) => x.evidence === "no evidence");
236
+ const named = gaps.slice(0, 3).map((x) => (x.source ? `${x.id} ← ${x.source}` : x.id)).join(", ");
237
+ const parts = [`${r.totals.pass}/${r.totals.total} PASS`];
238
+ if (r.totals.cut) parts.push(`${r.totals.cut} CUT (PO)`);
239
+ if (r.totals.no_evidence) parts.push(`${r.totals.no_evidence} no evidence (${named}${gaps.length > 3 ? ", …" : ""})`);
240
+ if (r.totals.inconsistencies) parts.push(`${r.totals.inconsistencies} inconsistency ${r.totals.inconsistencies === 1 ? "row" : "rows"}`);
241
+ return parts.join(" · ");
242
+ }
243
+
244
+ /**
245
+ * Render the matrix as a fixed-width table for a human reading a gate block.
246
+ * @param {object} r - A report from {@link projectRequirements}.
247
+ * @returns {string} The header, one row per requirement, the summary line and any inconsistencies.
248
+ */
249
+ export function renderTable(r) {
250
+ const out = [];
251
+ const head = ["REQ", "source", "status", "covering AC", "criterion", "verdict", "T0"];
252
+ const rows = r.rows.map((x) => [
253
+ x.id, x.source || "—", x.evidence,
254
+ x.covering_acs.length ? `${x.covering_acs[0].task_id}: ${x.covering_acs[0].ac.slice(0, 40)}${x.covering_acs.length > 1 ? ` (+${x.covering_acs.length - 1})` : ""}` : "—",
255
+ x.criteria.length ? `${x.criteria[0].criterion.slice(0, 40)}${x.criteria.length > 1 ? ` (+${x.criteria.length - 1})` : ""}` : "—",
256
+ x.criteria.length ? x.criteria.map((c) => c.verdict).join(",") : "—",
257
+ x.t0.length ? x.t0.map((h) => String(h).slice(0, 12)).join(",") : "—",
258
+ ]);
259
+ if (rows.length) {
260
+ const w = head.map((h, i) => Math.max(h.length, ...rows.map((row) => String(row[i]).length)));
261
+ const line = (row) => row.map((c, i) => String(c).padEnd(w[i])).join(" ").trimEnd();
262
+ out.push(line(head), line(w.map((n) => "-".repeat(n))), ...rows.map(line), "");
263
+ }
264
+ out.push(`Requirements: ${summaryLine(r)}`);
265
+ out.push(`run_id: ${r.run_id ?? "unknown"} · ledger rows read ${r.ledger.rows_read}, projected ${r.ledger.rows_projected}, other run ${r.ledger.rows_other_run}, unkeyed (run_id: unknown) ${r.ledger.rows_unknown_run}`);
266
+ for (const i of r.inconsistencies) {
267
+ out.push(` ⚠ ${i.requirement}: "${String(i.criterion).slice(0, 60)}" ${i.verdict} — ${i.why}`);
268
+ }
269
+ return out.join("\n");
270
+ }
271
+
272
+ export const ARGV_SPEC = {
273
+ usage: "harness.mjs probe requirements --slug <slug> [--run-id <id>] [--format json|table] [--cwd <dir>]",
274
+ _: { arity: 0, max: 0, name: "(no positional operands)" },
275
+ slug: { type: "str", required: true },
276
+ "run-id": { type: "str" },
277
+ format: { type: "enum", values: ["json", "table"], default: "json" },
278
+ cwd: { type: "path" },
279
+ };
280
+
281
+ /**
282
+ * Answer "which requirement has evidence, and from which criterion" from the artifacts on disk.
283
+ *
284
+ * @param {string[]} rawArgv - The subcommand's own arguments (harness.mjs strips the verb words).
285
+ * @returns {Promise<void>} Exits 0 with the projection — an absent run, an absent registry and an
286
+ * absent verdict are all answers, never errors.
287
+ */
288
+ export async function cli(rawArgv) {
289
+ const args = runArgs(ARGV_SPEC, rawArgv);
290
+ const cwd = resolve(args.cwd || process.cwd());
291
+ const report = projectRequirements({ cwd, slug: args.slug, runId: args.runId ?? undefined });
292
+ console.log(args.format === "table" ? renderTable(report) : JSON.stringify(report, null, 2));
293
+ process.exit(0);
294
+ }
295
+
296
+ if (isMain(import.meta.url)) cli(process.argv.slice(2));