shapeup-sdlc 3.4.0 → 3.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (43) hide show
  1. package/.claude-plugin/plugin.json +1 -1
  2. package/AGENTS.md +16 -4
  3. package/README.md +7 -3
  4. package/SECURITY.md +1 -1
  5. package/hooks/sandbox-guard.mjs +69 -6
  6. package/kernel/compile.mjs +33 -12
  7. package/kernel/harness.mjs +10 -4
  8. package/kernel/init/run-args.mjs +206 -0
  9. package/kernel/init/run.mjs +10 -0
  10. package/kernel/lib/contract.mjs +68 -1
  11. package/kernel/lib/paths.mjs +10 -0
  12. package/kernel/probe/concurrency.mjs +31 -6
  13. package/kernel/probe/digest.mjs +15 -1
  14. package/kernel/probe/owner.mjs +4 -1
  15. package/kernel/probe/requirements.mjs +296 -0
  16. package/kernel/probe/resume.mjs +195 -6
  17. package/kernel/probe/rounds.mjs +104 -0
  18. package/kernel/reduce/graph.mjs +5 -2
  19. package/kernel/reduce/ingest.mjs +69 -15
  20. package/kernel/reduce/ship.mjs +52 -31
  21. package/kernel/reduce/snapshot.mjs +23 -2
  22. package/kernel/report/export.mjs +54 -2
  23. package/kernel/report/facts.mjs +24 -2
  24. package/{skills/tech-lead → kernel}/schemas/domain.schema.json +20 -12
  25. package/kernel/verify/envelope.mjs +2 -2
  26. package/kernel/verify/skills.mjs +1 -1
  27. package/kernel/verify/spec.mjs +130 -4
  28. package/kernel/verify/trace.mjs +16 -7
  29. package/package.json +1 -1
  30. package/skills/ba-pitch-analyzer/SKILL.md +16 -1
  31. package/skills/coach/SKILL.md +8 -2
  32. package/skills/hill-chart/SKILL.md +3 -4
  33. package/skills/scope-architect/SKILL.md +16 -1
  34. package/skills/scope-hammer/SKILL.md +11 -2
  35. package/skills/spec-evaluator/SKILL.md +12 -1
  36. package/skills/tech-lead/SKILL.md +10 -10
  37. package/skills/tech-lead/references/gates.md +70 -12
  38. package/skills/tech-lead/references/protocol.md +4 -2
  39. package/skills/tech-lead/workflows/shapeup-run.js +176 -38
  40. package/skills/translator/SKILL.md +1 -1
  41. /package/{skills/tech-lead → kernel}/schemas/gate-answers.schema.json +0 -0
  42. /package/{skills/tech-lead → kernel}/schemas/work-order.schema.json +0 -0
  43. /package/{skills/tech-lead → kernel}/schemas/work-result.schema.json +0 -0
@@ -59,18 +59,27 @@
59
59
  import { existsSync, readdirSync, readFileSync, writeFileSync, mkdirSync } from "node:fs";
60
60
  import { dirname, join, resolve } from "node:path";
61
61
  import { runArgs } from "../lib/argv.mjs";
62
- import { splitFrontmatter } from "../lib/contract.mjs";
62
+ import { splitFrontmatter, uncoerce } from "../lib/contract.mjs";
63
63
  import { globToRegExp } from "../verify/spec.mjs";
64
64
  import {
65
65
  intake, harnessRun, wiringMap, projectProfile, scopesDir, resultsDir, ordersDir,
66
- orientDir, activeOrder, usecasesDir, breadboard, receipt, readReceipt,
66
+ orientDir, activeOrder, usecasesDir, breadboard, receipt, readReceipt, requirements,
67
67
  } from "../lib/paths.mjs";
68
68
  import { evalVerdict } from "./eval.mjs";
69
69
 
70
70
  /** The run-state values `references/protocol.md` (Part 4 — State) defines. A typo'd status is a rejection,
71
71
  * not a write — the whole point of this file is that a write nobody validates is a write nobody
72
- * can trust. */
73
- export const RUN_STATUSES = ["orienting", "mapping", "building", "evaluating", "shipped", "escalated"];
72
+ * can trust. `aborted` is the terminal counterpart `escalated` already was: a gate
73
+ * resolving "abort", or a hard stop, ends the run the same way an operator-declared escalation
74
+ * does — neither resumes on relaunch — so both are TERMINAL_STATUSES below. */
75
+ export const RUN_STATUSES = ["orienting", "mapping", "building", "evaluating", "shipped", "escalated", "aborted"];
76
+
77
+ /**
78
+ * The statuses a run does not come back from. `closeRun` refuses every other member of
79
+ * {@link RUN_STATUSES} — `orienting`/`mapping`/`building`/`evaluating` are mid-flight, and writing
80
+ * a close over one of those would stamp `closed_at` on a run a relaunch is still meant to resume.
81
+ */
82
+ export const TERMINAL_STATUSES = ["shipped", "aborted", "escalated"];
74
83
 
75
84
  /** ORIENT's four artifacts (skills/orient/SKILL.md §Outputs): three by exact name, plus a spike
76
85
  * whose filename carries the area it spiked (`spike-<area>.md`, or `spike-not-needed.md` when
@@ -394,6 +403,12 @@ export function deriveResumeState(cwd, slug) {
394
403
  orient_dir: `.shapeup/${slug}/orient/`,
395
404
  has_orient_artifacts: hasOrientArtifacts(cwd, slug),
396
405
  has_spec_tree: hasSpecTree(cwd, slug, hr.spec_folder || null),
406
+ // A PLAIN FACT, DELIBERATELY NOT A PHASE. The requirements registry is dispatched once, before
407
+ // ANALYZE, and the orchestrator guards that one dispatch on this boolean. It is NOT an entry in
408
+ // PHASE_ARTIFACT, and adding it there would be a migration hazard rather than a tidier shape:
409
+ // that map is also `nextPhase()`'s ordered list, so every run recorded before the registry
410
+ // existed would fast-forward to the registry instead of to `build` on its next relaunch.
411
+ has_requirements: existsSync(requirements(cwd, slug)),
397
412
  has_wiring_map: existsSync(wiringMap(cwd, slug)),
398
413
  project_profile_path: projectProfile(cwd, slug),
399
414
  has_project_profile: existsSync(projectProfile(cwd, slug)),
@@ -456,6 +471,166 @@ export function setRunStatus(cwd, slug, status) {
456
471
  return { ok: true, path: p, status };
457
472
  }
458
473
 
474
+ /**
475
+ * Rewrite `status:`, `closed_at:`, `closed_status:` and `close_cause:` together, in one pass,
476
+ * appending any of the last three that a pre-migration ledger carries no line for yet (the same
477
+ * tolerant-of-old-ledgers discipline `close_cause` itself shipped under).
478
+ *
479
+ * @param {string} body - The ledger's current text.
480
+ * @param {{status:string, closedAt:string, cause:(string|null)}} o - What to write. `cause` is
481
+ * already normalized prose (newlines collapsed, truncated) — this function only `uncoerce`s it.
482
+ * @returns {string} The rewritten text.
483
+ */
484
+ function writeCloseLines(body, { status, closedAt, cause }) {
485
+ const causeLine = `close_cause: ${uncoerce(cause || null)}`;
486
+ const closedStatusLine = `closed_status: ${status}`;
487
+ let out = body
488
+ .replace(/^status:.*$/m, `status: ${status}`)
489
+ .replace(/^closed_at:.*$/m, `closed_at: ${closedAt}`);
490
+ out = /^closed_status:.*$/m.test(out)
491
+ ? out.replace(/^closed_status:.*$/m, closedStatusLine)
492
+ // A ledger written before this field existed carries no line to replace — appended right after
493
+ // `closed_at:`, the one line every TERMINAL_STATUSES write also touches.
494
+ : out.replace(/^closed_at:.*$/m, (m) => `${m}\n${closedStatusLine}`);
495
+ out = /^close_cause:.*$/m.test(out)
496
+ ? out.replace(/^close_cause:.*$/m, causeLine)
497
+ : out.replace(/^closed_at:.*$/m, (m) => `${m}\n${causeLine}`);
498
+ return out;
499
+ }
500
+
501
+ /**
502
+ * Close the run: a terminal status, its cause, and a close timestamp, written together in ONE
503
+ * pass.
504
+ *
505
+ * Measured: after an EVAL worker escalated and the run aborted, `harness-run.md` still read
506
+ * `status: evaluating`, `closed_at: ~`, with no cause recorded anywhere — a live EVAL and a dead
507
+ * one were indistinguishable from the trace alone. Two writers made that possible: `setRunStatus`
508
+ * above replaces the `status:` line and NOTHING ELSE, and `closed_at` was written exactly once, as
509
+ * the literal `~`, by `init run` — nothing ever replaced it. This function is the one call site
510
+ * that closes a run, so a terminal RunReturn cannot leave one of the three facts behind.
511
+ *
512
+ * Refuses rather than silently no-ops, the same discipline as `setRunStatus`: an absent ledger, one
513
+ * missing the lines this writes, or a non-terminal `status` (closing a run still `building` would
514
+ * stamp a live run as done) is a fact to act on, not a write to skip quietly.
515
+ *
516
+ * THE ONCE-ONLY GUARD READS `closed_status`, NEVER `status`. REWORK (round 2): the guard used to key
517
+ * on `before.status`, and `status:` is a LIVE field every phase rewrites via `setRunStatus` —
518
+ * including the product's own ship path, which stamps `status: shipped`
519
+ * (`skills/tech-lead/workflows/shapeup-run.js`'s Ship phase) immediately before this call runs. So a
520
+ * run closed `aborted` at Preflight on one launch, relaunched, and carried through to a `shipped`
521
+ * RunReturn on a later one had its `status:` line rewritten to `shipped` by that ordinary phase
522
+ * traffic BEFORE `closeIfTerminal` ever called this function — the guard read `before.status ===
523
+ * "shipped"`, matched the very close it was about to perform, and treated a run that was actually
524
+ * closed `aborted` as already closed `shipped`: `ok:true`, an "idempotent no-op" that silently kept
525
+ * the abort's own `closed_at`/`close_cause` under a `status:` line now reading `shipped`. `closed_status`
526
+ * is written ONLY here, exactly once per distinct close-writing call, so nothing between two calls to
527
+ * this function can move it — it is the one field that actually answers "has this run been closed,
528
+ * and to what" regardless of how many times `status:` has been rewritten since.
529
+ *
530
+ * WHAT A SECOND CLOSE MEANS, decided explicitly rather than left implicit. `run_id` is reused across
531
+ * relaunches by design (AGENTS.md), so "the run's close" and "this launch's own close" are two
532
+ * different facts a single `closed_at`/`close_cause` pair cannot both hold:
533
+ * - The IDENTICAL status and the IDENTICAL cause is the ordinary case a retried or duplicated call
534
+ * produces (the same `withWarnings` call, or a relaunch that re-executes an already-applied
535
+ * close) — a true no-op, `ok:true`, nothing rewritten.
536
+ * - The SAME terminal status but a DIFFERENT cause is a SECOND, real close — most often a later
537
+ * relaunch aborting again for its own reason, or shipping again after an earlier ship's close
538
+ * record was never superseded. Discarding it (the pre-rework behavior) silently drops the later
539
+ * launch's own reason with no trace of the loss. It is recorded instead: this call's cause
540
+ * becomes the ledger's `close_cause`, folded together with the prior cause it is superseding —
541
+ * the earlier fact survives inside the new line rather than the ledger simply losing it — and the
542
+ * return carries `superseded:true` so a caller (`closeIfTerminal`) can flag the trace as degraded
543
+ * rather than reporting a clean success.
544
+ * - A DIFFERENT terminal status altogether (aborted vs. shipped) is refused outright — flipping the
545
+ * actual OUTCOME of a run after the fact is not a fact a later launch gets to silently overwrite,
546
+ * so the original `closed_status`/`closed_at`/`close_cause` are left completely untouched and
547
+ * handed back to the caller.
548
+ *
549
+ * @param {string} cwd - Project root.
550
+ * @param {string} slug - Feature slug.
551
+ * @param {{status:string, cause:(string|null)}} o - The terminal status (one of
552
+ * {@link TERMINAL_STATUSES}) and why the run ended there. `cause` travels through `uncoerce` (the
553
+ * one dialect `harness-run.md`'s frontmatter is read and written in), so free prose — quotes and
554
+ * colons included — round-trips as one frontmatter line; an embedded newline is collapsed to a
555
+ * space first, because this dialect is line-based and could not carry one either way.
556
+ * @returns {{ok:boolean, path:string, status:string, closed_at?:string, cause?:(string|null),
557
+ * reason?:string, closed_status?:string, superseded?:boolean, decision?:string,
558
+ * prior_cause?:(string|null), prior_closed_at?:string}} Outcome. A refused overwrite (already
559
+ * closed with a DIFFERENT terminal status) carries `closed_status`/`closed_at`/`cause` naming what
560
+ * is actually on disk. A successful supersede (same status, different cause) carries
561
+ * `superseded:true`, `decision:"superseded"` (the one-token signal the courier boundary in
562
+ * `shapeup-run.js` relays verbatim — see its own `cmd()` banner) and the prior close it folded in.
563
+ */
564
+ export function closeRun(cwd, slug, { status, cause = null } = {}) {
565
+ const p = harnessRun(cwd, slug);
566
+ if (!TERMINAL_STATUSES.includes(status)) {
567
+ return { ok: false, path: p, status, reason: `closeRun: "${status}" is not terminal — expected one of ${TERMINAL_STATUSES.join(" | ")}` };
568
+ }
569
+ if (!existsSync(p)) {
570
+ return { ok: false, path: p, status, reason: `no harness-run.md for slug "${slug}" — open the run with harness init run (GATE L0.1) before closing it` };
571
+ }
572
+ let body = readFileSync(p, "utf8");
573
+ if (!/^status:.*$/m.test(body) || !/^closed_at:.*$/m.test(body)) {
574
+ return { ok: false, path: p, status, reason: `harness-run.md carries no "status:"/"closed_at:" line to replace — the ledger's frontmatter is malformed (references/protocol.md)` };
575
+ }
576
+
577
+ // Truncated, not elided: a cause this long has already done its job in the run's own log — the
578
+ // ledger line is a pointer back to it, not the full transcript. Newlines are collapsed to spaces
579
+ // FIRST — this dialect is line-based, so a raw embedded newline would split one field into a value
580
+ // line and a stray, unparsed one.
581
+ const normCause = String(cause ?? "").replace(/\r?\n/g, " ").trim().slice(0, 4000) || null;
582
+
583
+ const before = parseFrontmatter(body);
584
+ const priorClosedStatus = before.closed_status && before.closed_status !== "~" ? before.closed_status : null;
585
+ const priorClosedAt = before.closed_at && before.closed_at !== "~" ? before.closed_at : null;
586
+ const priorCause = before.close_cause && before.close_cause !== "~" ? before.close_cause : null;
587
+
588
+ if (priorClosedStatus && priorClosedAt) {
589
+ if (priorClosedStatus === status && normCause === priorCause) {
590
+ // The identical fact, restated — a retried or duplicated call costs nothing.
591
+ return { ok: true, path: p, status, closed_at: priorClosedAt, cause: priorCause, decision: "idempotent", reason: `already closed as "${status}" at ${priorClosedAt} — idempotent no-op` };
592
+ }
593
+ if (priorClosedStatus !== status) {
594
+ // A DIFFERENT terminal status over an already-closed run — refused outright, the original
595
+ // close left completely untouched so the caller can see what it was refused permission to
596
+ // destroy, rather than losing it silently.
597
+ return {
598
+ ok: false, path: p, status,
599
+ reason: `closeRun: this run is already closed as "${priorClosedStatus}" at ${priorClosedAt} (cause: ${JSON.stringify(priorCause)}) — refusing to overwrite it with "${status}". A terminal close is a once-only fact; the first cause is not destroyed.`,
600
+ closed_status: priorClosedStatus, closed_at: priorClosedAt, cause: priorCause,
601
+ };
602
+ }
603
+ // SAME terminal status, a DIFFERENT cause — a second, real close (see the function banner's
604
+ // "what a second close means"). Superseded, not discarded: the prior cause is folded into the
605
+ // new line rather than lost, and the return says so explicitly.
606
+ const closedAt = new Date().toISOString();
607
+ const foldedCause = `${normCause || "no reason recorded"} — supersedes an earlier close recorded ${priorClosedAt} (cause: ${JSON.stringify(priorCause)})`.slice(0, 4000);
608
+ body = writeCloseLines(body, { status, closedAt, cause: foldedCause });
609
+ try { writeFileSync(p, body); } catch (e) {
610
+ return { ok: false, path: p, status, reason: `could not write the ledger: ${e.message}` };
611
+ }
612
+ const afterSup = parseFrontmatter(readFileSync(p, "utf8"));
613
+ if (afterSup.status !== status || !afterSup.closed_at || afterSup.closed_at === "~") {
614
+ return { ok: false, path: p, status, reason: `wrote the superseding close but the ledger reads back status="${afterSup.status}" closed_at="${afterSup.closed_at}" — the write did not take` };
615
+ }
616
+ return {
617
+ ok: true, path: p, status, closed_at: afterSup.closed_at, cause: afterSup.close_cause ?? null,
618
+ superseded: true, decision: "superseded", prior_cause: priorCause, prior_closed_at: priorClosedAt,
619
+ };
620
+ }
621
+
622
+ const closedAt = new Date().toISOString();
623
+ body = writeCloseLines(body, { status, closedAt, cause: normCause });
624
+ try { writeFileSync(p, body); } catch (e) {
625
+ return { ok: false, path: p, status, reason: `could not write the ledger: ${e.message}` };
626
+ }
627
+ const after = parseFrontmatter(readFileSync(p, "utf8"));
628
+ if (after.status !== status || !after.closed_at || after.closed_at === "~") {
629
+ return { ok: false, path: p, status, reason: `wrote the close but the ledger reads back status="${after.status}" closed_at="${after.closed_at}" — the write did not take` };
630
+ }
631
+ return { ok: true, path: p, status, closed_at: after.closed_at, cause: after.close_cause ?? null, decision: "closed" };
632
+ }
633
+
459
634
  /**
460
635
  * Point the substrate pointer at the order about to be executed.
461
636
  *
@@ -482,13 +657,17 @@ export function writeActiveOrder(cwd, slug, orderPath) {
482
657
 
483
658
  /** The typed argv contract (see `./lib/argv.mjs`). */
484
659
  export const ARGV_SPEC = {
485
- usage: "harness.mjs probe resume --slug <slug> [--cwd <dir>] [--require <phase> | --set-status <status> | --set-active-order <path>]",
660
+ usage: "harness.mjs probe resume --slug <slug> [--cwd <dir>] " +
661
+ "[--require <phase> | --set-status <status> | --set-active-order <path> | --close <status> [--cause <text>]]",
486
662
  _: { arity: 0, max: 0, name: "(no positional operands)" },
487
663
  slug: { type: "str", required: true },
488
664
  cwd: { type: "path" },
489
665
  require: { type: "enum", values: PHASES },
490
666
  "set-status": { type: "enum", values: RUN_STATUSES },
491
667
  "set-active-order": { type: "str" },
668
+ // The one call site that stamps a terminal status, its cause and closed_at together.
669
+ close: { type: "enum", values: TERMINAL_STATUSES },
670
+ cause: { type: "str" },
492
671
  };
493
672
 
494
673
  /**
@@ -502,11 +681,15 @@ export function cli(rawArgv) {
502
681
  const args = runArgs(ARGV_SPEC, rawArgv);
503
682
  const cwd = args.cwd || process.cwd();
504
683
 
505
- const ops = [args.require && "--require", args.setStatus && "--set-status", args.setActiveOrder && "--set-active-order"].filter(Boolean);
684
+ const ops = [args.require && "--require", args.setStatus && "--set-status", args.setActiveOrder && "--set-active-order", args.close && "--close"].filter(Boolean);
506
685
  if (ops.length > 1) {
507
686
  process.stderr.write(JSON.stringify({ error: "conflicting_flags", flags: ops, expected: "one operation per invocation" }) + "\n");
508
687
  process.exit(2);
509
688
  }
689
+ if (args.cause !== undefined && !args.close) {
690
+ process.stderr.write(JSON.stringify({ error: "conflicting_flags", flags: ["--cause"], expected: "--cause is only meaningful with --close" }) + "\n");
691
+ process.exit(2);
692
+ }
510
693
 
511
694
  // The post-condition. It prints the SAME ResumeState the derivation prints — plus which phase was
512
695
  // asked about and whether its artifact is there — so a caller that wants to act on the facts
@@ -535,6 +718,12 @@ export function cli(rawArgv) {
535
718
  process.exit(r.ok ? 0 : 3);
536
719
  }
537
720
 
721
+ if (args.close) {
722
+ const r = closeRun(cwd, args.slug, { status: args.close, cause: args.cause ?? null });
723
+ console.log(JSON.stringify(r));
724
+ process.exit(r.ok ? 0 : 3);
725
+ }
726
+
538
727
  console.log(JSON.stringify(deriveResumeState(cwd, args.slug)));
539
728
  process.exit(0);
540
729
  }
@@ -0,0 +1,104 @@
1
+ // rounds — how many BUILD/EVAL rounds a run actually reached, derived from disk.
2
+ //
3
+ // WHY THIS EXISTS (measured, not theorized).
4
+ //
5
+ // `harness-run.md`'s `rounds_used` frontmatter is written ONCE, at GATE L0.1 (`init run`), as `0`,
6
+ // and nothing in the round loop ever rewrites it — the orchestrator's own RunReturn carries the
7
+ // real count, but that value lives in a JS variable a relaunch loses, and it never reached the
8
+ // ledger. `reduce ship`'s own derivation partly compensated by counting the highest
9
+ // `evaluate-r<N>.json` result on disk, which is right for "rounds the judge saw" and wrong for
10
+ // "rounds the run built": measured after two full BUILD rounds with neither reaching EVAL,
11
+ // `round_count: 0` and `rounds_used: 0` both — a run that did real work reported having done none.
12
+ //
13
+ // TWO NUMBERS, NOT ONE. A round can be built and die before EVAL ever sees it (a circuit breaker,
14
+ // a kill mid-round), so "rounds built" and "rounds judged" are different facts and collapsing them
15
+ // into a single field is exactly what made the fallback silently wrong. Both are returned here, and
16
+ // both are meant to survive to wherever a run's numbers are read — the ship report and the export.
17
+ //
18
+ // PURE. Reads the run's own trace, writes nothing — the same discipline as `reduce ship`'s other
19
+ // derivations (`t0Summary`, `boardCensus`, …). A run before any of these artifacts existed, or a
20
+ // lane with no round concept at all (`--tiny`), falls back to the caller-supplied ledger value,
21
+ // non-regression with the earlier, EVAL-only behaviour.
22
+
23
+ import { existsSync, readdirSync, readFileSync } from "node:fs";
24
+ import { join } from "node:path";
25
+ import { ordersDir, verdictsDir, roundBuildDir, resultsDir } from "../lib/paths.mjs";
26
+
27
+ /** Parse a JSON file, returning null rather than throwing — every reader here is best-effort. */
28
+ function readJson(p) {
29
+ try { return JSON.parse(readFileSync(p, "utf8")); } catch { return null; }
30
+ }
31
+
32
+ /**
33
+ * The round a build-addressed order id encodes, e.g. `checkout/sc-01-r2-a1` → 2.
34
+ * @param {*} orderId - The order's own id, whatever shape it happens to be.
35
+ * @returns {(number|null)} The round, or null when the id carries none (`orient`, `wire`, `hammer`, …).
36
+ */
37
+ export function orderRound(orderId) {
38
+ const suffix = String(orderId ?? "").split("/").slice(1).join("/");
39
+ const m = suffix.match(/-r(\d+)(?:-a\d+)?$/);
40
+ return m ? Number(m[1]) : null;
41
+ }
42
+
43
+ const maxOf = (nums) => (nums.length ? Math.max(...nums) : null);
44
+
45
+ /**
46
+ * How many rounds this run actually reached — built, and separately, judged.
47
+ *
48
+ * `rounds_used` is the highest round carrying ANY build evidence: a compiled order, a T0 verdict
49
+ * artifact, a round build-gate artifact, or an EVAL result — the brief's own list, plus EVAL results
50
+ * because a judged round is, by construction, a round that was also built. `rounds_judged` is the
51
+ * highest round EVAL actually returned a verdict for (`evaluate-r<N>.json` on disk), kept as its own
52
+ * field rather than folded into the only number a reader can see.
53
+ *
54
+ * @param {string} cwd - Project root.
55
+ * @param {string} slug - Feature slug.
56
+ * @param {*} [fallback] - What to report for `rounds_used` when nothing on disk is derivable — the
57
+ * caller's own ledger value, passed through untouched (some callers hand this on already coerced
58
+ * to a number, some as the raw frontmatter string; this function does not care which).
59
+ * @returns {{rounds_used:*, rounds_judged:(number|null)}} Both counts.
60
+ */
61
+ export function deriveRounds(cwd, slug, fallback) {
62
+ const orderRounds = [];
63
+ const oDir = ordersDir(cwd, slug);
64
+ if (existsSync(oDir)) {
65
+ for (const f of readdirSync(oDir)) {
66
+ if (!f.endsWith(".json")) continue;
67
+ const r = orderRound(readJson(join(oDir, f))?.order_id);
68
+ if (r !== null) orderRounds.push(r);
69
+ }
70
+ }
71
+
72
+ const verdictRounds = [];
73
+ const vDir = verdictsDir(cwd, slug);
74
+ if (existsSync(vDir)) {
75
+ for (const f of readdirSync(vDir).filter((x) => x.endsWith(".json"))) {
76
+ const v = readJson(join(vDir, f));
77
+ if (typeof v?.round === "number") verdictRounds.push(v.round);
78
+ }
79
+ }
80
+
81
+ const buildGateRounds = [];
82
+ const bDir = roundBuildDir(cwd, slug);
83
+ if (existsSync(bDir)) {
84
+ for (const f of readdirSync(bDir)) {
85
+ const m = f.match(/^r(\d+)-t\d+\.json$/);
86
+ if (m) buildGateRounds.push(Number(m[1]));
87
+ }
88
+ }
89
+
90
+ const evalRounds = [];
91
+ const rDir = resultsDir(cwd, slug);
92
+ if (existsSync(rDir)) {
93
+ for (const f of readdirSync(rDir)) {
94
+ const m = f.match(/^evaluate-r(\d+)\.json$/);
95
+ if (m) evalRounds.push(Number(m[1]));
96
+ }
97
+ }
98
+
99
+ const built = maxOf([...orderRounds, ...verdictRounds, ...buildGateRounds, ...evalRounds]);
100
+ return {
101
+ rounds_used: built !== null ? built : fallback,
102
+ rounds_judged: maxOf(evalRounds),
103
+ };
104
+ }
@@ -32,7 +32,7 @@ import {
32
32
  localRoot, receipt as receiptPath, ordersDir, resultsDir, verdictsDir, trials as trialsPath,
33
33
  gates as gatesPath, scopesDir, usecasesDir, requirements as requirementsPath, wiringMap as wiringMapPath,
34
34
  } from "../lib/paths.mjs";
35
- import { readAllContracts, readContract, ucId, SCOPE_CONTRACT, WIRING_MAP } from "../lib/contract.mjs";
35
+ import { readAllContracts, readContract, ucId, reqId, SCOPE_CONTRACT, WIRING_MAP } from "../lib/contract.mjs";
36
36
  import { runIdFromReceipt } from "../lib/paths.mjs";
37
37
 
38
38
  /** The graph's home — one file per feature, beside the run trace it projects. */
@@ -263,7 +263,10 @@ export function project(cwd, slug) {
263
263
  // UseCase nodes it was built from and `--trace` stopped there instead of reaching the
264
264
  // objective. The contract now names both, so both are projectable.
265
265
  for (const uc of contract.use_cases || []) edge(id, "IMPLEMENTS", `uc:${slug}:${ucId(uc)}`);
266
- for (const req of contract.covers || []) edge(id, "COVERS", `req:${slug}:${req}`);
266
+ // Normalised to the registry's key space on the way in, the same mapping spec-lint applies:
267
+ // a contract citing the pitch's `R<n>` and a Requirement node keyed `REQ-<n>` are one node,
268
+ // and an edge drawn to the other spelling is an edge to a node the graph does not hold.
269
+ for (const req of contract.covers || []) edge(id, "COVERS", `req:${slug}:${reqId(req)}`);
267
270
  for (const dep of contract.depends_on || []) edge(id, "DEPENDS_ON", `scope:${slug}:${dep}`);
268
271
  }
269
272
  }
@@ -8,7 +8,16 @@
8
8
  // task_results[] → tick AC boxes, flip task frontmatter status, append Execution Log,
9
9
  // update tasks/_index.md row, propagate unblocks (old P3.1–P3.6)
10
10
  // discoveries[] → append to .shapeup/<slug>/discovery/ledger.md (old P3.7 / QA H.3)
11
- // verdict.criteria[] → append evaluation/.verdicts-<target>.jsonl (old evaluator B.0)
11
+ // deviations[] (when → append to the SAME discovery ledger, tagged distinctly. A worker's
12
+ // status:"escalated") protocol names deviations[] as the only channel a blocked worker has —
13
+ // there is no escalates[] field — and until now nothing routed it anywhere:
14
+ // a worker that correctly stopped rather than guessed produced a number and
15
+ // silence. Landing here reaches a reader that already exists — the census
16
+ // reads this ledger's open entries, and the ship report's own "Discovered,
17
+ // not built" section is generated from it.
18
+ // verdict.criteria[] → append evaluation/.verdicts-<target>.jsonl (old evaluator B.0), every
19
+ // row keyed by run_id and carrying the judge's traces_to[] anchor back to
20
+ // the requirement — see step 4 for why neither may be dropped here
12
21
  // verdict.refuted[] → un-tick refuted AC boxes + set eval_verdict frontmatter (old B.2/B.2b)
13
22
  // (the leg itself) → append one leg-completion row to .shapeup/<slug>/legs.jsonl
14
23
  //
@@ -31,7 +40,7 @@ import { tasksDir, localRoot, dispatchReceipts, legLedger, readRunId } from "../
31
40
  import { citationProblem } from "../probe/eval.mjs";
32
41
 
33
42
  const HERE = dirname(fileURLToPath(import.meta.url));
34
- const RESULT_SCHEMA = JSON.parse(readFileSync(resolve(HERE, "../../skills/tech-lead/schemas/work-result.schema.json"), "utf8"));
43
+ const RESULT_SCHEMA = JSON.parse(readFileSync(resolve(HERE, "../schemas/work-result.schema.json"), "utf8"));
35
44
 
36
45
  /**
37
46
  * @returns {string} Today's date as an ISO `YYYY-MM-DD` string (UTC), for log/frontmatter stamps.
@@ -193,7 +202,7 @@ export function updateBoardRow(indexBody, taskId, done) {
193
202
  * verdict{criteria[],refuted[]}).
194
203
  * @param {{cwd:string}} opts - cwd: working-directory root every LOCAL path resolves against.
195
204
  * @returns {{slug:string, tasks_updated:string[], acs_ticked:number, unblocked:string[],
196
- * discoveries_appended:number, refuted_unticked:number, verdict_lines:number}} A summary of every write performed.
205
+ * discoveries_appended:number, escalations_appended:number, refuted_unticked:number, verdict_lines:number}} A summary of every write performed.
197
206
  * @throws {Error} If a task/board/ledger file it must write is not writable (fs error propagates).
198
207
  * `evaluation/.verdicts-*.jsonl` under `.shapeup/<slug>/`.
199
208
  */
@@ -211,7 +220,7 @@ export function applyResult(result, { cwd }) {
211
220
  */
212
221
  function applyResultLocked(result, { cwd, slug }) {
213
222
  const local = localRoot(cwd, slug);
214
- const summary = { slug, tasks_updated: [], acs_ticked: 0, unblocked: [], discoveries_appended: 0, refuted_unticked: 0, verdict_lines: 0 };
223
+ const summary = { slug, tasks_updated: [], acs_ticked: 0, unblocked: [], discoveries_appended: 0, escalations_appended: 0, refuted_unticked: 0, verdict_lines: 0 };
215
224
 
216
225
  // 1. Task results → task files + board (old task-executor P3.1/P3.2/P3.6).
217
226
  const boardIndex = join(local, "tasks", "_index.md");
@@ -271,36 +280,81 @@ function applyResultLocked(result, { cwd, slug }) {
271
280
  }
272
281
  }
273
282
 
274
- // 3. Discoveries → the ledger (old P3.7 / QA H.3). Single writer: this script.
275
- if (result.discoveries?.length) {
283
+ // 3. Discoveries → the ledger (old P3.7 / QA H.3), plus an ESCALATE's deviations, tagged the same
284
+ // way and landed under the same heading. Single writer: this script. A worker's protocol names
285
+ // `deviations[]` as the only channel a blocked worker has — there is no `escalates[]` field —
286
+ // and routing it into THIS ledger is what makes it reach a reader that already exists: the
287
+ // ship report's own "Discovered, not built" section, and the census a scope's own worker reads
288
+ // before proposing a cut, both read this file's unresolved `+`/`~` entries.
289
+ const escalations = result.status === "escalated" ? (result.deviations || []) : [];
290
+ if (result.discoveries?.length || escalations.length) {
276
291
  const ledgerDir = join(local, "discovery");
277
292
  mkdirSync(ledgerDir, { recursive: true });
278
293
  const ledger = join(ledgerDir, "ledger.md");
279
294
  if (!existsSync(ledger)) writeFileSync(ledger, `---\nfeature: ${slug}\n---\n# Discovery Ledger — ${slug}\n`);
280
- const lines = result.discoveries.map((d) => {
281
- const tags = [d.lens ? `[lens:${d.lens}]` : "", d.severity_hint ? `severity-hint: ${d.severity_hint}` : "", d.test_gap ? `test-gap: ${d.test_gap}` : "", d.contradicts ? `contradicts: ${d.contradicts}` : "", d.traces_to?.length ? `traces_to: ${d.traces_to.join(", ")}` : ""].filter(Boolean);
282
- return `${d.marker} ${d.lens ? tags[0] + " " : ""}${d.line}${d.repro ? `\n repro: ${d.repro}` : ""}${tags.slice(d.lens ? 1 : 0).map((t) => `\n ${t}`).join("")}`;
283
- }).join("\n");
284
- appendFileSync(ledger, `\n## Discovered — ${result.order_id} (${today()})\n${lines}\n`);
285
- summary.discoveries_appended = result.discoveries.length;
295
+ // IDEMPOTENT ON THE ORDER, not merely append-only. `reduce ingest` is the single writer, but
296
+ // nothing stops the SAME order/result pair from being applied twice — a replayed ingest over an
297
+ // already-applied result, never a fresh attempt (a real re-attempt earns its own order_id,
298
+ // `r<N>-a<N+1>`). Measured: replaying one identical escalated WorkResult doubled its
299
+ // `[ESCALATE]` line in this ledger, and both GATE H's census and the ship report's "Discovered,
300
+ // not built" section count this file's open entries — so a replay silently inflated the count
301
+ // for a WorkResult that ran exactly once. A block heading names its order verbatim, so a second
302
+ // ingest of the same order recognises its own prior write and skips the append rather than
303
+ // duplicating it.
304
+ const existingLedger = readFileSync(ledger, "utf8");
305
+ /**
306
+ * The ledger heading one order's block is filed under — the idempotency key: a second ingest
307
+ * of the SAME order recognises its own prior write by this string, verbatim.
308
+ * @param {string} oid - `result.order_id` this block belongs to.
309
+ * @returns {string} The heading prefix (open-ended — the date suffix varies, the order id does not).
310
+ */
311
+ const headingFor = (oid) => `## Discovered — ${oid} (`;
312
+ const alreadyLogged = existingLedger.includes(headingFor(result.order_id));
313
+ if (alreadyLogged) {
314
+ summary.discoveries_appended = 0;
315
+ summary.escalations_appended = 0;
316
+ } else {
317
+ const discoveryLines = (result.discoveries || []).map((d) => {
318
+ const tags = [d.lens ? `[lens:${d.lens}]` : "", d.severity_hint ? `severity-hint: ${d.severity_hint}` : "", d.test_gap ? `test-gap: ${d.test_gap}` : "", d.contradicts ? `contradicts: ${d.contradicts}` : "", d.traces_to?.length ? `traces_to: ${d.traces_to.join(", ")}` : ""].filter(Boolean);
319
+ return `${d.marker} ${d.lens ? tags[0] + " " : ""}${d.line}${d.repro ? `\n repro: ${d.repro}` : ""}${tags.slice(d.lens ? 1 : 0).map((t) => `\n ${t}`).join("")}`;
320
+ });
321
+ // `+` (candidate work), matching the schema's own reading of that marker — an ESCALATE is
322
+ // exactly that: work a worker could not safely do without a decision only the census can make.
323
+ const escalateLines = escalations.map((d) => `+ [ESCALATE] ${d}`);
324
+ const lines = [...discoveryLines, ...escalateLines].join("\n");
325
+ appendFileSync(ledger, `\n${headingFor(result.order_id)}${today()})\n${lines}\n`);
326
+ summary.discoveries_appended = (result.discoveries || []).length;
327
+ summary.escalations_appended = escalations.length;
328
+ }
286
329
  }
287
330
 
288
331
  // 4. Verdict bookkeeping (old evaluator B.0/B.2/B.2b) — judge returns data, ingest writes.
332
+ //
333
+ // THE PROJECTION KEEPS THE TWO KEYS A LATER READER CANNOT RE-DERIVE. Measured on the first real
334
+ // verdict of a spined run: 85 of 97 criteria carried `traces_to` in the WorkResult and 0 of 97
335
+ // survived into this file, so the anchor from a criterion back to the requirement it grades was
336
+ // written by the judge and discarded one step later. And the row carried only the monotonic `run`
337
+ // counter, which restarts per ledger file and repeats across runs of one slug — `run_id` is the
338
+ // only key that separates them, so a projection over this file silently mixed two runs without
339
+ // it. A WorkResult carries no run key; it is read off the run's own receipt, the same derivation
340
+ // every other writer here uses.
289
341
  if (result.verdict) {
290
342
  const evalDir = join(local, "evaluation");
291
343
  mkdirSync(evalDir, { recursive: true });
292
344
  if (result.verdict.criteria?.length) {
293
345
  const target = result.order_id.split("/")[1] || "run";
294
346
  const ledger = join(evalDir, `.verdicts-${target}.jsonl`);
347
+ const runId = readRunId(cwd, slug);
295
348
  let run = 1;
296
349
  if (existsSync(ledger)) {
297
350
  const prior = readFileSync(ledger, "utf8").trim().split(/\n/).filter(Boolean).map((l) => { try { return JSON.parse(l); } catch { return null; } }).filter(Boolean);
298
351
  run = prior.reduce((mx, r) => Math.max(mx, r.run || 0), 0) + 1;
299
352
  }
300
353
  const lines = result.verdict.criteria.map((c) => JSON.stringify({
301
- run, dimension: c.dimension || "spec-conformance", criterion: c.criterion,
354
+ run, run_id: runId, dimension: c.dimension || "spec-conformance", criterion: c.criterion,
302
355
  verdict: c.verdict, confidence: c.confidence, reprobed: !!c.reprobed,
303
- evidence: c.evidence || "", at: new Date().toISOString(),
356
+ evidence: c.evidence || "", traces_to: Array.isArray(c.traces_to) ? c.traces_to : [],
357
+ at: new Date().toISOString(),
304
358
  })).join("\n");
305
359
  appendFileSync(ledger, lines + "\n");
306
360
  summary.verdict_lines = result.verdict.criteria.length;
@@ -644,5 +698,5 @@ export async function cli(rawArgv) {
644
698
  }
645
699
  }
646
700
 
647
- console.log(`✅ ingested ${result.order_id} — tasks: [${s.tasks_updated.join(", ")}] · ACs ticked: ${s.acs_ticked} · unblocked: [${s.unblocked.join(", ")}] · discoveries: ${s.discoveries_appended} · verdict lines: ${s.verdict_lines} · refuted un-ticked: ${s.refuted_unticked}`);
701
+ console.log(`✅ ingested ${result.order_id} — tasks: [${s.tasks_updated.join(", ")}] · ACs ticked: ${s.acs_ticked} · unblocked: [${s.unblocked.join(", ")}] · discoveries: ${s.discoveries_appended} · escalations: ${s.escalations_appended} · verdict lines: ${s.verdict_lines} · refuted un-ticked: ${s.refuted_unticked}`);
648
702
  }