shapeup-sdlc 3.5.0 → 3.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -59,7 +59,7 @@
59
59
  import { existsSync, readdirSync, readFileSync, writeFileSync, mkdirSync } from "node:fs";
60
60
  import { dirname, join, resolve } from "node:path";
61
61
  import { runArgs } from "../lib/argv.mjs";
62
- import { splitFrontmatter } from "../lib/contract.mjs";
62
+ import { splitFrontmatter, uncoerce } from "../lib/contract.mjs";
63
63
  import { globToRegExp } from "../verify/spec.mjs";
64
64
  import {
65
65
  intake, harnessRun, wiringMap, projectProfile, scopesDir, resultsDir, ordersDir,
@@ -69,8 +69,17 @@ import { evalVerdict } from "./eval.mjs";
69
69
 
70
70
  /** The run-state values `references/protocol.md` (Part 4 — State) defines. A typo'd status is a rejection,
71
71
  * not a write — the whole point of this file is that a write nobody validates is a write nobody
72
- * can trust. */
73
- export const RUN_STATUSES = ["orienting", "mapping", "building", "evaluating", "shipped", "escalated"];
72
+ * can trust. `aborted` is the terminal counterpart `escalated` already was: a gate
73
+ * resolving "abort", or a hard stop, ends the run the same way an operator-declared escalation
74
+ * does — neither resumes on relaunch — so both are TERMINAL_STATUSES below. */
75
+ export const RUN_STATUSES = ["orienting", "mapping", "building", "evaluating", "shipped", "escalated", "aborted"];
76
+
77
+ /**
78
+ * The statuses a run does not come back from. `closeRun` refuses every other member of
79
+ * {@link RUN_STATUSES} — `orienting`/`mapping`/`building`/`evaluating` are mid-flight, and writing
80
+ * a close over one of those would stamp `closed_at` on a run a relaunch is still meant to resume.
81
+ */
82
+ export const TERMINAL_STATUSES = ["shipped", "aborted", "escalated"];
74
83
 
75
84
  /** ORIENT's four artifacts (skills/orient/SKILL.md §Outputs): three by exact name, plus a spike
76
85
  * whose filename carries the area it spiked (`spike-<area>.md`, or `spike-not-needed.md` when
@@ -462,6 +471,166 @@ export function setRunStatus(cwd, slug, status) {
462
471
  return { ok: true, path: p, status };
463
472
  }
464
473
 
474
+ /**
475
+ * Rewrite `status:`, `closed_at:`, `closed_status:` and `close_cause:` together, in one pass,
476
+ * appending any of the last three that a pre-migration ledger carries no line for yet (the same
477
+ * tolerant-of-old-ledgers discipline `close_cause` itself shipped under).
478
+ *
479
+ * @param {string} body - The ledger's current text.
480
+ * @param {{status:string, closedAt:string, cause:(string|null)}} o - What to write. `cause` is
481
+ * already normalized prose (newlines collapsed, truncated) — this function only `uncoerce`s it.
482
+ * @returns {string} The rewritten text.
483
+ */
484
+ function writeCloseLines(body, { status, closedAt, cause }) {
485
+ const causeLine = `close_cause: ${uncoerce(cause || null)}`;
486
+ const closedStatusLine = `closed_status: ${status}`;
487
+ let out = body
488
+ .replace(/^status:.*$/m, `status: ${status}`)
489
+ .replace(/^closed_at:.*$/m, `closed_at: ${closedAt}`);
490
+ out = /^closed_status:.*$/m.test(out)
491
+ ? out.replace(/^closed_status:.*$/m, closedStatusLine)
492
+ // A ledger written before this field existed carries no line to replace — appended right after
493
+ // `closed_at:`, the one line every TERMINAL_STATUSES write also touches.
494
+ : out.replace(/^closed_at:.*$/m, (m) => `${m}\n${closedStatusLine}`);
495
+ out = /^close_cause:.*$/m.test(out)
496
+ ? out.replace(/^close_cause:.*$/m, causeLine)
497
+ : out.replace(/^closed_at:.*$/m, (m) => `${m}\n${causeLine}`);
498
+ return out;
499
+ }
500
+
501
+ /**
502
+ * Close the run: a terminal status, its cause, and a close timestamp, written together in ONE
503
+ * pass.
504
+ *
505
+ * Measured: after an EVAL worker escalated and the run aborted, `harness-run.md` still read
506
+ * `status: evaluating`, `closed_at: ~`, with no cause recorded anywhere — a live EVAL and a dead
507
+ * one were indistinguishable from the trace alone. Two writers made that possible: `setRunStatus`
508
+ * above replaces the `status:` line and NOTHING ELSE, and `closed_at` was written exactly once, as
509
+ * the literal `~`, by `init run` — nothing ever replaced it. This function is the one call site
510
+ * that closes a run, so a terminal RunReturn cannot leave one of the three facts behind.
511
+ *
512
+ * Refuses rather than silently no-ops, the same discipline as `setRunStatus`: an absent ledger, one
513
+ * missing the lines this writes, or a non-terminal `status` (closing a run still `building` would
514
+ * stamp a live run as done) is a fact to act on, not a write to skip quietly.
515
+ *
516
+ * THE ONCE-ONLY GUARD READS `closed_status`, NEVER `status`. REWORK (round 2): the guard used to key
517
+ * on `before.status`, and `status:` is a LIVE field every phase rewrites via `setRunStatus` —
518
+ * including the product's own ship path, which stamps `status: shipped`
519
+ * (`skills/tech-lead/workflows/shapeup-run.js`'s Ship phase) immediately before this call runs. So a
520
+ * run closed `aborted` at Preflight on one launch, relaunched, and carried through to a `shipped`
521
+ * RunReturn on a later one had its `status:` line rewritten to `shipped` by that ordinary phase
522
+ * traffic BEFORE `closeIfTerminal` ever called this function — the guard read `before.status ===
523
+ * "shipped"`, matched the very close it was about to perform, and treated a run that was actually
524
+ * closed `aborted` as already closed `shipped`: `ok:true`, an "idempotent no-op" that silently kept
525
+ * the abort's own `closed_at`/`close_cause` under a `status:` line now reading `shipped`. `closed_status`
526
+ * is written ONLY here, exactly once per distinct close-writing call, so nothing between two calls to
527
+ * this function can move it — it is the one field that actually answers "has this run been closed,
528
+ * and to what" regardless of how many times `status:` has been rewritten since.
529
+ *
530
+ * WHAT A SECOND CLOSE MEANS, decided explicitly rather than left implicit. `run_id` is reused across
531
+ * relaunches by design (AGENTS.md), so "the run's close" and "this launch's own close" are two
532
+ * different facts a single `closed_at`/`close_cause` pair cannot both hold:
533
+ * - The IDENTICAL status and the IDENTICAL cause is the ordinary case a retried or duplicated call
534
+ * produces (the same `withWarnings` call, or a relaunch that re-executes an already-applied
535
+ * close) — a true no-op, `ok:true`, nothing rewritten.
536
+ * - The SAME terminal status but a DIFFERENT cause is a SECOND, real close — most often a later
537
+ * relaunch aborting again for its own reason, or shipping again after an earlier ship's close
538
+ * record was never superseded. Discarding it (the pre-rework behavior) silently drops the later
539
+ * launch's own reason with no trace of the loss. It is recorded instead: this call's cause
540
+ * becomes the ledger's `close_cause`, folded together with the prior cause it is superseding —
541
+ * the earlier fact survives inside the new line rather than the ledger simply losing it — and the
542
+ * return carries `superseded:true` so a caller (`closeIfTerminal`) can flag the trace as degraded
543
+ * rather than reporting a clean success.
544
+ * - A DIFFERENT terminal status altogether (aborted vs. shipped) is refused outright — flipping the
545
+ * actual OUTCOME of a run after the fact is not a fact a later launch gets to silently overwrite,
546
+ * so the original `closed_status`/`closed_at`/`close_cause` are left completely untouched and
547
+ * handed back to the caller.
548
+ *
549
+ * @param {string} cwd - Project root.
550
+ * @param {string} slug - Feature slug.
551
+ * @param {{status:string, cause:(string|null)}} o - The terminal status (one of
552
+ * {@link TERMINAL_STATUSES}) and why the run ended there. `cause` travels through `uncoerce` (the
553
+ * one dialect `harness-run.md`'s frontmatter is read and written in), so free prose — quotes and
554
+ * colons included — round-trips as one frontmatter line; an embedded newline is collapsed to a
555
+ * space first, because this dialect is line-based and could not carry one either way.
556
+ * @returns {{ok:boolean, path:string, status:string, closed_at?:string, cause?:(string|null),
557
+ * reason?:string, closed_status?:string, superseded?:boolean, decision?:string,
558
+ * prior_cause?:(string|null), prior_closed_at?:string}} Outcome. A refused overwrite (already
559
+ * closed with a DIFFERENT terminal status) carries `closed_status`/`closed_at`/`cause` naming what
560
+ * is actually on disk. A successful supersede (same status, different cause) carries
561
+ * `superseded:true`, `decision:"superseded"` (the one-token signal the courier boundary in
562
+ * `shapeup-run.js` relays verbatim — see its own `cmd()` banner) and the prior close it folded in.
563
+ */
564
+ export function closeRun(cwd, slug, { status, cause = null } = {}) {
565
+ const p = harnessRun(cwd, slug);
566
+ if (!TERMINAL_STATUSES.includes(status)) {
567
+ return { ok: false, path: p, status, reason: `closeRun: "${status}" is not terminal — expected one of ${TERMINAL_STATUSES.join(" | ")}` };
568
+ }
569
+ if (!existsSync(p)) {
570
+ return { ok: false, path: p, status, reason: `no harness-run.md for slug "${slug}" — open the run with harness init run (GATE L0.1) before closing it` };
571
+ }
572
+ let body = readFileSync(p, "utf8");
573
+ if (!/^status:.*$/m.test(body) || !/^closed_at:.*$/m.test(body)) {
574
+ return { ok: false, path: p, status, reason: `harness-run.md carries no "status:"/"closed_at:" line to replace — the ledger's frontmatter is malformed (references/protocol.md)` };
575
+ }
576
+
577
+ // Truncated, not elided: a cause this long has already done its job in the run's own log — the
578
+ // ledger line is a pointer back to it, not the full transcript. Newlines are collapsed to spaces
579
+ // FIRST — this dialect is line-based, so a raw embedded newline would split one field into a value
580
+ // line and a stray, unparsed one.
581
+ const normCause = String(cause ?? "").replace(/\r?\n/g, " ").trim().slice(0, 4000) || null;
582
+
583
+ const before = parseFrontmatter(body);
584
+ const priorClosedStatus = before.closed_status && before.closed_status !== "~" ? before.closed_status : null;
585
+ const priorClosedAt = before.closed_at && before.closed_at !== "~" ? before.closed_at : null;
586
+ const priorCause = before.close_cause && before.close_cause !== "~" ? before.close_cause : null;
587
+
588
+ if (priorClosedStatus && priorClosedAt) {
589
+ if (priorClosedStatus === status && normCause === priorCause) {
590
+ // The identical fact, restated — a retried or duplicated call costs nothing.
591
+ return { ok: true, path: p, status, closed_at: priorClosedAt, cause: priorCause, decision: "idempotent", reason: `already closed as "${status}" at ${priorClosedAt} — idempotent no-op` };
592
+ }
593
+ if (priorClosedStatus !== status) {
594
+ // A DIFFERENT terminal status over an already-closed run — refused outright, the original
595
+ // close left completely untouched so the caller can see what it was refused permission to
596
+ // destroy, rather than losing it silently.
597
+ return {
598
+ ok: false, path: p, status,
599
+ reason: `closeRun: this run is already closed as "${priorClosedStatus}" at ${priorClosedAt} (cause: ${JSON.stringify(priorCause)}) — refusing to overwrite it with "${status}". A terminal close is a once-only fact; the first cause is not destroyed.`,
600
+ closed_status: priorClosedStatus, closed_at: priorClosedAt, cause: priorCause,
601
+ };
602
+ }
603
+ // SAME terminal status, a DIFFERENT cause — a second, real close (see the function banner's
604
+ // "what a second close means"). Superseded, not discarded: the prior cause is folded into the
605
+ // new line rather than lost, and the return says so explicitly.
606
+ const closedAt = new Date().toISOString();
607
+ const foldedCause = `${normCause || "no reason recorded"} — supersedes an earlier close recorded ${priorClosedAt} (cause: ${JSON.stringify(priorCause)})`.slice(0, 4000);
608
+ body = writeCloseLines(body, { status, closedAt, cause: foldedCause });
609
+ try { writeFileSync(p, body); } catch (e) {
610
+ return { ok: false, path: p, status, reason: `could not write the ledger: ${e.message}` };
611
+ }
612
+ const afterSup = parseFrontmatter(readFileSync(p, "utf8"));
613
+ if (afterSup.status !== status || !afterSup.closed_at || afterSup.closed_at === "~") {
614
+ return { ok: false, path: p, status, reason: `wrote the superseding close but the ledger reads back status="${afterSup.status}" closed_at="${afterSup.closed_at}" — the write did not take` };
615
+ }
616
+ return {
617
+ ok: true, path: p, status, closed_at: afterSup.closed_at, cause: afterSup.close_cause ?? null,
618
+ superseded: true, decision: "superseded", prior_cause: priorCause, prior_closed_at: priorClosedAt,
619
+ };
620
+ }
621
+
622
+ const closedAt = new Date().toISOString();
623
+ body = writeCloseLines(body, { status, closedAt, cause: normCause });
624
+ try { writeFileSync(p, body); } catch (e) {
625
+ return { ok: false, path: p, status, reason: `could not write the ledger: ${e.message}` };
626
+ }
627
+ const after = parseFrontmatter(readFileSync(p, "utf8"));
628
+ if (after.status !== status || !after.closed_at || after.closed_at === "~") {
629
+ return { ok: false, path: p, status, reason: `wrote the close but the ledger reads back status="${after.status}" closed_at="${after.closed_at}" — the write did not take` };
630
+ }
631
+ return { ok: true, path: p, status, closed_at: after.closed_at, cause: after.close_cause ?? null, decision: "closed" };
632
+ }
633
+
465
634
  /**
466
635
  * Point the substrate pointer at the order about to be executed.
467
636
  *
@@ -488,13 +657,17 @@ export function writeActiveOrder(cwd, slug, orderPath) {
488
657
 
489
658
  /** The typed argv contract (see `./lib/argv.mjs`). */
490
659
  export const ARGV_SPEC = {
491
- usage: "harness.mjs probe resume --slug <slug> [--cwd <dir>] [--require <phase> | --set-status <status> | --set-active-order <path>]",
660
+ usage: "harness.mjs probe resume --slug <slug> [--cwd <dir>] " +
661
+ "[--require <phase> | --set-status <status> | --set-active-order <path> | --close <status> [--cause <text>]]",
492
662
  _: { arity: 0, max: 0, name: "(no positional operands)" },
493
663
  slug: { type: "str", required: true },
494
664
  cwd: { type: "path" },
495
665
  require: { type: "enum", values: PHASES },
496
666
  "set-status": { type: "enum", values: RUN_STATUSES },
497
667
  "set-active-order": { type: "str" },
668
+ // The one call site that stamps a terminal status, its cause and closed_at together.
669
+ close: { type: "enum", values: TERMINAL_STATUSES },
670
+ cause: { type: "str" },
498
671
  };
499
672
 
500
673
  /**
@@ -508,11 +681,15 @@ export function cli(rawArgv) {
508
681
  const args = runArgs(ARGV_SPEC, rawArgv);
509
682
  const cwd = args.cwd || process.cwd();
510
683
 
511
- const ops = [args.require && "--require", args.setStatus && "--set-status", args.setActiveOrder && "--set-active-order"].filter(Boolean);
684
+ const ops = [args.require && "--require", args.setStatus && "--set-status", args.setActiveOrder && "--set-active-order", args.close && "--close"].filter(Boolean);
512
685
  if (ops.length > 1) {
513
686
  process.stderr.write(JSON.stringify({ error: "conflicting_flags", flags: ops, expected: "one operation per invocation" }) + "\n");
514
687
  process.exit(2);
515
688
  }
689
+ if (args.cause !== undefined && !args.close) {
690
+ process.stderr.write(JSON.stringify({ error: "conflicting_flags", flags: ["--cause"], expected: "--cause is only meaningful with --close" }) + "\n");
691
+ process.exit(2);
692
+ }
516
693
 
517
694
  // The post-condition. It prints the SAME ResumeState the derivation prints — plus which phase was
518
695
  // asked about and whether its artifact is there — so a caller that wants to act on the facts
@@ -541,6 +718,12 @@ export function cli(rawArgv) {
541
718
  process.exit(r.ok ? 0 : 3);
542
719
  }
543
720
 
721
+ if (args.close) {
722
+ const r = closeRun(cwd, args.slug, { status: args.close, cause: args.cause ?? null });
723
+ console.log(JSON.stringify(r));
724
+ process.exit(r.ok ? 0 : 3);
725
+ }
726
+
544
727
  console.log(JSON.stringify(deriveResumeState(cwd, args.slug)));
545
728
  process.exit(0);
546
729
  }
@@ -0,0 +1,104 @@
1
+ // rounds — how many BUILD/EVAL rounds a run actually reached, derived from disk.
2
+ //
3
+ // WHY THIS EXISTS (measured, not theorized).
4
+ //
5
+ // `harness-run.md`'s `rounds_used` frontmatter is written ONCE, at GATE L0.1 (`init run`), as `0`,
6
+ // and nothing in the round loop ever rewrites it — the orchestrator's own RunReturn carries the
7
+ // real count, but that value lives in a JS variable a relaunch loses, and it never reached the
8
+ // ledger. `reduce ship`'s own derivation partly compensated by counting the highest
9
+ // `evaluate-r<N>.json` result on disk, which is right for "rounds the judge saw" and wrong for
10
+ // "rounds the run built": measured after two full BUILD rounds with neither reaching EVAL,
11
+ // `round_count: 0` and `rounds_used: 0` both — a run that did real work reported having done none.
12
+ //
13
+ // TWO NUMBERS, NOT ONE. A round can be built and die before EVAL ever sees it (a circuit breaker,
14
+ // a kill mid-round), so "rounds built" and "rounds judged" are different facts and collapsing them
15
+ // into a single field is exactly what made the fallback silently wrong. Both are returned here, and
16
+ // both are meant to survive to wherever a run's numbers are read — the ship report and the export.
17
+ //
18
+ // PURE. Reads the run's own trace, writes nothing — the same discipline as `reduce ship`'s other
19
+ // derivations (`t0Summary`, `boardCensus`, …). A run before any of these artifacts existed, or a
20
+ // lane with no round concept at all (`--tiny`), falls back to the caller-supplied ledger value,
21
+ // non-regression with the earlier, EVAL-only behaviour.
22
+
23
+ import { existsSync, readdirSync, readFileSync } from "node:fs";
24
+ import { join } from "node:path";
25
+ import { ordersDir, verdictsDir, roundBuildDir, resultsDir } from "../lib/paths.mjs";
26
+
27
+ /** Parse a JSON file, returning null rather than throwing — every reader here is best-effort. */
28
+ function readJson(p) {
29
+ try { return JSON.parse(readFileSync(p, "utf8")); } catch { return null; }
30
+ }
31
+
32
+ /**
33
+ * The round a build-addressed order id encodes, e.g. `checkout/sc-01-r2-a1` → 2.
34
+ * @param {*} orderId - The order's own id, whatever shape it happens to be.
35
+ * @returns {(number|null)} The round, or null when the id carries none (`orient`, `wire`, `hammer`, …).
36
+ */
37
+ export function orderRound(orderId) {
38
+ const suffix = String(orderId ?? "").split("/").slice(1).join("/");
39
+ const m = suffix.match(/-r(\d+)(?:-a\d+)?$/);
40
+ return m ? Number(m[1]) : null;
41
+ }
42
+
43
+ const maxOf = (nums) => (nums.length ? Math.max(...nums) : null);
44
+
45
+ /**
46
+ * How many rounds this run actually reached — built, and separately, judged.
47
+ *
48
+ * `rounds_used` is the highest round carrying ANY build evidence: a compiled order, a T0 verdict
49
+ * artifact, a round build-gate artifact, or an EVAL result — the brief's own list, plus EVAL results
50
+ * because a judged round is, by construction, a round that was also built. `rounds_judged` is the
51
+ * highest round EVAL actually returned a verdict for (`evaluate-r<N>.json` on disk), kept as its own
52
+ * field rather than folded into the only number a reader can see.
53
+ *
54
+ * @param {string} cwd - Project root.
55
+ * @param {string} slug - Feature slug.
56
+ * @param {*} [fallback] - What to report for `rounds_used` when nothing on disk is derivable — the
57
+ * caller's own ledger value, passed through untouched (some callers hand this on already coerced
58
+ * to a number, some as the raw frontmatter string; this function does not care which).
59
+ * @returns {{rounds_used:*, rounds_judged:(number|null)}} Both counts.
60
+ */
61
+ export function deriveRounds(cwd, slug, fallback) {
62
+ const orderRounds = [];
63
+ const oDir = ordersDir(cwd, slug);
64
+ if (existsSync(oDir)) {
65
+ for (const f of readdirSync(oDir)) {
66
+ if (!f.endsWith(".json")) continue;
67
+ const r = orderRound(readJson(join(oDir, f))?.order_id);
68
+ if (r !== null) orderRounds.push(r);
69
+ }
70
+ }
71
+
72
+ const verdictRounds = [];
73
+ const vDir = verdictsDir(cwd, slug);
74
+ if (existsSync(vDir)) {
75
+ for (const f of readdirSync(vDir).filter((x) => x.endsWith(".json"))) {
76
+ const v = readJson(join(vDir, f));
77
+ if (typeof v?.round === "number") verdictRounds.push(v.round);
78
+ }
79
+ }
80
+
81
+ const buildGateRounds = [];
82
+ const bDir = roundBuildDir(cwd, slug);
83
+ if (existsSync(bDir)) {
84
+ for (const f of readdirSync(bDir)) {
85
+ const m = f.match(/^r(\d+)-t\d+\.json$/);
86
+ if (m) buildGateRounds.push(Number(m[1]));
87
+ }
88
+ }
89
+
90
+ const evalRounds = [];
91
+ const rDir = resultsDir(cwd, slug);
92
+ if (existsSync(rDir)) {
93
+ for (const f of readdirSync(rDir)) {
94
+ const m = f.match(/^evaluate-r(\d+)\.json$/);
95
+ if (m) evalRounds.push(Number(m[1]));
96
+ }
97
+ }
98
+
99
+ const built = maxOf([...orderRounds, ...verdictRounds, ...buildGateRounds, ...evalRounds]);
100
+ return {
101
+ rounds_used: built !== null ? built : fallback,
102
+ rounds_judged: maxOf(evalRounds),
103
+ };
104
+ }
@@ -8,6 +8,13 @@
8
8
  // task_results[] → tick AC boxes, flip task frontmatter status, append Execution Log,
9
9
  // update tasks/_index.md row, propagate unblocks (old P3.1–P3.6)
10
10
  // discoveries[] → append to .shapeup/<slug>/discovery/ledger.md (old P3.7 / QA H.3)
11
+ // deviations[] (when → append to the SAME discovery ledger, tagged distinctly. A worker's
12
+ // status:"escalated") protocol names deviations[] as the only channel a blocked worker has —
13
+ // there is no escalates[] field — and until now nothing routed it anywhere:
14
+ // a worker that correctly stopped rather than guessed produced a number and
15
+ // silence. Landing here reaches a reader that already exists — the census
16
+ // reads this ledger's open entries, and the ship report's own "Discovered,
17
+ // not built" section is generated from it.
11
18
  // verdict.criteria[] → append evaluation/.verdicts-<target>.jsonl (old evaluator B.0), every
12
19
  // row keyed by run_id and carrying the judge's traces_to[] anchor back to
13
20
  // the requirement — see step 4 for why neither may be dropped here
@@ -33,7 +40,7 @@ import { tasksDir, localRoot, dispatchReceipts, legLedger, readRunId } from "../
33
40
  import { citationProblem } from "../probe/eval.mjs";
34
41
 
35
42
  const HERE = dirname(fileURLToPath(import.meta.url));
36
- const RESULT_SCHEMA = JSON.parse(readFileSync(resolve(HERE, "../../skills/tech-lead/schemas/work-result.schema.json"), "utf8"));
43
+ const RESULT_SCHEMA = JSON.parse(readFileSync(resolve(HERE, "../schemas/work-result.schema.json"), "utf8"));
37
44
 
38
45
  /**
39
46
  * @returns {string} Today's date as an ISO `YYYY-MM-DD` string (UTC), for log/frontmatter stamps.
@@ -195,7 +202,7 @@ export function updateBoardRow(indexBody, taskId, done) {
195
202
  * verdict{criteria[],refuted[]}).
196
203
  * @param {{cwd:string}} opts - cwd: working-directory root every LOCAL path resolves against.
197
204
  * @returns {{slug:string, tasks_updated:string[], acs_ticked:number, unblocked:string[],
198
- * discoveries_appended:number, refuted_unticked:number, verdict_lines:number}} A summary of every write performed.
205
+ * discoveries_appended:number, escalations_appended:number, refuted_unticked:number, verdict_lines:number}} A summary of every write performed.
199
206
  * @throws {Error} If a task/board/ledger file it must write is not writable (fs error propagates).
200
207
  * `evaluation/.verdicts-*.jsonl` under `.shapeup/<slug>/`.
201
208
  */
@@ -213,7 +220,7 @@ export function applyResult(result, { cwd }) {
213
220
  */
214
221
  function applyResultLocked(result, { cwd, slug }) {
215
222
  const local = localRoot(cwd, slug);
216
- const summary = { slug, tasks_updated: [], acs_ticked: 0, unblocked: [], discoveries_appended: 0, refuted_unticked: 0, verdict_lines: 0 };
223
+ const summary = { slug, tasks_updated: [], acs_ticked: 0, unblocked: [], discoveries_appended: 0, escalations_appended: 0, refuted_unticked: 0, verdict_lines: 0 };
217
224
 
218
225
  // 1. Task results → task files + board (old task-executor P3.1/P3.2/P3.6).
219
226
  const boardIndex = join(local, "tasks", "_index.md");
@@ -273,18 +280,52 @@ function applyResultLocked(result, { cwd, slug }) {
273
280
  }
274
281
  }
275
282
 
276
- // 3. Discoveries → the ledger (old P3.7 / QA H.3). Single writer: this script.
277
- if (result.discoveries?.length) {
283
+ // 3. Discoveries → the ledger (old P3.7 / QA H.3), plus an ESCALATE's deviations, tagged the same
284
+ // way and landed under the same heading. Single writer: this script. A worker's protocol names
285
+ // `deviations[]` as the only channel a blocked worker has — there is no `escalates[]` field —
286
+ // and routing it into THIS ledger is what makes it reach a reader that already exists: the
287
+ // ship report's own "Discovered, not built" section, and the census a scope's own worker reads
288
+ // before proposing a cut, both read this file's unresolved `+`/`~` entries.
289
+ const escalations = result.status === "escalated" ? (result.deviations || []) : [];
290
+ if (result.discoveries?.length || escalations.length) {
278
291
  const ledgerDir = join(local, "discovery");
279
292
  mkdirSync(ledgerDir, { recursive: true });
280
293
  const ledger = join(ledgerDir, "ledger.md");
281
294
  if (!existsSync(ledger)) writeFileSync(ledger, `---\nfeature: ${slug}\n---\n# Discovery Ledger — ${slug}\n`);
282
- const lines = result.discoveries.map((d) => {
283
- const tags = [d.lens ? `[lens:${d.lens}]` : "", d.severity_hint ? `severity-hint: ${d.severity_hint}` : "", d.test_gap ? `test-gap: ${d.test_gap}` : "", d.contradicts ? `contradicts: ${d.contradicts}` : "", d.traces_to?.length ? `traces_to: ${d.traces_to.join(", ")}` : ""].filter(Boolean);
284
- return `${d.marker} ${d.lens ? tags[0] + " " : ""}${d.line}${d.repro ? `\n repro: ${d.repro}` : ""}${tags.slice(d.lens ? 1 : 0).map((t) => `\n ${t}`).join("")}`;
285
- }).join("\n");
286
- appendFileSync(ledger, `\n## Discovered — ${result.order_id} (${today()})\n${lines}\n`);
287
- summary.discoveries_appended = result.discoveries.length;
295
+ // IDEMPOTENT ON THE ORDER, not merely append-only. `reduce ingest` is the single writer, but
296
+ // nothing stops the SAME order/result pair from being applied twice — a replayed ingest over an
297
+ // already-applied result, never a fresh attempt (a real re-attempt earns its own order_id,
298
+ // `r<N>-a<N+1>`). Measured: replaying one identical escalated WorkResult doubled its
299
+ // `[ESCALATE]` line in this ledger, and both GATE H's census and the ship report's "Discovered,
300
+ // not built" section count this file's open entries — so a replay silently inflated the count
301
+ // for a WorkResult that ran exactly once. A block heading names its order verbatim, so a second
302
+ // ingest of the same order recognises its own prior write and skips the append rather than
303
+ // duplicating it.
304
+ const existingLedger = readFileSync(ledger, "utf8");
305
+ /**
306
+ * The ledger heading one order's block is filed under — the idempotency key: a second ingest
307
+ * of the SAME order recognises its own prior write by this string, verbatim.
308
+ * @param {string} oid - `result.order_id` this block belongs to.
309
+ * @returns {string} The heading prefix (open-ended — the date suffix varies, the order id does not).
310
+ */
311
+ const headingFor = (oid) => `## Discovered — ${oid} (`;
312
+ const alreadyLogged = existingLedger.includes(headingFor(result.order_id));
313
+ if (alreadyLogged) {
314
+ summary.discoveries_appended = 0;
315
+ summary.escalations_appended = 0;
316
+ } else {
317
+ const discoveryLines = (result.discoveries || []).map((d) => {
318
+ const tags = [d.lens ? `[lens:${d.lens}]` : "", d.severity_hint ? `severity-hint: ${d.severity_hint}` : "", d.test_gap ? `test-gap: ${d.test_gap}` : "", d.contradicts ? `contradicts: ${d.contradicts}` : "", d.traces_to?.length ? `traces_to: ${d.traces_to.join(", ")}` : ""].filter(Boolean);
319
+ return `${d.marker} ${d.lens ? tags[0] + " " : ""}${d.line}${d.repro ? `\n repro: ${d.repro}` : ""}${tags.slice(d.lens ? 1 : 0).map((t) => `\n ${t}`).join("")}`;
320
+ });
321
+ // `+` (candidate work), matching the schema's own reading of that marker — an ESCALATE is
322
+ // exactly that: work a worker could not safely do without a decision only the census can make.
323
+ const escalateLines = escalations.map((d) => `+ [ESCALATE] ${d}`);
324
+ const lines = [...discoveryLines, ...escalateLines].join("\n");
325
+ appendFileSync(ledger, `\n${headingFor(result.order_id)}${today()})\n${lines}\n`);
326
+ summary.discoveries_appended = (result.discoveries || []).length;
327
+ summary.escalations_appended = escalations.length;
328
+ }
288
329
  }
289
330
 
290
331
  // 4. Verdict bookkeeping (old evaluator B.0/B.2/B.2b) — judge returns data, ingest writes.
@@ -657,5 +698,5 @@ export async function cli(rawArgv) {
657
698
  }
658
699
  }
659
700
 
660
- console.log(`✅ ingested ${result.order_id} — tasks: [${s.tasks_updated.join(", ")}] · ACs ticked: ${s.acs_ticked} · unblocked: [${s.unblocked.join(", ")}] · discoveries: ${s.discoveries_appended} · verdict lines: ${s.verdict_lines} · refuted un-ticked: ${s.refuted_unticked}`);
701
+ console.log(`✅ ingested ${result.order_id} — tasks: [${s.tasks_updated.join(", ")}] · ACs ticked: ${s.acs_ticked} · unblocked: [${s.unblocked.join(", ")}] · discoveries: ${s.discoveries_appended} · escalations: ${s.escalations_appended} · verdict lines: ${s.verdict_lines} · refuted un-ticked: ${s.refuted_unticked}`);
661
702
  }
@@ -31,12 +31,13 @@ import { join, dirname } from "node:path";
31
31
  import { runArgs } from "../lib/argv.mjs";
32
32
  import {
33
33
  report as reportPath, tasksDir, verdictsDir, trials, evaluationDir, qaDir,
34
- roundLedger, discoveryLedger, receipt as receiptPath, harnessRun, relShared, resultsDir,
34
+ roundLedger, discoveryLedger, receipt as receiptPath, harnessRun, relShared,
35
35
  activeOrder,
36
36
  } from "../lib/paths.mjs";
37
37
  import { readTrials } from "../verify/t0.mjs";
38
38
  import { ratchetReport } from "../probe/stats.mjs";
39
39
  import { projectRequirements, summaryLine } from "../probe/requirements.mjs";
40
+ import { deriveRounds } from "../probe/rounds.mjs";
40
41
  import { collectDiff, scanDiff, summarize } from "./leftovers.mjs";
41
42
 
42
43
  /** @returns {string} Today as `YYYY-MM-DD` (UTC). */
@@ -167,13 +168,13 @@ export function section(md, heading) {
167
168
  */
168
169
  export function buildReport(facts) {
169
170
  const {
170
- slug, at, verdict, qa, rounds, board, t0, artifacts, ratchet,
171
+ slug, at, verdict, qa, rounds, roundsJudged, board, t0, artifacts, ratchet,
171
172
  evalCriteria, evalBugs, qaFindings, decisions, discovered, intakeSha, leftovers, requirements,
172
173
  } = facts;
173
174
 
174
175
  const L = [];
175
176
  L.push("---", "type: ship-report", `feature: ${slug}`, `date: ${at}`,
176
- `verdict: ${verdict}`, `rounds_used: ${rounds ?? "~"}`, `qa: ${qa}`,
177
+ `verdict: ${verdict}`, `rounds_used: ${rounds ?? "~"}`, `rounds_judged: ${roundsJudged ?? "~"}`, `qa: ${qa}`,
177
178
  `intake_sha256: ${intakeSha ?? "~"}`, "---", "");
178
179
  L.push(`# ${slug} — ship report`, "");
179
180
  L.push("Frozen at GATE L4. Every figure below is derived from run artifacts on disk — the trial",
@@ -183,6 +184,10 @@ export function buildReport(facts) {
183
184
  L.push("| | |", "|---|---|");
184
185
  L.push(`| Verdict | **${verdict}** |`);
185
186
  L.push(`| Rounds used | ${rounds ?? "—"} |`);
187
+ // A round built and a round judged are different facts — a round can die before EVAL ever sees
188
+ // it, so this row is its own line rather than folded into "Rounds used" above. Omitted when EVAL
189
+ // never ran at all, the same way the sections below it are.
190
+ if (roundsJudged != null) L.push(`| Rounds judged | ${roundsJudged} |`);
186
191
  L.push(`| Board | ${board.done}/${board.total} tasks done |`);
187
192
  L.push(`| T0 artifacts | ${artifacts} |`);
188
193
  L.push(`| QA | ${qa} |`);
@@ -291,32 +296,6 @@ export function buildReport(facts) {
291
296
  return L.join("\n");
292
297
  }
293
298
 
294
- /**
295
- * How many BUILD/EVAL rounds this run actually completed.
296
- *
297
- * `harness-run.md`'s `rounds_used` frontmatter field is written ONCE, at GATE L0.1 (`init run`),
298
- * as `0` — nothing in the round loop ever rewrites it as rounds complete. The orchestrator's own
299
- * `RunReturn` carries the real count (`shapeup-run.js`'s `rounds_used: round`), but that value
300
- * never reaches `reduce ship`, so every real run's report printed "Rounds used | 0" beside its own
301
- * `results/evaluate-r1.json` — a claim the frontmatter makes about the run, contradicted by the
302
- * artifact sitting next to it. Derived instead, the same way `probe resume`'s `eval_rounds_done`
303
- * already does: the highest `evaluate-r<N>.json` result on disk. Falls back to the frontmatter
304
- * value only when no EVAL round ever ran (the `--tiny` lane has no round concept at all), so a
305
- * bare or tiny run's reporting is unchanged.
306
- * @param {string} cwd - Project root.
307
- * @param {string} slug - Feature slug.
308
- * @param {(string|undefined)} fallback - `run.rounds_used` from the frontmatter.
309
- * @returns {(number|string|undefined)} The derived round count, or the fallback.
310
- */
311
- function roundsUsed(cwd, slug, fallback) {
312
- const dir = resultsDir(cwd, slug);
313
- const done = (existsSync(dir) ? readdirSync(dir) : [])
314
- .map((f) => f.match(/^evaluate-r(\d+)\.json$/))
315
- .filter(Boolean)
316
- .map((m) => Number(m[1]));
317
- return done.length ? Math.max(...done) : fallback;
318
- }
319
-
320
299
  /**
321
300
  * Gather every fact from disk and render the report.
322
301
  * @param {{cwd:string, slug:string, verdict?:string, qa?:string}} opts - Inputs.
@@ -331,12 +310,18 @@ export function generate({ cwd, slug, verdict, qa }) {
331
310
  const ledger = readIf(roundLedger(cwd, slug));
332
311
  const discovery = readIf(discoveryLedger(cwd, slug));
333
312
 
313
+ // Two numbers, not one: `rounds` (built — an order, a T0 verdict, a build-gate artifact or an
314
+ // EVAL result) and `roundsJudged` (EVAL actually returned a verdict for), kept separate so a run
315
+ // whose later rounds never reached EVAL still reports the rounds it built.
316
+ const derivedRounds = deriveRounds(cwd, slug, run.rounds_used);
317
+
334
318
  const facts = {
335
319
  slug,
336
320
  at: today(),
337
321
  verdict: verdict || run.final_verdict || "not-evaluated",
338
322
  qa: qa || (huntReport ? "run" : "skipped"),
339
- rounds: roundsUsed(cwd, slug, run.rounds_used),
323
+ rounds: derivedRounds.rounds_used,
324
+ roundsJudged: derivedRounds.rounds_judged,
340
325
  intakeSha: receipt.intake_sha256,
341
326
  board: boardCensus(cwd, slug),
342
327
  t0: t0Summary(cwd, slug),
@@ -25,6 +25,7 @@ import { resolve, join } from "node:path";
25
25
  import { validate } from "../verify/envelope.mjs";
26
26
  import { runArgs } from "../lib/argv.mjs";
27
27
  import { localDir, localRoot, relLocal, globLocal, runSnapshot as runSnapshotPath } from "../lib/paths.mjs";
28
+ import { deriveRounds } from "../probe/rounds.mjs";
28
29
 
29
30
  /**
30
31
  * Read a JSON file, tolerating absence/parse errors.
@@ -126,8 +127,28 @@ export function deriveSnapshot(cwd) {
126
127
  if (existsSync(runPath)) {
127
128
  try {
128
129
  const fm = frontmatter(readFileSync(runPath, "utf8"));
129
- if (MID_RUN.has(fm.status) || ["shipped", "escalated"].includes(fm.status)) snapshot.status = fm.status;
130
- if (/^\d+$/.test(fm.rounds_used || "")) snapshot.rounds_used = Number(fm.rounds_used);
130
+ // The terminal allowlist mirrors kernel/probe/resume.mjs's TERMINAL_STATUSES, not just the two
131
+ // members it used to carry — a schema/allowlist that disagrees with RUN_STATUSES is silent on
132
+ // both sides here (findRun only ever surfaces a MID_RUN run today, so this whole branch is
133
+ // unreachable in practice), but is exactly the divergence class this file's own imports exist
134
+ // to close elsewhere (deriveRounds, above). "aborted" is a real RUN_STATUSES member.
135
+ if (MID_RUN.has(fm.status) || ["shipped", "escalated", "aborted"].includes(fm.status)) snapshot.status = fm.status;
136
+ // MIGRATED to the same mechanical derivation `reduce ship` and `report export` already use
137
+ // (kernel/probe/rounds.mjs), rather than reading `harness-run.md`'s `rounds_used` literally.
138
+ // That frontmatter line is written ONCE, as 0, by `init run`, and nothing in the round loop
139
+ // ever rewrites it — so the third mechanical reader of "how many rounds did this run build"
140
+ // was the one still reporting the pre-fix number. Measured on a two-round, no-EVAL fixture:
141
+ // this line alone reported `rounds_used: 0` beside its own `round: 2` a few lines below,
142
+ // the exact disagreement `deriveRounds` exists to close. `fm.rounds_used` still travels in as
143
+ // the fallback for a run neither this fix nor deriveRounds can see evidence for (a `--tiny`
144
+ // lane, or a run from before any of these artifacts existed).
145
+ const derivedRounds = deriveRounds(cwd, run.slug, fm.rounds_used);
146
+ if (Number.isFinite(Number(derivedRounds.rounds_used))) snapshot.rounds_used = Number(derivedRounds.rounds_used);
147
+ // Kept as its OWN field, never folded into rounds_used — a round built is not a round judged,
148
+ // and collapsing the two into one number is exactly the ambiguity this migration removes.
149
+ // null (no EVAL result on disk yet) is a real answer and is not written at all, the same
150
+ // optional-field discipline every other snapshot field here follows.
151
+ if (Number.isFinite(derivedRounds.rounds_judged)) snapshot.rounds_judged = derivedRounds.rounds_judged;
131
152
  if (/^\d+$/.test(fm.max_rounds || "")) snapshot.max_rounds = Number(fm.max_rounds);
132
153
  if (fm.auto_level) snapshot.auto_level = fm.auto_level;
133
154
  if (fm.spec_folder) snapshot.spec_folder = fm.spec_folder;