shapeup-sdlc 3.7.0 → 3.7.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -33,6 +33,7 @@
33
33
 
34
34
  import { existsSync, readFileSync, readdirSync } from "node:fs";
35
35
  import { join, resolve } from "node:path";
36
+ import { createHash } from "node:crypto";
36
37
  import { runArgs } from "../lib/argv.mjs";
37
38
  import { resultsDir, scopesDir } from "../lib/paths.mjs";
38
39
 
@@ -59,6 +60,58 @@ export function isScoped(cwd, slug) {
59
60
  catch { return false; }
60
61
  }
61
62
 
63
+ /**
64
+ * Why one T0 citation does not resolve, or null when the kernel can positively confirm it does.
65
+ *
66
+ * Re-checks the two facts a citation can lie about and the schema promises are enforced
67
+ * (`T0Citation`: "the evaluator RECOMPUTES sha256 from disk — a handed hash is never trusted"):
68
+ * that the bytes at `path` hash to the cited `sha256`, and that what those bytes actually say is a
69
+ * green T0 verdict — a PASS/FAIL cannot ride on an artifact that itself recorded red. NOT checked:
70
+ * whether `path` is one of the order's own `payload.t0_artifacts` (see the note on
71
+ * {@link citationProblem} for why that is left to the operator rather than enforced here).
72
+ *
73
+ * FAILS OPEN ON "CANNOT TELL", CLOSED ON "PROVEN WRONG" — and the two are not the same fact. A read
74
+ * that fails for a reason that says something DEFINITE about what is (or isn't) at `path` is proof,
75
+ * same as a hash mismatch: `ENOENT` (nothing there) and `EISDIR` (a directory, never a file — T0
76
+ * verdicts are always files, per `writeArtifact`) both mean no such artifact was ever produced, so
77
+ * both refuse. `verify t0` writes verdict artifacts immutably and never deletes one
78
+ * (`writeArtifact`'s `wx` flag — see kernel/verify/t0.mjs), so that absence or shape mismatch is a
79
+ * positive fact about the citation, not a guess. Anything else a read can fail with — permission
80
+ * denied, a symlink loop, a transient I/O error — says nothing about the citation's honesty, only
81
+ * that THIS MACHINE could not check it just now, so it is not refused on that ground alone: the
82
+ * fail-open discipline this repo's guards already use elsewhere for an unproven bad state.
83
+ *
84
+ * @param {string} cwd - Project root.
85
+ * @param {object} citation - One `T0Citation` (`scope_id`, `path`, `sha256`). Read defensively:
86
+ * `reduce ingest` only reaches this after the enclosing WorkResult passed schema validation, but
87
+ * `probe eval` reaches it over a result file it merely `JSON.parse`s, so a malformed citation must
88
+ * fail this check rather than throw.
89
+ * @returns {(string|null)} A reason phrased for an operator, or null when the file at `path` exists,
90
+ * hashes to the cited `sha256`, and its own `overall` reads "green".
91
+ */
92
+ function unresolvedCitation(cwd, citation) {
93
+ const rel = typeof citation?.path === "string" ? citation.path : "";
94
+ if (!rel) return "names no artifact path";
95
+ let text;
96
+ try {
97
+ text = readFileSync(resolve(cwd, rel), "utf8");
98
+ } catch (e) {
99
+ if (e.code === "ENOENT") return `cites ${rel}, which does not exist on disk`;
100
+ if (e.code === "EISDIR") return `cites ${rel}, which is a directory, not a T0 verdict file`;
101
+ // EACCES, ELOOP, EIO, EMFILE… — this machine failing to look, not evidence against the
102
+ // citation, so it is not refused on that ground.
103
+ return null;
104
+ }
105
+ const actual = createHash("sha256").update(text).digest("hex");
106
+ const claimed = typeof citation.sha256 === "string" ? citation.sha256.toLowerCase() : "";
107
+ if (actual !== claimed) return `cites ${rel} with sha256 ${citation.sha256}, but the file on disk hashes to ${actual}`;
108
+ let body;
109
+ try { body = JSON.parse(text); }
110
+ catch { return `cites ${rel}, whose bytes match the hash but do not read as a T0 verdict`; }
111
+ if (body?.overall !== "green") return `cites ${rel}, whose own verdict is "${body?.overall ?? "unknown"}", not green`;
112
+ return null;
113
+ }
114
+
62
115
  /**
63
116
  * Why a verdict cannot stand as its round's judgement on T0 grounds, or null when it can.
64
117
  *
@@ -68,21 +121,40 @@ export function isScoped(cwd, slug) {
68
121
  * and branched on like any other. It is checked here so the round loop, the resume derivation, the
69
122
  * hill and ingest all refuse the same verdict for the same reason.
70
123
  *
71
- * PRESENCE, NOT HASHES. The evaluator re-hashes what it cites; a slip transcribing a digest is not
72
- * evidence the verdict is wrong, and refusing a round over one would cost a whole re-evaluation.
124
+ * RESOLVES, NOT JUST PRESENT. A citation naming a path and a hash used to be taken on faith: any
125
+ * non-empty `t0_citations[]` passed, whatever it pointed at. Measured live: a PASS citing a path
126
+ * that does not exist on disk, and a PASS citing a real artifact whose own verdict was red, both
127
+ * ingested clean — presence stood in for a re-hash the schema had already promised. Each citation is
128
+ * now re-checked through {@link unresolvedCitation}.
129
+ *
130
+ * WHAT THIS DOES NOT CHECK: whether a citation is drawn from the order's own `payload.t0_artifacts`
131
+ * list. That would need the compiled order, which this function's callers do not equally have —
132
+ * `reduce ingest` holds it, but `probe eval` (and the round-loop/resume/hill readers behind it)
133
+ * knows only (cwd, slug, round), and a scope's T0 attempt can legitimately go green again LATER than
134
+ * whatever list was frozen at compile time (`greenVerdict` already treats "newest green" as
135
+ * authoritative for exactly this reason — see kernel/probe/t0.mjs). Enforcing membership only where
136
+ * the order happens to be on hand would let one channel refuse a citation the other accepts, for
137
+ * evidence that may simply be fresher than the order — worse than leaving it unenforced.
73
138
  *
74
139
  * @param {string} cwd - Project root.
75
140
  * @param {string} slug - Feature slug.
76
141
  * @param {object} verdict - The WorkResult's `verdict` block.
77
- * @returns {(string|null)} The problem, phrased for an operator; null for a cited verdict, an
78
- * unscoped spec, or a block with no PASS/FAIL in it (there is no judgement to invalidate).
142
+ * @returns {(string|null)} The problem, phrased for an operator; null for a verdict whose every
143
+ * citation resolves, an unscoped spec, or a block with no PASS/FAIL in it (there is no judgement
144
+ * to invalidate).
79
145
  */
80
146
  export function citationProblem(cwd, slug, verdict) {
81
147
  if (verdict?.overall !== "PASS" && verdict?.overall !== "FAIL") return null;
82
- if (Array.isArray(verdict.t0_citations) && verdict.t0_citations.length) return null;
83
148
  if (!isScoped(cwd, slug)) return null;
84
- return `the ${verdict.overall} verdict cites no T0 artifact, and a verdict on a scoped spec must ` +
85
- "cite the T0 verdict it re-hashed (the order lists them under payload.t0_artifacts)";
149
+ if (!Array.isArray(verdict.t0_citations) || !verdict.t0_citations.length) {
150
+ return `the ${verdict.overall} verdict cites no T0 artifact, and a verdict on a scoped spec must ` +
151
+ "cite the T0 verdict it re-hashed (the order lists them under payload.t0_artifacts)";
152
+ }
153
+ for (const citation of verdict.t0_citations) {
154
+ const reason = unresolvedCitation(cwd, citation);
155
+ if (reason) return `the ${verdict.overall} verdict ${reason} — a T0 citation is re-hashed from disk, never taken on the handed word`;
156
+ }
157
+ return null;
86
158
  }
87
159
 
88
160
  /**
@@ -56,14 +56,14 @@
56
56
  // Exit: 0 ok · 2 malformed argv (nothing ran) · 3 the target the operation needs is not on disk ·
57
57
  // 6 the required phase's artifact is NOT on disk (the phase did not complete).
58
58
 
59
- import { existsSync, readdirSync, readFileSync, writeFileSync, mkdirSync } from "node:fs";
59
+ import { existsSync, readdirSync, readFileSync, writeFileSync, mkdirSync, rmSync } from "node:fs";
60
60
  import { dirname, join, resolve } from "node:path";
61
61
  import { runArgs } from "../lib/argv.mjs";
62
62
  import { splitFrontmatter, uncoerce } from "../lib/contract.mjs";
63
63
  import { globToRegExp } from "../verify/spec.mjs";
64
64
  import {
65
65
  intake, harnessRun, wiringMap, projectProfile, scopesDir, resultsDir, ordersDir,
66
- orientDir, activeOrder, usecasesDir, breadboard, receipt, readReceipt, requirements,
66
+ orientDir, activeOrder, activeScope, usecasesDir, breadboard, receipt, readReceipt, requirements,
67
67
  exportRunDir,
68
68
  } from "../lib/paths.mjs";
69
69
  import { evalVerdict } from "./eval.mjs";
@@ -679,10 +679,46 @@ export function closeRun(cwd, slug, { status, cause = null, withExport = true }
679
679
  // The Ship phase (`shapeup-run.js`) already exports a shipped run itself, before this call ever
680
680
  // runs — Stage 2 adds the endings that wrote nothing, and leaves that path untouched.
681
681
  const shouldExport = withExport && status !== "shipped";
682
- const withExportWarning = (result) => {
683
- if (!shouldExport) return result;
684
- const warning = exportOnClose(cwd, slug);
685
- return warning ? { ...result, export_warning: warning } : result;
682
+
683
+ /**
684
+ * Everything a close owes the checkout once the ledger line is written: export the run's
685
+ * records, then retire its pointers.
686
+ *
687
+ * The export runs FIRST, but not because it has to: `exportOnClose` is handed the slug and keys
688
+ * its output by the receipt's `run_id`, so it never reads the pointers this retires. (The bare
689
+ * `reduce export` CLI does read `active-scope` — only to work out which run the operator meant
690
+ * when they named none.) Reporting before teardown is a defensive default, not a correctness
691
+ * requirement, and it is recorded as such so the next reader does not defend an ordering that
692
+ * carries nothing. Measured: swapping the two leaves every check green.
693
+ *
694
+ * WHY RETIRE AT ALL. The substrate fence is enforced while an order is compiled and unanswered,
695
+ * and a run that ends any way other than shipping leaves exactly that by construction. Until this
696
+ * ran, the fence outlived the run: after a close that exited 0 and recorded everything, an
697
+ * ordinary write anywhere in the project was still denied, with no dispatch in flight. The
698
+ * operator's obvious remedy did not help either — `init run --force` is documented as "abandon
699
+ * the open run and start over", never as "release a stuck fence".
700
+ *
701
+ * AND RETIRING IS NOT ANSWERING. The pointer says "a run is in flight"; the order's missing
702
+ * result says "nobody came back". Only the first is untrue after a close. The abandoned order
703
+ * stays unanswered, the attempt census still sees nothing spent on it, and `init run --force`
704
+ * remains the one thing that writes a synthetic result — because a close that quietly claimed the
705
+ * work was answered would spend an attempt budget on work nobody did.
706
+ *
707
+ * Best-effort, like the export: a pointer that cannot be removed degrades the close and is
708
+ * reported in its return, but never turns a close into a non-close.
709
+ */
710
+ const finishClose = (result) => {
711
+ const warning = shouldExport ? exportOnClose(cwd, slug) : null;
712
+ const stuck = [];
713
+ for (const pointer of [activeOrder(cwd), activeScope(cwd)]) {
714
+ try { rmSync(pointer, { force: true }); } catch { /* fall through to the check below */ }
715
+ if (existsSync(pointer)) stuck.push(pointer);
716
+ }
717
+ return {
718
+ ...result,
719
+ ...(warning ? { export_warning: warning } : {}),
720
+ ...(stuck.length ? { pointer_warning: `could not retire ${stuck.join(", ")} — the substrate fence may still deny writes until it is removed by hand` } : {}),
721
+ };
686
722
  };
687
723
 
688
724
  // Truncated, not elided: a cause this long has already done its job in the run's own log — the
@@ -699,7 +735,7 @@ export function closeRun(cwd, slug, { status, cause = null, withExport = true }
699
735
  if (priorClosedStatus && priorClosedAt) {
700
736
  if (priorClosedStatus === status && normCause === priorCause) {
701
737
  // The identical fact, restated — a retried or duplicated call costs nothing.
702
- return withExportWarning({ ok: true, path: p, status, closed_at: priorClosedAt, cause: priorCause, decision: "idempotent", reason: `already closed as "${status}" at ${priorClosedAt} — idempotent no-op` });
738
+ return finishClose({ ok: true, path: p, status, closed_at: priorClosedAt, cause: priorCause, decision: "idempotent", reason: `already closed as "${status}" at ${priorClosedAt} — idempotent no-op` });
703
739
  }
704
740
  if (priorClosedStatus !== status) {
705
741
  // A DIFFERENT terminal status over an already-closed run — refused outright, the original
@@ -724,7 +760,7 @@ export function closeRun(cwd, slug, { status, cause = null, withExport = true }
724
760
  if (afterSup.status !== status || !afterSup.closed_at || afterSup.closed_at === "~") {
725
761
  return { ok: false, path: p, status, reason: `wrote the superseding close but the ledger reads back status="${afterSup.status}" closed_at="${afterSup.closed_at}" — the write did not take` };
726
762
  }
727
- return withExportWarning({
763
+ return finishClose({
728
764
  ok: true, path: p, status, closed_at: afterSup.closed_at, cause: afterSup.close_cause ?? null,
729
765
  superseded: true, decision: "superseded", prior_cause: priorCause, prior_closed_at: priorClosedAt,
730
766
  });
@@ -739,7 +775,7 @@ export function closeRun(cwd, slug, { status, cause = null, withExport = true }
739
775
  if (after.status !== status || !after.closed_at || after.closed_at === "~") {
740
776
  return { ok: false, path: p, status, reason: `wrote the close but the ledger reads back status="${after.status}" closed_at="${after.closed_at}" — the write did not take` };
741
777
  }
742
- return withExportWarning({ ok: true, path: p, status, closed_at: after.closed_at, cause: after.close_cause ?? null, decision: "closed" });
778
+ return finishClose({ ok: true, path: p, status, closed_at: after.closed_at, cause: after.close_cause ?? null, decision: "closed" });
743
779
  }
744
780
 
745
781
  /**
@@ -6,24 +6,158 @@
6
6
  import { readFileSync, writeFileSync, existsSync, readdirSync, mkdirSync } from "node:fs";
7
7
  import { resolve, join } from "node:path";
8
8
  import { runArgs } from "../lib/argv.mjs";
9
- import { scopesDir, hillDir, verdictsDir, resultsDir, discoveryLedger } from "../lib/paths.mjs";
9
+ import { scopesDir, hillDir, verdictsDir, resultsDir, discoveryLedger, localRoot } from "../lib/paths.mjs";
10
10
  import { readAllContracts, SCOPE_CONTRACT } from "../lib/contract.mjs";
11
11
  import { evalVerdict } from "../probe/eval.mjs";
12
12
  import { redBuildRounds } from "../verify/build.mjs";
13
13
 
14
+ // ---------------------------------------------------------------------------------------------
15
+ // THE DISCOVERY LEDGER, AND WHY ITS ABSENCE IS NOT A ZERO.
16
+ //
17
+ // The ledger arm is the only thing that promotes a scope from UPHILL_UNKNOWN to UPHILL_SOLVED —
18
+ // "the open questions are closed". It used to be read as `scopeUnknowns[id] || 0`, which made
19
+ // "nobody has filed a ledger yet" and "every unknown is closed" the same value, and that value
20
+ // selects the FLATTERING phase. An absent value and a real one must not share a signature; when
21
+ // they do, the run reports progress it has no evidence for. So the count is `null` — not `0` —
22
+ // whenever the ledger could not actually be read and understood, and `null` promotes nothing.
23
+ //
24
+ // THE HEADING IS THE ATTRIBUTION KEY, and it has to match what the ingest step really writes:
25
+ // `## Discovered — <order_id> (<date>)`, where `order_id` is `<slug>/<suffix>`. Two ways that
26
+ // parse used to fail silently, both ending in the same false zero:
27
+ // * It required a COLON between the slug and the suffix. The order-envelope schema pins
28
+ // `order_id` to a slug, a SLASH, and a suffix drawn from `[a-z0-9.-]` — a colon cannot appear
29
+ // in a schema-valid order id at all, so the scope was never captured from any real ledger and
30
+ // every scope on every project read zero unknowns regardless of what was open.
31
+ // * It matched the em dash only, so a heading typed with a plain hyphen contributed nothing.
32
+ // Both are fixed by matching the dash as a class and splitting the order id on its "/", and by
33
+ // resolving the scope against the CONTRACTS ON DISK rather than guessing at the suffix's shape: a
34
+ // build suffix is `<scope>-r<N>-a<M>` and a scoped operation's is `<operation>-<scope>[-r<N>]`, so
35
+ // a pattern that tried to carve the scope out positionally captured the round suffix along with it.
36
+ // ---------------------------------------------------------------------------------------------
37
+
38
+ /** A ledger block heading, naming the order it files: `## Discovered — <slug>/<suffix> (<date>)`. */
39
+ const LEDGER_HEADING = /^##\s+Discovered\s+[—–-]\s+(\S+)/;
40
+
41
+ /**
42
+ * Index the scopes by the filename-safe form `harness compile` puts in an order suffix, so a
43
+ * heading is matched against ids that actually exist rather than parsed speculatively.
44
+ *
45
+ * @param {Array<object>} scopes - Parsed scope contracts.
46
+ * @returns {Map<string,string>} Compile's suffix form → the contract's own `scope_id`.
47
+ */
48
+ function scopeSuffixIndex(scopes) {
49
+ const ix = new Map();
50
+ for (const s of scopes) {
51
+ const id = String(s?.scope_id || "");
52
+ if (!id) continue;
53
+ // Mirrors `harness compile`'s own normalisation of a scope id into an order suffix.
54
+ const key = id.toLowerCase().replace(/[^a-z0-9.-]/g, "-").replace(/^[^a-z0-9]+/, "");
55
+ if (key) ix.set(key, id);
56
+ }
57
+ return ix;
58
+ }
59
+
60
+ /**
61
+ * The scope a ledger heading's order belongs to, or `null` when it belongs to none.
62
+ *
63
+ * Operation-level dispatches (orient, analyze, wire, evaluate, hunt, hammer) carry no scope in
64
+ * their order id by construction, so "no scope" is a real and common answer here, not a parse
65
+ * failure — their rows are feature-wide and are deliberately credited to nobody.
66
+ *
67
+ * @param {string} orderId - The order id as the heading names it (`<slug>/<suffix>`).
68
+ * @param {Map<string,string>} ix - The index from `scopeSuffixIndex`.
69
+ * @returns {string|null} The owning contract's `scope_id`, or null.
70
+ */
71
+ function scopeOfHeading(orderId, ix) {
72
+ const slash = orderId.indexOf("/");
73
+ const suffix = slash === -1 ? orderId : orderId.slice(slash + 1);
74
+ const core = suffix.replace(/-r\d+(?:-a\d+)?$/, "");
75
+ // Longest match wins, so a project holding both `pages` and `pages-admin` attributes each
76
+ // heading to the scope it actually names rather than to whichever was indexed first.
77
+ let best = null;
78
+ let bestLen = -1;
79
+ for (const [key, id] of ix) {
80
+ if ((core === key || core.endsWith(`-${key}`)) && key.length > bestLen) { best = id; bestLen = key.length; }
81
+ }
82
+ return best;
83
+ }
84
+
85
+ /**
86
+ * Open unknowns (`~` rows) per scope, read from the discovery ledger.
87
+ *
88
+ * @param {string} ledgerPath - Path to the run's discovery ledger.
89
+ * @param {Array<object>} scopes - Parsed scope contracts, used to attribute each block.
90
+ * @returns {Object<string,number>|null} Counts per `scope_id` — a scope with a block and no open
91
+ * row is a genuine `0`. `null` means the ledger was NOT read: it is absent, unreadable, or
92
+ * nothing in it could be attributed to a scope. `null` is never treated as zero, because "we
93
+ * understood none of this file" is not evidence that nothing is open.
94
+ */
95
+ function ledgerUnknowns(ledgerPath, scopes) {
96
+ if (!existsSync(ledgerPath)) return null;
97
+ let text;
98
+ try { text = readFileSync(ledgerPath, "utf8"); } catch { return null; }
99
+
100
+ const ix = scopeSuffixIndex(scopes);
101
+ const counts = {};
102
+ let headings = 0;
103
+ let attributed = 0;
104
+ let current = null;
105
+
106
+ for (const line of text.split("\n")) {
107
+ if (line.startsWith("## ")) {
108
+ // Reset on EVERY heading, matched or not. Carrying the previous block's scope across an
109
+ // unrecognised heading would credit one scope's open rows to another.
110
+ current = null;
111
+ const m = line.match(LEDGER_HEADING);
112
+ if (!m) continue;
113
+ headings++;
114
+ const id = scopeOfHeading(m[1], ix);
115
+ if (id) { attributed++; current = id; counts[id] = counts[id] || 0; }
116
+ continue;
117
+ }
118
+ if (current && line.startsWith("~ ")) counts[current]++;
119
+ }
120
+
121
+ // A ledger whose blocks we could not attribute to a single scope tells us nothing per scope.
122
+ // Reporting that as all-zero is the same false signature the colon-vs-slash bug produced.
123
+ if (headings === 0 || attributed === 0) return null;
124
+ return counts;
125
+ }
126
+
127
+ /**
128
+ * The phase currently recorded in a committed hill shard.
129
+ *
130
+ * @param {string} hDir - The committed hill directory.
131
+ * @param {string} id - Scope id.
132
+ * @returns {string|null} The recorded phase, or null when no shard exists for that scope.
133
+ */
134
+ function committedPhase(hDir, id) {
135
+ const p = join(hDir, `${id}.yml`);
136
+ if (!existsSync(p)) return null;
137
+ try { return readFileSync(p, "utf8").match(/^phase:\s*(\S+)/m)?.[1] ?? null; } catch { return null; }
138
+ }
139
+
14
140
  /**
15
141
  * Derive and write the hill phase for all scopes mechanically based on T0, T1, and ledger facts.
16
142
  *
17
143
  * The derived phase follows these progression rules (facts move dots, not authors):
18
- * - UPHILL_UNKNOWN: open unknowns > 0 in the ledger for this scope
19
- * - UPHILL_SOLVED: unknowns 0, no T0-green yet
144
+ * - UPHILL_UNKNOWN: open unknowns > 0 in the ledger for this scope — and the floor the scope sits
145
+ * at whenever the ledger has not answered at all, which is where every run legitimately begins
146
+ * - UPHILL_SOLVED: the ledger was read and reports zero open unknowns, no T0-green yet
20
147
  * - DOWNHILL_EXECUTION: ≥1 T0-green in a round whose build gate is not red; T1/seesaw pending
21
148
  * - FINISHED: T1 PASS ∧ seesaw green
22
149
  *
23
150
  * @param {string} cwd - The project root directory.
24
151
  * @param {string} slug - The feature slug being built.
25
- * @returns {Array<{scope_id: string, phase: string, changed: boolean}>} A report of all scopes processed, their derived phase, and whether the hill shard on disk was modified.
26
- * Side effects: writes to `shapeup/<slug>/hill/<scope-id>.yml` for each scope.
152
+ * @returns {Array<{scope_id: string, phase: (string|null), changed: boolean, derived: boolean,
153
+ * unknowns: (number|null), reason?: string}>} A report of all scopes processed. `derived` says
154
+ * whether the phase was computed from run evidence at all: when it is false the phase is
155
+ * whatever the committed shard already records (or null when there is none), nothing was
156
+ * written, and `reason` names why. `unknowns` is the ledger count behind the phase, or null
157
+ * when the ledger could not be read — the two are deliberately distinguishable in the output as
158
+ * well as in the derivation.
159
+ * Side effects: writes to `shapeup/<slug>/hill/<scope-id>.yml` for each scope, EXCEPT on the
160
+ * refusal path below, which writes nothing at all.
27
161
  */
28
162
  export function deriveHill(cwd, slug) {
29
163
  const scopes = readAllContracts(scopesDir(cwd, slug), SCOPE_CONTRACT).map((x) => x.contract);
@@ -31,6 +165,44 @@ export function deriveHill(cwd, slug) {
31
165
  const ledgerPath = discoveryLedger(cwd, slug);
32
166
  const hDir = hillDir(cwd, slug);
33
167
 
168
+ // -------------------------------------------------------------------------------------------
169
+ // A DERIVATION THAT CANNOT SEE THE RUN TRACE MUST NOT WRITE THE DELIVERABLE.
170
+ //
171
+ // This function reads one tier and writes the other. Every input below — T0 verdicts, the EVAL
172
+ // results, the round build gates, the discovery ledger — lives in the gitignored run trace,
173
+ // while the shards it writes are committed and outlive it: after a ship the run trace is cleaned
174
+ // up and the shards are the ONLY surviving record of where the work got to. So when the run
175
+ // trace for this slug is not on disk, every input is provably absent, and anything derived from
176
+ // that is derived from nothing.
177
+ //
178
+ // This is not a hypothetical. Pulling a branch mid-run is a supported state: the puller gets the
179
+ // committed spec, scopes and shards and no run trace of their own. Their first launch used to
180
+ // re-derive every scope from the empty set and overwrite the shards with the result — losing
181
+ // committed history rather than misreporting it.
182
+ //
183
+ // WHY REFUSE RATHER THAN WRITE AN "UNKNOWN" PHASE. Writing anything here destroys the record
184
+ // just as thoroughly; a shard that says "I could not look" has still replaced the one that said
185
+ // FINISHED, and the phase enum is a committed data format that a reader parses as current
186
+ // status. Not writing already expresses "no opinion" exactly, and it needs no new enum value.
187
+ //
188
+ // WHY THE CONDITION IS THE TIER'S EXISTENCE AND NOT "the phase would go down". Moving a dot
189
+ // backwards is correct and must stay possible — this is a pure function of the artifacts present
190
+ // and reports what they currently support, in both directions. A guard phrased as "never lower a
191
+ // phase" would quietly turn a derived value into a high-water mark, which is a different defect
192
+ // wearing this one's clothes. The condition is the narrowest one that is positively provable:
193
+ // the tier holding every input is not there.
194
+ // -------------------------------------------------------------------------------------------
195
+ if (!existsSync(localRoot(cwd, slug))) {
196
+ return scopes.map((s) => ({
197
+ scope_id: s.scope_id,
198
+ phase: committedPhase(hDir, s.scope_id),
199
+ changed: false,
200
+ derived: false,
201
+ unknowns: null,
202
+ reason: "local-run-trace-absent",
203
+ }));
204
+ }
205
+
34
206
  if (!existsSync(hDir)) mkdirSync(hDir, { recursive: true });
35
207
 
36
208
  // 1. Check if T1 Evaluation passed — the LATEST evaluate round's verdict, read the same way
@@ -95,28 +267,20 @@ export function deriveHill(cwd, slug) {
95
267
  }
96
268
  }
97
269
 
98
- // 3. Ledger unknowns per scope
99
- const scopeUnknowns = {};
100
- if (existsSync(ledgerPath)) {
101
- const lines = readFileSync(ledgerPath, "utf8").split("\n");
102
- let currentScope = null;
103
- for (const line of lines) {
104
- const m = line.match(/^## Discovered — .*?:([\w.-]+)-a\d+/);
105
- if (m) {
106
- currentScope = m[1];
107
- }
108
- if (currentScope && line.startsWith("~ ")) {
109
- scopeUnknowns[currentScope] = (scopeUnknowns[currentScope] || 0) + 1;
110
- }
111
- }
112
- }
113
-
270
+ // 3. Ledger unknowns per scope — `null` for every scope when the ledger itself was not readable
271
+ // or nothing in it named a scope (see `ledgerUnknowns`). Only a real count can promote.
272
+ const scopeUnknowns = ledgerUnknowns(ledgerPath, scopes);
273
+
114
274
  const report = [];
115
275
  for (const s of scopes) {
116
276
  const id = s.scope_id;
117
277
  const t0 = t0Facts[id] || { hasGreen: false, seesawGreen: false };
118
- const unknowns = scopeUnknowns[id] || 0;
119
-
278
+ // `null` = the ledger did not answer; a number = it did. `|| 0` collapsed the two.
279
+ const unknowns = scopeUnknowns === null ? null : (scopeUnknowns[id] || 0);
280
+
281
+ // UPHILL_UNKNOWN is the floor, and the honest answer whenever nothing has promoted a scope off
282
+ // it — including before Orient has filed anything, which is where every run legitimately
283
+ // starts. Only an ANSWERED count of zero promotes to UPHILL_SOLVED; `null` never does.
120
284
  let phase = "UPHILL_UNKNOWN";
121
285
  if (t1Pass && t0.hasGreen && t0.seesawGreen) {
122
286
  phase = "FINISHED";
@@ -125,7 +289,7 @@ export function deriveHill(cwd, slug) {
125
289
  } else if (unknowns === 0) {
126
290
  phase = "UPHILL_SOLVED";
127
291
  }
128
-
292
+
129
293
  const yaml = `scope_id: ${id}\nphase: ${phase}\n`;
130
294
  const out = join(hDir, `${id}.yml`);
131
295
  let changed = false;
@@ -133,7 +297,7 @@ export function deriveHill(cwd, slug) {
133
297
  writeFileSync(out, yaml);
134
298
  changed = true;
135
299
  }
136
- report.push({ scope_id: id, phase, changed });
300
+ report.push({ scope_id: id, phase, changed, derived: true, unknowns });
137
301
  }
138
302
  return report;
139
303
  }
@@ -156,5 +320,16 @@ export async function cli(rawArgv) {
156
320
  const args = runArgs(ARGV_SPEC, rawArgv);
157
321
  const cwd = resolve(args.cwd || process.cwd());
158
322
  const report = deriveHill(cwd, args.slug);
323
+ // A refusal that is visible only as a missing write reads exactly like a derivation that
324
+ // happened to agree with what was already on disk, so say it out loud. It is not an error —
325
+ // the caller runs this advisorily several times a run, and declining to derive from nothing is
326
+ // the correct outcome, not a failure — so the exit code stays 0 and the warning goes to stderr.
327
+ const underived = report.filter((r) => r.derived === false);
328
+ if (underived.length) {
329
+ console.error(
330
+ `hill: derived nothing for ${underived.length} scope(s) (${underived[0].reason}) — ` +
331
+ `the run trace this phase is derived from is not on disk, so the committed shards were left as they are.`,
332
+ );
333
+ }
159
334
  console.log(JSON.stringify(report, null, 2));
160
335
  }
@@ -32,7 +32,7 @@ import { runArgs } from "../lib/argv.mjs";
32
32
  import {
33
33
  report as reportPath, tasksDir, verdictsDir, trials, evaluationDir, qaDir,
34
34
  roundLedger, discoveryLedger, receipt as receiptPath, harnessRun, relShared,
35
- activeOrder,
35
+ activeOrder, runArgsPath, readReceipt, runIdFromReceipt,
36
36
  } from "../lib/paths.mjs";
37
37
  import { readTrials } from "../verify/t0.mjs";
38
38
  import { ratchetReport } from "../probe/stats.mjs";
@@ -78,7 +78,7 @@ export function frontmatter(text) {
78
78
  */
79
79
  export function boardCensus(cwd, slug) {
80
80
  const dir = tasksDir(cwd, slug);
81
- const out = { total: 0, done: 0, unfinished: [] };
81
+ const out = { total: 0, done: 0, unfinished: [], anchors: {} };
82
82
  if (!existsSync(dir)) return out;
83
83
  for (const f of readdirSync(dir)) {
84
84
  if (!/^TASK-[\w.-]+\.md$/i.test(f)) continue;
@@ -86,6 +86,12 @@ export function boardCensus(cwd, slug) {
86
86
  const fm = frontmatter(body);
87
87
  const id = fm.id || f.replace(/\.md$/, "");
88
88
  out.total++;
89
+ // The COMMITTED anchor for this board id. `use_case_refs` is the tier-direction rule's own
90
+ // sanctioned direction (LOCAL names SHARED), and it is what the frozen report cites instead of
91
+ // the id — boards renumber per machine, use cases do not.
92
+ const ucs = String(fm.use_case_refs ?? "").replace(/^\[|\]$/g, "")
93
+ .split(",").map((x) => x.trim()).filter(Boolean);
94
+ out.anchors[id] = ucs;
89
95
  if (fm.status === "done") out.done++;
90
96
  else out.unfinished.push(id);
91
97
  }
@@ -93,6 +99,31 @@ export function boardCensus(cwd, slug) {
93
99
  return out;
94
100
  }
95
101
 
102
+ /**
103
+ * Replace every board id in free prose with its committed anchor.
104
+ *
105
+ * THE WRITE BOUNDARY, not the column, and that is the whole point. Board ids reached the frozen
106
+ * report three different ways — the unfinished-task callout, the covering-AC column's own prefix,
107
+ * and INSIDE acceptance-criterion prose a planner wrote ("given the seeded todos (TASK-006)"). The
108
+ * third is upstream free text, so a fix that only changes what the columns interpolate still
109
+ * commits a file the next run's spec-lint reds. Everything written into the committed report passes
110
+ * through here.
111
+ *
112
+ * An id with no resolvable use case becomes a neutral phrase rather than the id: the report loses a
113
+ * pointer that never resolved off this machine anyway, and keeps the sentence around it.
114
+ *
115
+ * @param {*} text - Any value destined for the committed report.
116
+ * @param {Record<string, string[]>} anchors - Board id → its `use_case_refs`.
117
+ * @returns {string} The text with every `TASK-…` replaced by a stable anchor.
118
+ */
119
+ export function deboard(text, anchors = {}) {
120
+ return String(text ?? "").replace(/\bTASK-[A-Za-z0-9][\w.-]*/g, (id) => {
121
+ const ucs = anchors[id];
122
+ if (ucs && ucs.length) return ucs.join("/");
123
+ return "a board task";
124
+ });
125
+ }
126
+
96
127
  /**
97
128
  * Per-scope T0 outcome, reduced from the trial ledger.
98
129
  *
@@ -194,7 +225,10 @@ export function buildReport(facts) {
194
225
  L.push("");
195
226
 
196
227
  if (board.unfinished.length) {
197
- L.push(`> **${board.unfinished.length} task(s) did not finish:** ${board.unfinished.join(", ")}.`,
228
+ // Anchored, never enumerated by board id: the ids renumber per machine, and a committed file
229
+ // carrying one reds the NEXT run of this pitch at L1b.
230
+ const unfinishedAnchors = [...new Set(board.unfinished.flatMap((id) => board.anchors?.[id] ?? []))];
231
+ L.push(`> **${board.unfinished.length} task(s) did not finish**${unfinishedAnchors.length ? ` — use cases: ${unfinishedAnchors.join(", ")}` : ""}.`,
198
232
  "> The verdict above grades what was built, not what was planned.", "");
199
233
  }
200
234
 
@@ -241,7 +275,9 @@ export function buildReport(facts) {
241
275
  const cell = (s) => String(s).replace(/\|/g, "\\|");
242
276
  L.push("| REQ | source | evidence | covering AC | criterion | T0 |", "|---|---|---|---|---|---|");
243
277
  for (const r of requirements.rows) {
244
- const ac = r.covering_acs.length ? `${r.covering_acs[0].task_id}: ${r.covering_acs[0].ac}${r.covering_acs.length > 1 ? ` (+${r.covering_acs.length - 1})` : ""}` : "—";
278
+ const ac = r.covering_acs.length
279
+ ? `${deboard(r.covering_acs[0].task_id, facts.board?.anchors)}: ${deboard(r.covering_acs[0].ac, facts.board?.anchors)}${r.covering_acs.length > 1 ? ` (+${r.covering_acs.length - 1})` : ""}`
280
+ : "—";
245
281
  const crit = r.criteria.length ? `${r.criteria[0].criterion}${r.criteria.length > 1 ? ` (+${r.criteria.length - 1})` : ""} → ${r.criteria.map((c) => c.verdict).join(",")}` : "—";
246
282
  const t0h = r.t0.length ? r.t0.map((h) => String(h).slice(0, 12)).join(", ") : "—";
247
283
  L.push(`| ${r.id} | ${cell(r.source || "—")} | ${r.evidence} | ${cell(ac)} | ${cell(crit)} | ${t0h} |`);
@@ -362,9 +398,51 @@ export const ARGV_SPEC = {
362
398
  * @returns {(Promise<void>|void)} Settles when the subcommand has written its output; most paths
363
399
  * call `process.exit()` with the subcommand's documented code rather than returning.
364
400
  */
401
+ /**
402
+ * Did THIS run skip evaluation? Read from the run's own recorded arguments, and believed only when
403
+ * the record can be shown to belong to this run.
404
+ *
405
+ * WHY THE KERNEL ASKS AT ALL. A run launched with `--no-eval` verified nothing, and the protocol
406
+ * has always said such a run ships as `not-evaluated`, "recorded plainly — never silently
407
+ * upgraded". That promise lived entirely inside the orchestrator's control flow, where one
408
+ * assignment downstream of the branch restores the old behaviour with every check still green —
409
+ * measured, not supposed. The gate block tells the human the truth either way; the committed report
410
+ * a teammate inherits on `git pull` is the artifact that was lying, so the refusal belongs at the
411
+ * writer of that artifact, where a future orchestrator edit cannot reach it.
412
+ *
413
+ * POSITIVELY PROVEN OR NOT AT ALL. The record is written at launch, later than the receipt that
414
+ * mints the run key, so a freshly opened run can still find the PREVIOUS run's record on disk. A
415
+ * stale flag would refuse a ship that verified everything — a worse failure than the one this
416
+ * closes. The flag therefore counts only when the record names the same run the receipt does;
417
+ * a missing, unreadable or differently-keyed record proves nothing and permits.
418
+ *
419
+ * @param {string} cwd - Project root.
420
+ * @param {string} slug - Feature slug.
421
+ * @returns {boolean} True only when this run's own record says evaluation was skipped.
422
+ */
423
+ function evalWasSkipped(cwd, slug) {
424
+ try {
425
+ const record = JSON.parse(readFileSync(runArgsPath(cwd, slug), "utf8"));
426
+ if (record?.noEval !== true) return false;
427
+ const mine = runIdFromReceipt(readReceipt(receiptPath(cwd, slug)));
428
+ return Boolean(mine) && record.runId === mine;
429
+ } catch { return false; }
430
+ }
431
+
432
+ /** The verdict values that assert the feature was graded and passed. */
433
+ const PASSING = new Set(["PASS", "pass"]);
434
+
365
435
  export async function cli(rawArgv) {
366
436
  const args = runArgs(ARGV_SPEC, rawArgv);
367
437
  const cwd = args.cwd || process.cwd();
438
+ if (PASSING.has(String(args.verdict ?? "")) && evalWasSkipped(cwd, args.slug)) {
439
+ console.error(
440
+ "✋ reduce ship: this run was launched with --no-eval, so nothing graded it — refusing to freeze a report " +
441
+ `that says ${args.verdict}. Ship it as --verdict not-evaluated, which the report, the ledger and the ` +
442
+ "sign-off block all carry.",
443
+ );
444
+ process.exit(3);
445
+ }
368
446
  const { markdown, path } = generate({ cwd, slug: args.slug, verdict: args.verdict, qa: args.qa });
369
447
  if (args.stdout) {
370
448
  process.stdout.write(markdown);