shapeup-sdlc 3.3.0 → 3.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude-plugin/plugin.json +1 -1
- package/AGENTS.md +4 -2
- package/README.md +6 -2
- package/SECURITY.md +1 -1
- package/hooks/sandbox-guard.mjs +69 -6
- package/kernel/compile.mjs +15 -6
- package/kernel/harness.mjs +7 -2
- package/kernel/lib/contract.mjs +88 -5
- package/kernel/probe/requirements.mjs +296 -0
- package/kernel/probe/resume.mjs +7 -1
- package/kernel/reduce/board.mjs +26 -2
- package/kernel/reduce/graph.mjs +31 -14
- package/kernel/reduce/ingest.mjs +16 -3
- package/kernel/reduce/ship.mjs +37 -1
- package/kernel/verify/spec.mjs +130 -4
- package/kernel/verify/trace.mjs +16 -7
- package/package.json +1 -1
- package/skills/ba-pitch-analyzer/SKILL.md +16 -1
- package/skills/scope-architect/SKILL.md +16 -1
- package/skills/scope-hammer/SKILL.md +10 -1
- package/skills/spec-evaluator/SKILL.md +12 -1
- package/skills/tech-lead/references/gates.md +22 -1
- package/skills/tech-lead/schemas/domain.schema.json +5 -0
- package/skills/tech-lead/workflows/shapeup-run.js +88 -11
|
@@ -0,0 +1,296 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
// probe requirements — the way back: a pitch clause, the criterion that graded it, the verdict.
|
|
3
|
+
//
|
|
4
|
+
// WHY THIS IS A QUERY AND NOT A SENTENCE, which is the same reason `probe owner` is one. The
|
|
5
|
+
// requirement matrix is read at GATE L4 and cited by GATE H's census, and both are places where a
|
|
6
|
+
// narrated figure is indistinguishable from a measured one. Measured on the first verdict any run
|
|
7
|
+
// of the spine produced: 85 of 97 criterion rows carried a `traces_to` anchor in the WorkResult and
|
|
8
|
+
// 0 of 97 survived into the projection ingest wrote — so a matrix assembled from memory would have
|
|
9
|
+
// been assembled from an edge that no longer existed on disk. Every figure below is derived from
|
|
10
|
+
// files: the committed registry, the board's `covers:` clauses, the verdict ledger, and the EVAL
|
|
11
|
+
// result's own T0 citations. Nothing is passed in and nothing is written.
|
|
12
|
+
//
|
|
13
|
+
// THE JOIN, and which half is authoritative. A requirement has EVIDENCE when an acceptance
|
|
14
|
+
// criterion covers it (`(covers: REQ-…)` on the AC line — the planning-time edge, reviewed at L1b)
|
|
15
|
+
// AND a criterion that names it passed. `traces_to[]` is the navigation path from the judge's
|
|
16
|
+
// criterion back to the requirement, exactly what the schema already calls it: an anchor, never a
|
|
17
|
+
// grading input. A criterion whose `traces_to` names a REQ that no AC covers is printed as an
|
|
18
|
+
// INCONSISTENCY row and counted as nothing — folding it in would derive one L4 line from two
|
|
19
|
+
// unreconciled sources, which is the failure `probe owner` exists to prevent.
|
|
20
|
+
//
|
|
21
|
+
// IT PROJECTS ONE RUN. `order_id`, round and attempt all repeat across runs of one feature; the
|
|
22
|
+
// run key is the only thing that separates them, so a projection that ignored it would silently mix
|
|
23
|
+
// two runs of one slug. Rows written before the key reached the ledger carry no `run_id`; they are
|
|
24
|
+
// reported as unknown rather than folded into the run being projected.
|
|
25
|
+
//
|
|
26
|
+
// AN EMPTY JOIN READS AS CLEARLY AS A FULL ONE. A tree with no registry, a board with no `covers:`
|
|
27
|
+
// and a run with no verdict are all legitimate states, and each answers "no evidence" rather than
|
|
28
|
+
// failing — that answer is the point of the query, not an error in it.
|
|
29
|
+
//
|
|
30
|
+
// Usage:
|
|
31
|
+
// node "${CLAUDE_PLUGIN_ROOT}/kernel/harness.mjs" probe requirements --slug <slug> [--run-id <id>] [--format json|table] [--cwd <dir>]
|
|
32
|
+
//
|
|
33
|
+
// Exit code: 0 = answered (an empty projection is an answer), 2 = bad argv.
|
|
34
|
+
|
|
35
|
+
import { existsSync, readFileSync, readdirSync } from "node:fs";
|
|
36
|
+
import { resolve, join } from "node:path";
|
|
37
|
+
import { runArgs, isMain } from "../lib/argv.mjs";
|
|
38
|
+
import { requirements as requirementsFile, evaluationDir, resultsDir, readRunId } from "../lib/paths.mjs";
|
|
39
|
+
import { parseRequirements, coveredReqIds } from "../verify/trace.mjs";
|
|
40
|
+
import { readBoard } from "../compile.mjs";
|
|
41
|
+
import { reqId } from "../lib/contract.mjs";
|
|
42
|
+
|
|
43
|
+
/**
|
|
44
|
+
* Read a file, tolerating absence — every input to this projection is optional by design.
|
|
45
|
+
* @param {string} p - Absolute path.
|
|
46
|
+
* @returns {(string|null)} The contents, or null when the file is missing or unreadable.
|
|
47
|
+
*/
|
|
48
|
+
function readIf(p) {
|
|
49
|
+
try { return existsSync(p) ? readFileSync(p, "utf8") : null; } catch { return null; }
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
/**
|
|
53
|
+
* The acceptance criteria on the LOCAL board that cover each requirement.
|
|
54
|
+
*
|
|
55
|
+
* The planning-time half of the join, and the authoritative one: a `covers:` clause is written when
|
|
56
|
+
* the plan is still cheap to change and is reviewed at L1b, whereas `traces_to` is written by the
|
|
57
|
+
* judge after the fact.
|
|
58
|
+
*
|
|
59
|
+
* NOT A SECOND COVERS-CLOSURE. Whether a requirement is covered is still decided by
|
|
60
|
+
* `coveredReqIds` in the oracle — there is one implementation of that question and this is not it.
|
|
61
|
+
* This walk exists only to keep the AC the closure discards, so the matrix can print WHICH
|
|
62
|
+
* criterion covers the clause instead of only that one does.
|
|
63
|
+
*
|
|
64
|
+
* @param {Array<object>} board - Task entries from `readBoard` (the only parser that carries
|
|
65
|
+
* `acceptance_criteria`; the scheduling view does not).
|
|
66
|
+
* @returns {Map<string, Array<{task_id:string, ac:string}>>} REQ-id → the ACs naming it, in board
|
|
67
|
+
* order. A requirement no AC covers is simply absent.
|
|
68
|
+
*/
|
|
69
|
+
export function coveringAcs(board) {
|
|
70
|
+
const out = new Map();
|
|
71
|
+
for (const task of board || []) {
|
|
72
|
+
for (const ac of task.acceptance_criteria || []) {
|
|
73
|
+
if (typeof ac !== "object" || !Array.isArray(ac.covers)) continue;
|
|
74
|
+
for (const raw of ac.covers) {
|
|
75
|
+
const id = reqId(raw).toUpperCase();
|
|
76
|
+
if (!out.has(id)) out.set(id, []);
|
|
77
|
+
out.get(id).push({ task_id: task.id, ac: ac.text });
|
|
78
|
+
}
|
|
79
|
+
}
|
|
80
|
+
}
|
|
81
|
+
return out;
|
|
82
|
+
}
|
|
83
|
+
|
|
84
|
+
/**
|
|
85
|
+
* Every criterion row ingest projected, tagged with the EVAL target its ledger belongs to.
|
|
86
|
+
*
|
|
87
|
+
* The target is read off the FILE NAME rather than the row: `.verdicts-<target>.jsonl` is written
|
|
88
|
+
* per order, so the file is what says which dispatch produced the rows, and the same file is what
|
|
89
|
+
* locates the WorkResult carrying that dispatch's T0 citations.
|
|
90
|
+
*
|
|
91
|
+
* @param {string} cwd - Project root.
|
|
92
|
+
* @param {string} slug - Feature slug.
|
|
93
|
+
* @returns {Array<object>} One entry per ledger line, each the stored row plus `target`. Malformed
|
|
94
|
+
* lines are skipped; a missing evaluation directory yields [].
|
|
95
|
+
*/
|
|
96
|
+
export function readVerdictRows(cwd, slug) {
|
|
97
|
+
const dir = evaluationDir(cwd, slug);
|
|
98
|
+
let files = [];
|
|
99
|
+
try { files = readdirSync(dir).filter((f) => /^\.verdicts-.+\.jsonl$/.test(f)).sort(); } catch { return []; }
|
|
100
|
+
const rows = [];
|
|
101
|
+
for (const f of files) {
|
|
102
|
+
const target = f.replace(/^\.verdicts-/, "").replace(/\.jsonl$/, "");
|
|
103
|
+
for (const line of (readIf(join(dir, f)) || "").split(/\r?\n/)) {
|
|
104
|
+
if (!line.trim()) continue;
|
|
105
|
+
try { rows.push({ ...JSON.parse(line), target }); } catch { /* a truncated line is not a verdict */ }
|
|
106
|
+
}
|
|
107
|
+
}
|
|
108
|
+
return rows;
|
|
109
|
+
}
|
|
110
|
+
|
|
111
|
+
/**
|
|
112
|
+
* The T0 artifacts an EVAL dispatch re-hashed, read from the WorkResult it wrote.
|
|
113
|
+
*
|
|
114
|
+
* The verdict ledger records criteria, not citations, so the machine fact a generator cannot
|
|
115
|
+
* fabricate lives one file over — in `results/<target>.json`. Read here rather than recomputed: the
|
|
116
|
+
* evaluator's own citation is what the round was accepted on.
|
|
117
|
+
*
|
|
118
|
+
* @param {string} cwd - Project root.
|
|
119
|
+
* @param {string} slug - Feature slug.
|
|
120
|
+
* @param {string} target - The EVAL order suffix (`evaluate-r1`), from the ledger's file name.
|
|
121
|
+
* @returns {Array<{scope_id:(string|null), path:(string|null), sha256:(string|null)}>} The cited
|
|
122
|
+
* artifacts; [] when the result is absent, unreadable or cites none.
|
|
123
|
+
*/
|
|
124
|
+
export function t0Citations(cwd, slug, target) {
|
|
125
|
+
const body = readIf(join(resultsDir(cwd, slug), `${target}.json`));
|
|
126
|
+
if (!body) return [];
|
|
127
|
+
try {
|
|
128
|
+
const cites = JSON.parse(body)?.verdict?.t0_citations;
|
|
129
|
+
if (!Array.isArray(cites)) return [];
|
|
130
|
+
return cites.map((c) => ({ scope_id: c?.scope_id ?? null, path: c?.path ?? null, sha256: c?.sha256 ?? null }));
|
|
131
|
+
} catch { return []; }
|
|
132
|
+
}
|
|
133
|
+
|
|
134
|
+
/**
|
|
135
|
+
* Project one run's requirement matrix from the artifacts on disk.
|
|
136
|
+
*
|
|
137
|
+
* @param {{cwd:string, slug:string, runId?:(string|null)}} opts - Project root, feature slug, and
|
|
138
|
+
* the run to project; omitted, it is resolved from the run's own receipt.
|
|
139
|
+
* @returns {{slug:string, run_id:(string|null), registry:boolean, totals:object,
|
|
140
|
+
* rows:Array<object>, inconsistencies:Array<object>, ledger:object}} `rows` is one entry per
|
|
141
|
+
* registered clause — its source, status, covering ACs, the criteria that named it and their
|
|
142
|
+
* verdicts, and each criterion's T0 citations — with `evidence` one of `PASS`, `no evidence` or
|
|
143
|
+
* `cut`. `inconsistencies` holds criteria tracing to a REQ no AC covers. `ledger` says how many
|
|
144
|
+
* rows were read, projected, skipped as another run's, and left unkeyed. No registry ⇒ `rows` is
|
|
145
|
+
* empty and `registry` is false: an empty projection, not an error.
|
|
146
|
+
*/
|
|
147
|
+
export function projectRequirements({ cwd, slug, runId = undefined }) {
|
|
148
|
+
const regText = readIf(requirementsFile(cwd, slug));
|
|
149
|
+
const clauses = regText === null ? [] : parseRequirements(regText);
|
|
150
|
+
const board = readBoard(cwd, slug);
|
|
151
|
+
// `covered` decides; `acs` only says which criterion did the covering.
|
|
152
|
+
const covered = coveredReqIds(board);
|
|
153
|
+
const acs = coveringAcs(board);
|
|
154
|
+
const run = runId === undefined ? readRunId(cwd, slug) : runId;
|
|
155
|
+
|
|
156
|
+
const all = readVerdictRows(cwd, slug);
|
|
157
|
+
const ledger = { rows_read: all.length, rows_projected: 0, rows_other_run: 0, rows_unknown_run: 0 };
|
|
158
|
+
const mine = [];
|
|
159
|
+
for (const r of all) {
|
|
160
|
+
if (r.run_id === undefined || r.run_id === null || r.run_id === "") { ledger.rows_unknown_run++; continue; }
|
|
161
|
+
if (run !== null && r.run_id === run) { ledger.rows_projected++; mine.push(r); continue; }
|
|
162
|
+
ledger.rows_other_run++;
|
|
163
|
+
}
|
|
164
|
+
|
|
165
|
+
// The criteria that name each requirement, and the T0 artifacts the round that graded them cited.
|
|
166
|
+
const t0Cache = new Map();
|
|
167
|
+
/**
|
|
168
|
+
* The cited T0 artifacts for one EVAL target, read once per target.
|
|
169
|
+
* @param {string} target - The EVAL order suffix.
|
|
170
|
+
* @returns {Array<object>} The citation rows.
|
|
171
|
+
*/
|
|
172
|
+
const t0For = (target) => {
|
|
173
|
+
if (!t0Cache.has(target)) t0Cache.set(target, t0Citations(cwd, slug, target));
|
|
174
|
+
return t0Cache.get(target);
|
|
175
|
+
};
|
|
176
|
+
|
|
177
|
+
const byReq = new Map();
|
|
178
|
+
const inconsistencies = [];
|
|
179
|
+
for (const r of mine) {
|
|
180
|
+
const anchors = Array.isArray(r.traces_to) ? r.traces_to : [];
|
|
181
|
+
for (const raw of anchors) {
|
|
182
|
+
const id = reqId(raw).toUpperCase();
|
|
183
|
+
const entry = {
|
|
184
|
+
criterion: r.criterion ?? "", dimension: r.dimension ?? "", verdict: r.verdict ?? "",
|
|
185
|
+
confidence: r.confidence ?? null, evidence: r.evidence ?? "", target: r.target,
|
|
186
|
+
run: r.run ?? null, run_id: r.run_id ?? null, t0: t0For(r.target),
|
|
187
|
+
};
|
|
188
|
+
// THE AUTHORITATIVE HALF DECIDES. An anchor pointing at a requirement no acceptance criterion
|
|
189
|
+
// covers is a claim the plan never made; it is printed so somebody can reconcile it, and
|
|
190
|
+
// counted as nothing.
|
|
191
|
+
if (!covered.has(id)) { inconsistencies.push({ requirement: id, ...entry, why: "no acceptance criterion covers this requirement — the anchor resolves to nothing the plan claimed" }); continue; }
|
|
192
|
+
if (!byReq.has(id)) byReq.set(id, []);
|
|
193
|
+
byReq.get(id).push(entry);
|
|
194
|
+
}
|
|
195
|
+
}
|
|
196
|
+
|
|
197
|
+
const rows = clauses.map((c) => {
|
|
198
|
+
const id = c.id.toUpperCase();
|
|
199
|
+
const covering = acs.get(id) || [];
|
|
200
|
+
const criteria = byReq.get(id) || [];
|
|
201
|
+
const cut = c.status !== "covered";
|
|
202
|
+
const passed = criteria.some((x) => String(x.verdict).toUpperCase() === "PASS");
|
|
203
|
+
return {
|
|
204
|
+
id: c.id,
|
|
205
|
+
source: c.source || "",
|
|
206
|
+
clause: c.clause || "",
|
|
207
|
+
status: c.status,
|
|
208
|
+
covering_acs: covering,
|
|
209
|
+
criteria,
|
|
210
|
+
t0: [...new Set(criteria.flatMap((x) => x.t0.map((t) => t.sha256).filter(Boolean)))],
|
|
211
|
+
evidence: cut ? "cut" : passed ? "PASS" : "no evidence",
|
|
212
|
+
};
|
|
213
|
+
});
|
|
214
|
+
|
|
215
|
+
const totals = {
|
|
216
|
+
total: rows.length,
|
|
217
|
+
pass: rows.filter((r) => r.evidence === "PASS").length,
|
|
218
|
+
cut: rows.filter((r) => r.evidence === "cut").length,
|
|
219
|
+
no_evidence: rows.filter((r) => r.evidence === "no evidence").length,
|
|
220
|
+
inconsistencies: inconsistencies.length,
|
|
221
|
+
};
|
|
222
|
+
return { slug, run_id: run, registry: regText !== null, totals, rows, inconsistencies, ledger };
|
|
223
|
+
}
|
|
224
|
+
|
|
225
|
+
/**
|
|
226
|
+
* The one-line summary GATE L4 prints and GATE H's census cites.
|
|
227
|
+
*
|
|
228
|
+
* @param {object} r - A report from {@link projectRequirements}.
|
|
229
|
+
* @returns {string} `15/17 PASS · 1 CUT (PO) · 1 no evidence (REQ-12 ← R12)`, or the empty-run
|
|
230
|
+
* phrasing when there is no registry to project.
|
|
231
|
+
*/
|
|
232
|
+
export function summaryLine(r) {
|
|
233
|
+
if (!r.registry) return "n/a (no registry)";
|
|
234
|
+
if (!r.totals.total) return "0 requirements registered";
|
|
235
|
+
const gaps = r.rows.filter((x) => x.evidence === "no evidence");
|
|
236
|
+
const named = gaps.slice(0, 3).map((x) => (x.source ? `${x.id} ← ${x.source}` : x.id)).join(", ");
|
|
237
|
+
const parts = [`${r.totals.pass}/${r.totals.total} PASS`];
|
|
238
|
+
if (r.totals.cut) parts.push(`${r.totals.cut} CUT (PO)`);
|
|
239
|
+
if (r.totals.no_evidence) parts.push(`${r.totals.no_evidence} no evidence (${named}${gaps.length > 3 ? ", …" : ""})`);
|
|
240
|
+
if (r.totals.inconsistencies) parts.push(`${r.totals.inconsistencies} inconsistency ${r.totals.inconsistencies === 1 ? "row" : "rows"}`);
|
|
241
|
+
return parts.join(" · ");
|
|
242
|
+
}
|
|
243
|
+
|
|
244
|
+
/**
|
|
245
|
+
* Render the matrix as a fixed-width table for a human reading a gate block.
|
|
246
|
+
* @param {object} r - A report from {@link projectRequirements}.
|
|
247
|
+
* @returns {string} The header, one row per requirement, the summary line and any inconsistencies.
|
|
248
|
+
*/
|
|
249
|
+
export function renderTable(r) {
|
|
250
|
+
const out = [];
|
|
251
|
+
const head = ["REQ", "source", "status", "covering AC", "criterion", "verdict", "T0"];
|
|
252
|
+
const rows = r.rows.map((x) => [
|
|
253
|
+
x.id, x.source || "—", x.evidence,
|
|
254
|
+
x.covering_acs.length ? `${x.covering_acs[0].task_id}: ${x.covering_acs[0].ac.slice(0, 40)}${x.covering_acs.length > 1 ? ` (+${x.covering_acs.length - 1})` : ""}` : "—",
|
|
255
|
+
x.criteria.length ? `${x.criteria[0].criterion.slice(0, 40)}${x.criteria.length > 1 ? ` (+${x.criteria.length - 1})` : ""}` : "—",
|
|
256
|
+
x.criteria.length ? x.criteria.map((c) => c.verdict).join(",") : "—",
|
|
257
|
+
x.t0.length ? x.t0.map((h) => String(h).slice(0, 12)).join(",") : "—",
|
|
258
|
+
]);
|
|
259
|
+
if (rows.length) {
|
|
260
|
+
const w = head.map((h, i) => Math.max(h.length, ...rows.map((row) => String(row[i]).length)));
|
|
261
|
+
const line = (row) => row.map((c, i) => String(c).padEnd(w[i])).join(" ").trimEnd();
|
|
262
|
+
out.push(line(head), line(w.map((n) => "-".repeat(n))), ...rows.map(line), "");
|
|
263
|
+
}
|
|
264
|
+
out.push(`Requirements: ${summaryLine(r)}`);
|
|
265
|
+
out.push(`run_id: ${r.run_id ?? "unknown"} · ledger rows read ${r.ledger.rows_read}, projected ${r.ledger.rows_projected}, other run ${r.ledger.rows_other_run}, unkeyed (run_id: unknown) ${r.ledger.rows_unknown_run}`);
|
|
266
|
+
for (const i of r.inconsistencies) {
|
|
267
|
+
out.push(` ⚠ ${i.requirement}: "${String(i.criterion).slice(0, 60)}" ${i.verdict} — ${i.why}`);
|
|
268
|
+
}
|
|
269
|
+
return out.join("\n");
|
|
270
|
+
}
|
|
271
|
+
|
|
272
|
+
export const ARGV_SPEC = {
|
|
273
|
+
usage: "harness.mjs probe requirements --slug <slug> [--run-id <id>] [--format json|table] [--cwd <dir>]",
|
|
274
|
+
_: { arity: 0, max: 0, name: "(no positional operands)" },
|
|
275
|
+
slug: { type: "str", required: true },
|
|
276
|
+
"run-id": { type: "str" },
|
|
277
|
+
format: { type: "enum", values: ["json", "table"], default: "json" },
|
|
278
|
+
cwd: { type: "path" },
|
|
279
|
+
};
|
|
280
|
+
|
|
281
|
+
/**
|
|
282
|
+
* Answer "which requirement has evidence, and from which criterion" from the artifacts on disk.
|
|
283
|
+
*
|
|
284
|
+
* @param {string[]} rawArgv - The subcommand's own arguments (harness.mjs strips the verb words).
|
|
285
|
+
* @returns {Promise<void>} Exits 0 with the projection — an absent run, an absent registry and an
|
|
286
|
+
* absent verdict are all answers, never errors.
|
|
287
|
+
*/
|
|
288
|
+
export async function cli(rawArgv) {
|
|
289
|
+
const args = runArgs(ARGV_SPEC, rawArgv);
|
|
290
|
+
const cwd = resolve(args.cwd || process.cwd());
|
|
291
|
+
const report = projectRequirements({ cwd, slug: args.slug, runId: args.runId ?? undefined });
|
|
292
|
+
console.log(args.format === "table" ? renderTable(report) : JSON.stringify(report, null, 2));
|
|
293
|
+
process.exit(0);
|
|
294
|
+
}
|
|
295
|
+
|
|
296
|
+
if (isMain(import.meta.url)) cli(process.argv.slice(2));
|
package/kernel/probe/resume.mjs
CHANGED
|
@@ -63,7 +63,7 @@ import { splitFrontmatter } from "../lib/contract.mjs";
|
|
|
63
63
|
import { globToRegExp } from "../verify/spec.mjs";
|
|
64
64
|
import {
|
|
65
65
|
intake, harnessRun, wiringMap, projectProfile, scopesDir, resultsDir, ordersDir,
|
|
66
|
-
orientDir, activeOrder, usecasesDir, breadboard, receipt, readReceipt,
|
|
66
|
+
orientDir, activeOrder, usecasesDir, breadboard, receipt, readReceipt, requirements,
|
|
67
67
|
} from "../lib/paths.mjs";
|
|
68
68
|
import { evalVerdict } from "./eval.mjs";
|
|
69
69
|
|
|
@@ -394,6 +394,12 @@ export function deriveResumeState(cwd, slug) {
|
|
|
394
394
|
orient_dir: `.shapeup/${slug}/orient/`,
|
|
395
395
|
has_orient_artifacts: hasOrientArtifacts(cwd, slug),
|
|
396
396
|
has_spec_tree: hasSpecTree(cwd, slug, hr.spec_folder || null),
|
|
397
|
+
// A PLAIN FACT, DELIBERATELY NOT A PHASE. The requirements registry is dispatched once, before
|
|
398
|
+
// ANALYZE, and the orchestrator guards that one dispatch on this boolean. It is NOT an entry in
|
|
399
|
+
// PHASE_ARTIFACT, and adding it there would be a migration hazard rather than a tidier shape:
|
|
400
|
+
// that map is also `nextPhase()`'s ordered list, so every run recorded before the registry
|
|
401
|
+
// existed would fast-forward to the registry instead of to `build` on its next relaunch.
|
|
402
|
+
has_requirements: existsSync(requirements(cwd, slug)),
|
|
397
403
|
has_wiring_map: existsSync(wiringMap(cwd, slug)),
|
|
398
404
|
project_profile_path: projectProfile(cwd, slug),
|
|
399
405
|
has_project_profile: existsSync(projectProfile(cwd, slug)),
|
package/kernel/reduce/board.mjs
CHANGED
|
@@ -68,7 +68,11 @@ const listField = (fm, key) => {
|
|
|
68
68
|
*/
|
|
69
69
|
export function parseBoard(tasksDir) {
|
|
70
70
|
if (!existsSync(tasksDir)) return [];
|
|
71
|
-
|
|
71
|
+
// SORTED, because `criticalPath` breaks ties on strict `>` and therefore keeps the FIRST chain it
|
|
72
|
+
// meets among equal-hours chains. Directory order is a filesystem detail (APFS happens to return
|
|
73
|
+
// sorted; ext4's hash order does not, and neither does a rename), so an unsorted read made a
|
|
74
|
+
// derived value depend on which machine ran it. Every sibling reader in the kernel already sorts.
|
|
75
|
+
return readdirSync(tasksDir).sort()
|
|
72
76
|
.filter((f) => /^TASK-[\w.-]+\.md$/i.test(f))
|
|
73
77
|
.map((f) => {
|
|
74
78
|
const body = readFileSync(join(tasksDir, f), "utf8");
|
|
@@ -130,10 +134,30 @@ export function criticalPath(tasks) {
|
|
|
130
134
|
}
|
|
131
135
|
return (memo[id] = { hours: best.hours + t.hours, chain: [...best.chain, id] });
|
|
132
136
|
};
|
|
137
|
+
// TIES BREAK ON CONTENT, NOT ON ARRIVAL. `>` alone keeps whichever equal-hours chain is MET
|
|
138
|
+
// FIRST, which is input order — so this function returned a different critical path for the same
|
|
139
|
+
// board depending only on how its task list happened to be ordered. Measured: two disjoint
|
|
140
|
+
// 5-hour chains, five permutations of one list, two different answers.
|
|
141
|
+
//
|
|
142
|
+
// Sorting the reader that feeds it (`parseBoard`) makes the input stable and therefore hides this
|
|
143
|
+
// on any one machine, but it leaves the ORDER-DEPENDENCE in place one call up — a caller with its
|
|
144
|
+
// own ordering, or a future second reader, re-opens it. A derived value has to be a function of
|
|
145
|
+
// the board, not of the walk that produced it, so the tie is resolved here: same hours, then the
|
|
146
|
+
// lexicographically smaller chain. Deterministic on every machine and for every caller.
|
|
147
|
+
/**
|
|
148
|
+
* Is chain `c` a better critical path than the incumbent `b`?
|
|
149
|
+
* @param {{hours:number, chain:string[]}} c - The candidate chain.
|
|
150
|
+
* @param {{hours:number, chain:string[]}} b - The incumbent best.
|
|
151
|
+
* @returns {boolean} True when `c` has more hours, or ties on hours and sorts first by chain
|
|
152
|
+
* content — so the answer is a function of the board rather than of the iteration order.
|
|
153
|
+
*/
|
|
154
|
+
const better = (c, b) => c.hours > b.hours
|
|
155
|
+
|| (c.hours === b.hours && c.chain.length > 0
|
|
156
|
+
&& (b.chain.length === 0 || c.chain.join("\u0000") < b.chain.join("\u0000")));
|
|
133
157
|
let best = { hours: 0, chain: [] };
|
|
134
158
|
for (const t of tasks) {
|
|
135
159
|
const c = longest(t.id);
|
|
136
|
-
if (c
|
|
160
|
+
if (better(c, best)) best = c;
|
|
137
161
|
}
|
|
138
162
|
return best;
|
|
139
163
|
}
|
package/kernel/reduce/graph.mjs
CHANGED
|
@@ -32,7 +32,7 @@ import {
|
|
|
32
32
|
localRoot, receipt as receiptPath, ordersDir, resultsDir, verdictsDir, trials as trialsPath,
|
|
33
33
|
gates as gatesPath, scopesDir, usecasesDir, requirements as requirementsPath, wiringMap as wiringMapPath,
|
|
34
34
|
} from "../lib/paths.mjs";
|
|
35
|
-
import { readAllContracts, readContract, ucId, SCOPE_CONTRACT, WIRING_MAP } from "../lib/contract.mjs";
|
|
35
|
+
import { readAllContracts, readContract, ucId, reqId, SCOPE_CONTRACT, WIRING_MAP } from "../lib/contract.mjs";
|
|
36
36
|
import { runIdFromReceipt } from "../lib/paths.mjs";
|
|
37
37
|
|
|
38
38
|
/** The graph's home — one file per feature, beside the run trace it projects. */
|
|
@@ -43,7 +43,7 @@ export const WORK_NODES = ["Run", "Order", "Result", "Verdict", "Trial", "GateDe
|
|
|
43
43
|
export const DOMAIN_NODES = ["Scope", "UseCase", "Requirement", "Seam"];
|
|
44
44
|
|
|
45
45
|
/** Edge types. Each names a direction that is meaningful to read backwards. */
|
|
46
|
-
export const EDGES = ["PRODUCED", "EVALUATES", "SUPERSEDES", "COVERS", "DEPENDS_ON", "DERIVED_FROM"];
|
|
46
|
+
export const EDGES = ["PRODUCED", "EVALUATES", "SUPERSEDES", "COVERS", "DEPENDS_ON", "DERIVED_FROM", "IMPLEMENTS"];
|
|
47
47
|
|
|
48
48
|
/**
|
|
49
49
|
* Read the graph as a log and fold it into nodes and edges.
|
|
@@ -63,7 +63,10 @@ export function readGraph(cwd, slug) {
|
|
|
63
63
|
let row;
|
|
64
64
|
try { row = JSON.parse(line); } catch { continue; } // a torn line proves nothing; skip it
|
|
65
65
|
lines++;
|
|
66
|
-
|
|
66
|
+
// LAST LINE WINS, as the banner says — a REPLACE, not a merge. Merging kept attributes from
|
|
67
|
+
// superseded lines alive forever, so an attribute removed from an artifact survived in the
|
|
68
|
+
// projection and a rebuilt graph stopped matching the maintained one.
|
|
69
|
+
if (row.k === "node" && row.id) nodes.set(row.id, row);
|
|
67
70
|
else if (row.k === "edge" && row.from && row.to && row.t) edges.set(`${row.from}|${row.t}|${row.to}`, row);
|
|
68
71
|
}
|
|
69
72
|
return { nodes, edges, lines };
|
|
@@ -194,16 +197,19 @@ export function project(cwd, slug) {
|
|
|
194
197
|
source: g.source ?? null, note: g.note ?? null, round: g.round ?? null,
|
|
195
198
|
run_id: g.run_id ?? runId ?? null,
|
|
196
199
|
});
|
|
197
|
-
//
|
|
198
|
-
//
|
|
199
|
-
//
|
|
200
|
-
// `
|
|
200
|
+
// EVERY gate depends on the Run, UNCONDITIONALLY — and a round-scoped one ALSO depends on that
|
|
201
|
+
// round's T0 verdict(s), the evidence it was decided against.
|
|
202
|
+
//
|
|
203
|
+
// The Run edge used to be an `else` FALLBACK, and a conditional edge TARGET cannot survive an
|
|
204
|
+
// append-only log. Gates are crossed BEFORE the round's verdict artifact lands, so the first
|
|
205
|
+
// projection minted `DEPENDS_ON run` and a later one added the verdict edges beside it —
|
|
206
|
+
// and `appendGraph` has no tombstone, so the fallback became permanent. The incremental graph
|
|
207
|
+
// then carried an edge a rebuild does not imply, breaking the one property this file exists to
|
|
208
|
+
// promise: that it can be deleted and rebuilt identically. Unconditional is also truer — a gate
|
|
209
|
+
// does depend on its run. The fix is not a guard; it is removing the condition.
|
|
210
|
+
if (runNode) edge(id, "DEPENDS_ON", runNode);
|
|
201
211
|
const roundVerdicts = (g.gate === "L2" || g.gate === "L3") ? verdictIdsByRound.get(g.round) : null;
|
|
202
|
-
|
|
203
|
-
for (const vid of roundVerdicts) edge(id, "DEPENDS_ON", vid);
|
|
204
|
-
} else if (runNode) {
|
|
205
|
-
edge(id, "DEPENDS_ON", runNode);
|
|
206
|
-
}
|
|
212
|
+
for (const vid of roundVerdicts || []) edge(id, "DEPENDS_ON", vid);
|
|
207
213
|
}
|
|
208
214
|
}
|
|
209
215
|
|
|
@@ -221,7 +227,15 @@ export function project(cwd, slug) {
|
|
|
221
227
|
// projected to two nodes. `--trace` is supposed to reach "the execution record"; it reached
|
|
222
228
|
// whichever scope happened to be written last. `baseline_trial` is chosen from the same
|
|
223
229
|
// scope's prior rows, so the SUPERSEDES edge resolves inside the same partition.
|
|
224
|
-
|
|
230
|
+
//
|
|
231
|
+
// AND THE RUN IS PART OF THE KEY TOO, for the same reason one step out. A trial ordinal
|
|
232
|
+
// restarts at 1 in the next run while `trials.jsonl` is APPEND-ONLY, so both runs' rows live
|
|
233
|
+
// on disk together and `trial:<slug>:<scope>:1` named two of them — the second run silently
|
|
234
|
+
// overwrote the first run's execution record, which is the identical defect the paragraph
|
|
235
|
+
// above records fixing once already, one key component short. `baseline_trial` is chosen from
|
|
236
|
+
// the same run's rows, so SUPERSEDES still resolves inside the partition.
|
|
237
|
+
const trialRun = t.run_id ?? runId ?? "norun";
|
|
238
|
+
const trialKey = (n) => `trial:${slug}:${trialRun}:${t.scope_id ? `${t.scope_id}:` : ""}${n}`;
|
|
225
239
|
const id = trialKey(t.trial);
|
|
226
240
|
node(id, "Trial", {
|
|
227
241
|
trial: t.trial, round: t.round ?? null, attempt: t.attempt ?? null,
|
|
@@ -249,7 +263,10 @@ export function project(cwd, slug) {
|
|
|
249
263
|
// UseCase nodes it was built from and `--trace` stopped there instead of reaching the
|
|
250
264
|
// objective. The contract now names both, so both are projectable.
|
|
251
265
|
for (const uc of contract.use_cases || []) edge(id, "IMPLEMENTS", `uc:${slug}:${ucId(uc)}`);
|
|
252
|
-
|
|
266
|
+
// Normalised to the registry's key space on the way in, the same mapping spec-lint applies:
|
|
267
|
+
// a contract citing the pitch's `R<n>` and a Requirement node keyed `REQ-<n>` are one node,
|
|
268
|
+
// and an edge drawn to the other spelling is an edge to a node the graph does not hold.
|
|
269
|
+
for (const req of contract.covers || []) edge(id, "COVERS", `req:${slug}:${reqId(req)}`);
|
|
253
270
|
for (const dep of contract.depends_on || []) edge(id, "DEPENDS_ON", `scope:${slug}:${dep}`);
|
|
254
271
|
}
|
|
255
272
|
}
|
package/kernel/reduce/ingest.mjs
CHANGED
|
@@ -8,7 +8,9 @@
|
|
|
8
8
|
// task_results[] → tick AC boxes, flip task frontmatter status, append Execution Log,
|
|
9
9
|
// update tasks/_index.md row, propagate unblocks (old P3.1–P3.6)
|
|
10
10
|
// discoveries[] → append to .shapeup/<slug>/discovery/ledger.md (old P3.7 / QA H.3)
|
|
11
|
-
// verdict.criteria[] → append evaluation/.verdicts-<target>.jsonl (old evaluator B.0)
|
|
11
|
+
// verdict.criteria[] → append evaluation/.verdicts-<target>.jsonl (old evaluator B.0), every
|
|
12
|
+
// row keyed by run_id and carrying the judge's traces_to[] anchor back to
|
|
13
|
+
// the requirement — see step 4 for why neither may be dropped here
|
|
12
14
|
// verdict.refuted[] → un-tick refuted AC boxes + set eval_verdict frontmatter (old B.2/B.2b)
|
|
13
15
|
// (the leg itself) → append one leg-completion row to .shapeup/<slug>/legs.jsonl
|
|
14
16
|
//
|
|
@@ -286,21 +288,32 @@ function applyResultLocked(result, { cwd, slug }) {
|
|
|
286
288
|
}
|
|
287
289
|
|
|
288
290
|
// 4. Verdict bookkeeping (old evaluator B.0/B.2/B.2b) — judge returns data, ingest writes.
|
|
291
|
+
//
|
|
292
|
+
// THE PROJECTION KEEPS THE TWO KEYS A LATER READER CANNOT RE-DERIVE. Measured on the first real
|
|
293
|
+
// verdict of a spined run: 85 of 97 criteria carried `traces_to` in the WorkResult and 0 of 97
|
|
294
|
+
// survived into this file, so the anchor from a criterion back to the requirement it grades was
|
|
295
|
+
// written by the judge and discarded one step later. And the row carried only the monotonic `run`
|
|
296
|
+
// counter, which restarts per ledger file and repeats across runs of one slug — `run_id` is the
|
|
297
|
+
// only key that separates them, so a projection over this file silently mixed two runs without
|
|
298
|
+
// it. A WorkResult carries no run key; it is read off the run's own receipt, the same derivation
|
|
299
|
+
// every other writer here uses.
|
|
289
300
|
if (result.verdict) {
|
|
290
301
|
const evalDir = join(local, "evaluation");
|
|
291
302
|
mkdirSync(evalDir, { recursive: true });
|
|
292
303
|
if (result.verdict.criteria?.length) {
|
|
293
304
|
const target = result.order_id.split("/")[1] || "run";
|
|
294
305
|
const ledger = join(evalDir, `.verdicts-${target}.jsonl`);
|
|
306
|
+
const runId = readRunId(cwd, slug);
|
|
295
307
|
let run = 1;
|
|
296
308
|
if (existsSync(ledger)) {
|
|
297
309
|
const prior = readFileSync(ledger, "utf8").trim().split(/\n/).filter(Boolean).map((l) => { try { return JSON.parse(l); } catch { return null; } }).filter(Boolean);
|
|
298
310
|
run = prior.reduce((mx, r) => Math.max(mx, r.run || 0), 0) + 1;
|
|
299
311
|
}
|
|
300
312
|
const lines = result.verdict.criteria.map((c) => JSON.stringify({
|
|
301
|
-
run, dimension: c.dimension || "spec-conformance", criterion: c.criterion,
|
|
313
|
+
run, run_id: runId, dimension: c.dimension || "spec-conformance", criterion: c.criterion,
|
|
302
314
|
verdict: c.verdict, confidence: c.confidence, reprobed: !!c.reprobed,
|
|
303
|
-
evidence: c.evidence || "",
|
|
315
|
+
evidence: c.evidence || "", traces_to: Array.isArray(c.traces_to) ? c.traces_to : [],
|
|
316
|
+
at: new Date().toISOString(),
|
|
304
317
|
})).join("\n");
|
|
305
318
|
appendFileSync(ledger, lines + "\n");
|
|
306
319
|
summary.verdict_lines = result.verdict.criteria.length;
|
package/kernel/reduce/ship.mjs
CHANGED
|
@@ -36,6 +36,7 @@ import {
|
|
|
36
36
|
} from "../lib/paths.mjs";
|
|
37
37
|
import { readTrials } from "../verify/t0.mjs";
|
|
38
38
|
import { ratchetReport } from "../probe/stats.mjs";
|
|
39
|
+
import { projectRequirements, summaryLine } from "../probe/requirements.mjs";
|
|
39
40
|
import { collectDiff, scanDiff, summarize } from "./leftovers.mjs";
|
|
40
41
|
|
|
41
42
|
/** @returns {string} Today as `YYYY-MM-DD` (UTC). */
|
|
@@ -167,7 +168,7 @@ export function section(md, heading) {
|
|
|
167
168
|
export function buildReport(facts) {
|
|
168
169
|
const {
|
|
169
170
|
slug, at, verdict, qa, rounds, board, t0, artifacts, ratchet,
|
|
170
|
-
evalCriteria, evalBugs, qaFindings, decisions, discovered, intakeSha, leftovers,
|
|
171
|
+
evalCriteria, evalBugs, qaFindings, decisions, discovered, intakeSha, leftovers, requirements,
|
|
171
172
|
} = facts;
|
|
172
173
|
|
|
173
174
|
const L = [];
|
|
@@ -214,6 +215,40 @@ export function buildReport(facts) {
|
|
|
214
215
|
L.push("");
|
|
215
216
|
}
|
|
216
217
|
|
|
218
|
+
// The requirement matrix — the way back from a verdict to the clause the pitch asked for, frozen
|
|
219
|
+
// at the one moment the run's local evidence still exists. Omitted entirely when the run has no
|
|
220
|
+
// registry: a table of nothing reads as "no requirements", which is a different claim from "this
|
|
221
|
+
// run predates the registry". Derived like every other figure here, by the same probe the L4 line
|
|
222
|
+
// and GATE H's census read, so the three cannot disagree.
|
|
223
|
+
if (requirements?.registry && requirements.rows.length) {
|
|
224
|
+
L.push("## Requirements", "");
|
|
225
|
+
L.push("One row per registered clause. A requirement has evidence when an acceptance criterion",
|
|
226
|
+
"covers it AND a criterion grading it passed — `covers:` is the join, the judge's anchor is the",
|
|
227
|
+
"path back. This is a projection, never a verdict: it never blocked this ship.", "");
|
|
228
|
+
L.push(`**${summaryLine(requirements)}** · run \`${requirements.run_id ?? "unknown"}\``, "");
|
|
229
|
+
// A clause, an AC and a criterion are all free prose, and a literal pipe in any of them breaks
|
|
230
|
+
// the row into columns nobody wrote — a frozen report that misrenders its own evidence.
|
|
231
|
+
/**
|
|
232
|
+
* Escape a free-prose value for a Markdown table cell.
|
|
233
|
+
* @param {string} s - The value.
|
|
234
|
+
* @returns {string} The value with every literal pipe escaped.
|
|
235
|
+
*/
|
|
236
|
+
const cell = (s) => String(s).replace(/\|/g, "\\|");
|
|
237
|
+
L.push("| REQ | source | evidence | covering AC | criterion | T0 |", "|---|---|---|---|---|---|");
|
|
238
|
+
for (const r of requirements.rows) {
|
|
239
|
+
const ac = r.covering_acs.length ? `${r.covering_acs[0].task_id}: ${r.covering_acs[0].ac}${r.covering_acs.length > 1 ? ` (+${r.covering_acs.length - 1})` : ""}` : "—";
|
|
240
|
+
const crit = r.criteria.length ? `${r.criteria[0].criterion}${r.criteria.length > 1 ? ` (+${r.criteria.length - 1})` : ""} → ${r.criteria.map((c) => c.verdict).join(",")}` : "—";
|
|
241
|
+
const t0h = r.t0.length ? r.t0.map((h) => String(h).slice(0, 12)).join(", ") : "—";
|
|
242
|
+
L.push(`| ${r.id} | ${cell(r.source || "—")} | ${r.evidence} | ${cell(ac)} | ${cell(crit)} | ${t0h} |`);
|
|
243
|
+
}
|
|
244
|
+
L.push("");
|
|
245
|
+
if (requirements.inconsistencies.length) {
|
|
246
|
+
L.push("Anchored to a requirement no acceptance criterion covers — reconcile, do not count as evidence:", "");
|
|
247
|
+
for (const i of requirements.inconsistencies) L.push(`- ${i.requirement} ← "${i.criterion}" (${i.verdict})`);
|
|
248
|
+
L.push("");
|
|
249
|
+
}
|
|
250
|
+
}
|
|
251
|
+
|
|
217
252
|
// The ratchet aggregate is derived, ~10 scalars that do not grow with the run, which is why it
|
|
218
253
|
// can live in the committed tier while `metrics/` correctly stays gitignored (ADR-0001: a
|
|
219
254
|
// committed shard keyed on $HOSTNAME only grows). Without this the instrument existed and was
|
|
@@ -306,6 +341,7 @@ export function generate({ cwd, slug, verdict, qa }) {
|
|
|
306
341
|
board: boardCensus(cwd, slug),
|
|
307
342
|
t0: t0Summary(cwd, slug),
|
|
308
343
|
ratchet: ratchetReport(readTrials(trials(cwd, slug))),
|
|
344
|
+
requirements: projectRequirements({ cwd, slug }),
|
|
309
345
|
artifacts: verdictArtifactCount(cwd, slug),
|
|
310
346
|
evalCriteria: section(evalReport, /^#+\s.*criteria/i) || section(evalReport, /^#+\s*spec-conformance/i),
|
|
311
347
|
evalBugs: section(evalReport, /^#+\s*Bugs?\b/i),
|