shapeup-sdlc 3.7.8 → 3.7.12
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude-plugin/plugin.json +1 -1
- package/README.md +6 -4
- package/hooks/sandbox-guard.mjs +9 -0
- package/kernel/compile.mjs +67 -15
- package/kernel/probe/eval.mjs +19 -5
- package/kernel/probe/resume.mjs +24 -3
- package/kernel/probe/staged.mjs +83 -0
- package/kernel/report/export.mjs +5 -0
- package/kernel/schemas/domain.schema.json +34 -1
- package/kernel/verify/env.mjs +197 -0
- package/kernel/verify/spec.mjs +70 -0
- package/kernel/verify/t0.mjs +22 -4
- package/package.json +1 -1
- package/skills/ba-pitch-analyzer/assets/templates/synthesis.tmpl.md +5 -0
- package/skills/tech-lead/workflows/shapeup-run.js +18 -1
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "shapeup-sdlc-plugin",
|
|
3
3
|
"displayName": "ShapeUp SDLC Plugin",
|
|
4
|
-
"version": "3.7.
|
|
4
|
+
"version": "3.7.12",
|
|
5
5
|
"description": "Shape Up SDLC harness for Claude Code: shaping, intake, orient, scope-mapping, building (T0-verified, sandboxed, scope-contracted), evaluation and QA skills orchestrated by a tech-lead.",
|
|
6
6
|
"author": {
|
|
7
7
|
"name": "Liberty Nguyen",
|
package/README.md
CHANGED
|
@@ -48,10 +48,12 @@ Two limits, stated here because the point of this section is that a claim withou
|
|
|
48
48
|
behind it is the thing this harness exists to prevent: the **seesaw** regression arm is declared
|
|
49
49
|
and not yet wired (no run writes its registry — wiring it is an open Betting Table decision), and
|
|
50
50
|
the citation **re-hash** the kernel performs proves self-consistency, not provenance — the digest
|
|
51
|
-
and the cited verdict are checked, the scope, round and run the artifact belongs to are not.
|
|
52
|
-
|
|
53
|
-
about the
|
|
54
|
-
|
|
51
|
+
and the cited verdict are checked, the scope, round and run the artifact belongs to are not. Both are
|
|
52
|
+
open items in `shapeup/knowledge-base/harness-defects.md`, not shipped guarantees. A T0 artifact is
|
|
53
|
+
also evidence about the machine that produced it, and now says so: each verdict carries where it
|
|
54
|
+
ran — the absolute path, the git tree, the resolved toolchain, lockfile digests, declared cache
|
|
55
|
+
directories and a digest over an allowlist of environment values — so a disagreeing re-run can be
|
|
56
|
+
told from a regression. That block measures and judges nothing.
|
|
55
57
|
→ *Prevents: "done" asserted with nothing behind it.*
|
|
56
58
|
|
|
57
59
|
**3. Parallel work can't corrupt shared state.** Each scope gets a write-whitelist of files
|
package/hooks/sandbox-guard.mjs
CHANGED
|
@@ -329,6 +329,11 @@ async function main() {
|
|
|
329
329
|
allowed: [...(o.substrate.allowed || []), ...(o.substrate.shared || [])],
|
|
330
330
|
appendOnly: o.substrate.append_only || [],
|
|
331
331
|
frozen: o.substrate.frozen || [],
|
|
332
|
+
// The one exception to "frozen outranks everything", and it is the compiler's to grant, never
|
|
333
|
+
// the worker's to request: the paths THIS order authors, named from its own identity. A build
|
|
334
|
+
// leg's substrate freezes the whole run trace, so without this its own WorkResult — its
|
|
335
|
+
// documented last step — would be denied along with every channel it must not touch.
|
|
336
|
+
own: o.substrate.own || [],
|
|
332
337
|
})).filter((c) => c.allowed.length || c.appendOnly.length || c.frozen.length);
|
|
333
338
|
|
|
334
339
|
if (contracts.length === 0) defer("no live order declares write/append/frozen boundaries", "no-whitelist");
|
|
@@ -363,6 +368,10 @@ async function main() {
|
|
|
363
368
|
// a planner is graded against — so the compiler emitted a declaration with no enforcer, which is
|
|
364
369
|
// the exact state this hook exists to end. A path a live contract freezes is a violation
|
|
365
370
|
// wherever it lives.
|
|
371
|
+
// `own` first, and only against the contract that declared it: another live order's exception
|
|
372
|
+
// never licenses this write. A path no contract claims as its own falls through to the freeze.
|
|
373
|
+
if (contracts.some((c) => matchesAny(rel, c.own, fold))) continue;
|
|
374
|
+
|
|
366
375
|
const freezer = contracts.find((c) => matchesAny(rel, c.frozen, fold));
|
|
367
376
|
if (freezer) {
|
|
368
377
|
violations.push(rel);
|
package/kernel/compile.mjs
CHANGED
|
@@ -211,13 +211,37 @@ export const OP_OWNER = {
|
|
|
211
211
|
* operation, so mode/flag differences are enforced by the sandbox hook reading the order's substrate, not trusted to prose.
|
|
212
212
|
* @param {string} operation - The order's operation (execute|fix|spike|analyze|reconcile|
|
|
213
213
|
* retrofit-surface|coverage|map-scopes|wire|evaluate|orient|hunt|translate|hammer|coach|scan|research).
|
|
214
|
-
* @param {{slug?:string, specDir?:string, scope?:object}} [ctx] - slug (names
|
|
215
|
-
* specDir (overrides the default spec path), scope (contract supplying
|
|
216
|
-
*
|
|
217
|
-
*
|
|
218
|
-
*
|
|
214
|
+
* @param {{slug?:string, specDir?:string, scope?:object, ownStem?:string}} [ctx] - slug (names
|
|
215
|
+
* LOCAL/SHARED roots), specDir (overrides the default spec path), scope (contract supplying
|
|
216
|
+
* allowed/shared substrates), ownStem (this order's own file stem, so a build leg may write its
|
|
217
|
+
* own WorkResult and no one else's).
|
|
218
|
+
* @returns {{allowed:string[], shared?:string[], frozen?:string[], append_only?:string[], own?:string[]}}
|
|
219
|
+
* The substrate contract: globs the worker may write (`allowed`), shared-write globs, read-only
|
|
220
|
+
* `frozen` globs, `append_only` globs, and `own` — the paths this order may write DESPITE a
|
|
221
|
+
* broader freeze, derived by the compiler from the order's own identity and never requested.
|
|
222
|
+
* An unknown operation returns a LOCAL-only default.
|
|
219
223
|
*/
|
|
220
|
-
export function substrateFor(operation,
|
|
224
|
+
export function substrateFor(operation, ctx = {}) {
|
|
225
|
+
// EVERY ORDER NAMES THE RESULT IT ANSWERS WITH. A worker used to infer that path from its order's
|
|
226
|
+
// own filename — the one thing in the envelope that was convention rather than contract — so the
|
|
227
|
+
// one file every dispatch must write was the one the order did not mention. It is `own` for every
|
|
228
|
+
// operation now: the compiler derives it from the order's identity, which is also what keeps a leg
|
|
229
|
+
// from writing somebody else's.
|
|
230
|
+
const base = substrateTemplate(operation, ctx);
|
|
231
|
+
if (!ctx.ownStem) return base;
|
|
232
|
+
const ownResult = `${globLocal(ctx.slug)}/results/${ctx.ownStem}.json`;
|
|
233
|
+
return { ...base, own: [...new Set([...(base.own || []), ownResult])] };
|
|
234
|
+
}
|
|
235
|
+
|
|
236
|
+
/**
|
|
237
|
+
* The per-operation template {@link substrateFor} builds on.
|
|
238
|
+
*
|
|
239
|
+
* @param {string} operation - The order's operation.
|
|
240
|
+
* @param {object} [ctx] - As {@link substrateFor}: slug, specDir, scope, ownStem.
|
|
241
|
+
* @returns {{allowed:string[], shared?:string[], frozen?:string[], append_only?:string[], own?:string[]}}
|
|
242
|
+
* The operation's write contract before the result path every order names is merged in.
|
|
243
|
+
*/
|
|
244
|
+
function substrateTemplate(operation, { slug, specDir, scope, ownStem = null } = {}) {
|
|
221
245
|
const local = globLocal(slug);
|
|
222
246
|
const spec = specDir || globShared(slug, "spec");
|
|
223
247
|
const scopesDir = globShared(slug, "scopes");
|
|
@@ -266,15 +290,30 @@ export function substrateFor(operation, { slug, specDir, scope } = {}) {
|
|
|
266
290
|
const FROZEN_ATTESTATION = [`${local}/receipts/**`, `${local}/legs.jsonl`, `${local}/t0/verdicts/**`, `${local}/tasks/_index.md`];
|
|
267
291
|
switch (operation) {
|
|
268
292
|
case "execute": case "fix": case "spike":
|
|
269
|
-
//
|
|
270
|
-
//
|
|
271
|
-
//
|
|
272
|
-
//
|
|
273
|
-
//
|
|
293
|
+
// THE RUN TRACE IS THE KERNEL'S, EXCEPT WHAT THIS LEG AUTHORS. The freeze used to be a list of
|
|
294
|
+
// channels a defect had named — the staged pitch, the receipts, the leg ledger, the T0
|
|
295
|
+
// verdicts, the board index — and each review found more of the same class: the trial ledger,
|
|
296
|
+
// the gate ledger, the round build gates, the graph, the run args, the run ledger itself. A
|
|
297
|
+
// list that grows one defect at a time is not a boundary. So the boundary is inverted here:
|
|
298
|
+
// everything under the run trace is frozen for a build leg, and `own` carries the short,
|
|
299
|
+
// derived list of what such a leg actually writes. Everything else under there is written by
|
|
300
|
+
// the kernel in its own process or by the hook layer, and neither goes through this guard —
|
|
301
|
+
// so freezing it costs no legitimate write. `${local}/**` subsumes FROZEN_INTAKE and
|
|
302
|
+
// FROZEN_ATTESTATION for this operation, and a check asserts that rather than restating them.
|
|
274
303
|
return {
|
|
275
304
|
allowed: [...(scope?.allowed_file_substrate || []), `${local}/spikes/**`],
|
|
276
305
|
shared: scope?.shared_substrate || [],
|
|
277
|
-
frozen: [
|
|
306
|
+
frozen: [`${local}/**`],
|
|
307
|
+
own: [
|
|
308
|
+
// The result this order answers with, and no sibling's: a leg that can write another
|
|
309
|
+
// leg's WorkResult can report work nobody did, and ingest would have no way to tell.
|
|
310
|
+
// With no stem the caller is asking about the operation in general, not about one order,
|
|
311
|
+
// and the honest answer is the whole directory rather than a guess at which file.
|
|
312
|
+
...(ownStem ? [`${local}/results/${ownStem}.json`] : [`${local}/results/**`]),
|
|
313
|
+
`${local}/tasks/TASK-*.md`,
|
|
314
|
+
`${local}/discovery/**`,
|
|
315
|
+
`${local}/spikes/**`,
|
|
316
|
+
],
|
|
278
317
|
};
|
|
279
318
|
case "analyze":
|
|
280
319
|
return { allowed: [`${spec}/**`, `${local}/**`], frozen: [...FROZEN_INTAKE] };
|
|
@@ -315,9 +354,22 @@ export function substrateFor(operation, { slug, specDir, scope } = {}) {
|
|
|
315
354
|
frozen: [...FROZEN_SPEC_CORE, ...FROZEN_INTAKE, `${scopesDir}/**`, globShared(slug, "project-profile.md")],
|
|
316
355
|
};
|
|
317
356
|
case "evaluate":
|
|
318
|
-
|
|
357
|
+
// THE JUDGE MAY NOT WRITE THE EVIDENCE IT CITES. Its substrate froze the spec, the pitch and
|
|
358
|
+
// the board — and not `t0/verdicts/**`, so a judge could write a green verdict artifact into
|
|
359
|
+
// the canonical directory under the run-trace carve-out and then cite it, correctly hashed.
|
|
360
|
+
// Same inversion as a build leg: the run trace is the kernel's, and `own` names what the
|
|
361
|
+
// judge authors — its report and the envelope that answers its order.
|
|
362
|
+
return {
|
|
363
|
+
allowed: [],
|
|
364
|
+
frozen: [`${spec}/**`, `${local}/**`],
|
|
365
|
+
own: [`${local}/evaluation/**`, ...(ownStem ? [`${local}/results/${ownStem}.json`] : [`${local}/results/**`])],
|
|
366
|
+
};
|
|
319
367
|
case "hunt":
|
|
320
|
-
return {
|
|
368
|
+
return {
|
|
369
|
+
allowed: [],
|
|
370
|
+
frozen: [`${spec}/**`, `${local}/**`],
|
|
371
|
+
own: [`${local}/qa/**`, ...(ownStem ? [`${local}/results/${ownStem}.json`] : [`${local}/results/**`])],
|
|
372
|
+
};
|
|
321
373
|
case "orient":
|
|
322
374
|
return { allowed: [`${local}/orient/**`], frozen: [`${spec}/**`] };
|
|
323
375
|
case "translate":
|
|
@@ -782,7 +834,7 @@ export function compileOrder({
|
|
|
782
834
|
mode,
|
|
783
835
|
...(operation ? { operation } : {}),
|
|
784
836
|
...(interaction ? { interaction } : {}),
|
|
785
|
-
substrate: substrateFor(operation, { slug, specDir, scope }),
|
|
837
|
+
substrate: substrateFor(operation, { slug, specDir, scope, ownStem: suffix }),
|
|
786
838
|
payload: {
|
|
787
839
|
...(scope ? { scope_contract: scope } : {}),
|
|
788
840
|
...(tasks?.length ? { tasks } : {}),
|
package/kernel/probe/eval.mjs
CHANGED
|
@@ -31,11 +31,11 @@
|
|
|
31
31
|
// cause as its first deviation. A bare `ok: false` reached the operator as a sub-agent that died
|
|
32
32
|
// after retries, while the one sentence naming the actual cause sat in a file nobody was pointed at.
|
|
33
33
|
|
|
34
|
-
import { existsSync, readFileSync, readdirSync } from "node:fs";
|
|
35
|
-
import { join, resolve } from "node:path";
|
|
34
|
+
import { existsSync, readFileSync, readdirSync, realpathSync } from "node:fs";
|
|
35
|
+
import { join, resolve, resolve as resolvePath, sep } from "node:path";
|
|
36
36
|
import { createHash } from "node:crypto";
|
|
37
37
|
import { runArgs } from "../lib/argv.mjs";
|
|
38
|
-
import { resultsDir, scopesDir, readRunId } from "../lib/paths.mjs";
|
|
38
|
+
import { resultsDir, scopesDir, readRunId, verdictsDir } from "../lib/paths.mjs";
|
|
39
39
|
|
|
40
40
|
/** Longest `reason` reported. A deviation is prose written by a worker and can run to paragraphs. */
|
|
41
41
|
const REASON_MAX = 400;
|
|
@@ -89,7 +89,7 @@ export function isScoped(cwd, slug) {
|
|
|
89
89
|
* @returns {(string|null)} A reason phrased for an operator, or null when the file at `path` exists,
|
|
90
90
|
* hashes to the cited `sha256`, and its own `overall` reads "green".
|
|
91
91
|
*/
|
|
92
|
-
function unresolvedCitation(cwd, citation, { round = null, runId = null } = {}) {
|
|
92
|
+
function unresolvedCitation(cwd, citation, { round = null, runId = null, verdicts = null } = {}) {
|
|
93
93
|
const rel = typeof citation?.path === "string" ? citation.path : "";
|
|
94
94
|
if (!rel) return "names no artifact path";
|
|
95
95
|
let text;
|
|
@@ -109,6 +109,19 @@ function unresolvedCitation(cwd, citation, { round = null, runId = null } = {})
|
|
|
109
109
|
try { body = JSON.parse(text); }
|
|
110
110
|
catch { return `cites ${rel}, whose bytes match the hash but do not read as a T0 verdict`; }
|
|
111
111
|
if (body?.overall !== "green") return `cites ${rel}, whose own verdict is "${body?.overall ?? "unknown"}", not green`;
|
|
112
|
+
// INSIDE THIS RUN'S OWN VERDICTS DIRECTORY, resolved — asked of a file that exists, so "does not
|
|
113
|
+
// exist" and "resolves somewhere else" stay different answers. Accepted before this check: a
|
|
114
|
+
// green artifact in the source tree, one outside the project reached by `../`, one by absolute
|
|
115
|
+
// path, and a symlink in the verdicts directory pointing at a forged file. Every one hashed
|
|
116
|
+
// correctly, because a digest says the bytes are the file's and nothing about which file it
|
|
117
|
+
// should have been.
|
|
118
|
+
if (verdicts) {
|
|
119
|
+
const real = (() => { try { return realpathSync.native(resolvePath(cwd, rel)); } catch { return resolvePath(cwd, rel); } })();
|
|
120
|
+
const home = (() => { try { return realpathSync.native(verdicts); } catch { return verdicts; } })();
|
|
121
|
+
if (!real.startsWith(home.endsWith(sep) ? home : home + sep)) {
|
|
122
|
+
return `cites ${rel}, which resolves outside this run's own verdicts directory (${home}) — a verdict cites what this run's own verifier wrote, not a file the judge can reach`;
|
|
123
|
+
}
|
|
124
|
+
}
|
|
112
125
|
// THE ARTIFACT HAS TO BE THE ONE THE CITATION SAYS IT IS. A re-hash proves the bytes are the
|
|
113
126
|
// file's; it says nothing about whose verdict the file holds. A PASS citing scope alpha's green
|
|
114
127
|
// artifact while declaring scope beta, or a prior round's, or a prior run's over the same slug,
|
|
@@ -198,8 +211,9 @@ export function citationProblem(cwd, slug, verdict, { round = null } = {}) {
|
|
|
198
211
|
"cite the T0 verdict it re-hashed (the order lists them under payload.t0_artifacts)";
|
|
199
212
|
}
|
|
200
213
|
const runId = readRunId(cwd, slug);
|
|
214
|
+
const verdicts = verdictsDir(cwd, slug);
|
|
201
215
|
for (const citation of verdict.t0_citations) {
|
|
202
|
-
const reason = unresolvedCitation(cwd, citation, { round, runId });
|
|
216
|
+
const reason = unresolvedCitation(cwd, citation, { round, runId, verdicts });
|
|
203
217
|
if (reason) return `the ${verdict.overall} verdict ${reason} — a T0 citation is re-hashed from disk, never taken on the handed word`;
|
|
204
218
|
}
|
|
205
219
|
return null;
|
package/kernel/probe/resume.mjs
CHANGED
|
@@ -65,6 +65,7 @@ import { parseBoard } from "../reduce/board.mjs";
|
|
|
65
65
|
import { intake, harnessRun, wiringMap, projectProfile, scopesDir, resultsDir, ordersDir, orientDir, activeOrder, activeScope, usecasesDir, breadboard, receipt, readReceipt, requirements, exportRunDir, lastRun, readRunId, tasksDir, gates, verdictsDir, roundBuildDir } from "../lib/paths.mjs";
|
|
66
66
|
import { evalVerdict } from "./eval.mjs";
|
|
67
67
|
import { deriveRounds } from "./rounds.mjs";
|
|
68
|
+
import { stagedWorkflowDrift, driftWarning } from "./staged.mjs";
|
|
68
69
|
import { collectRun, writeRun } from "../report/export.mjs";
|
|
69
70
|
|
|
70
71
|
/** The run-state values `references/protocol.md` (Part 4 — State) defines. A typo'd status is a rejection,
|
|
@@ -445,9 +446,11 @@ export function nextPhase(f) {
|
|
|
445
446
|
*
|
|
446
447
|
* @param {string} cwd - Project root.
|
|
447
448
|
* @param {string} slug - Feature slug.
|
|
449
|
+
* @param {object} [opts] - `pluginRoot`, when the caller knows where the plugin it is running from
|
|
450
|
+
* lives: the state then also reports whether the staged orchestrator is the installed one.
|
|
448
451
|
* @returns {object} The ResumeState record (domain.schema.json $defs/ResumeState).
|
|
449
452
|
*/
|
|
450
|
-
export function deriveResumeState(cwd, slug) {
|
|
453
|
+
export function deriveResumeState(cwd, slug, { pluginRoot = null } = {}) {
|
|
451
454
|
const hrPath = harnessRun(cwd, slug);
|
|
452
455
|
const hr = existsSync(hrPath) ? parseFrontmatter(readFileSync(hrPath, "utf8")) : {};
|
|
453
456
|
|
|
@@ -517,7 +520,22 @@ export function deriveResumeState(cwd, slug) {
|
|
|
517
520
|
.map((f) => Number(f.match(/\d+/)[0]))
|
|
518
521
|
.filter((n) => evalVerdict(cwd, slug, n).found),
|
|
519
522
|
};
|
|
520
|
-
|
|
523
|
+
// WHICH ORCHESTRATOR THIS LAUNCH WILL RUN. A run keeps the workflow copy it opened with — right,
|
|
524
|
+
// and invisible: a relaunch after an upgrade executes the old one and reports normally, so every
|
|
525
|
+
// observation is of the previous release. Reported, never enforced (see probe/staged.mjs).
|
|
526
|
+
const staged = stagedWorkflowDrift(cwd, pluginRoot, slug);
|
|
527
|
+
const warning = driftWarning(staged);
|
|
528
|
+
return {
|
|
529
|
+
...facts,
|
|
530
|
+
staged_workflow: {
|
|
531
|
+
checked: staged.checked,
|
|
532
|
+
drift: staged.drift.map((d) => d.file),
|
|
533
|
+
installed_version: staged.installed_version,
|
|
534
|
+
run_version: staged.run_version,
|
|
535
|
+
...(warning ? { warning } : {}),
|
|
536
|
+
},
|
|
537
|
+
next_phase: nextPhase(facts),
|
|
538
|
+
};
|
|
521
539
|
}
|
|
522
540
|
|
|
523
541
|
/**
|
|
@@ -933,6 +951,9 @@ export const ARGV_SPEC = {
|
|
|
933
951
|
// Not `--close`: the caller (shapeup-run.js's closeIfTerminal) hands over a RunReturn arm, never
|
|
934
952
|
// a status it decided was terminal itself — RUN_RETURN_CLOSE/closeArm above make that call.
|
|
935
953
|
"close-arm": { type: "str" },
|
|
954
|
+
// The launch hands its own plugin root so the state probe can say whether the orchestrator about
|
|
955
|
+
// to run is the installed one. Optional: a caller that does not know it gets `checked: false`.
|
|
956
|
+
"plugin-root": { type: "path" },
|
|
936
957
|
cause: { type: "str" },
|
|
937
958
|
};
|
|
938
959
|
|
|
@@ -1001,7 +1022,7 @@ export function cli(rawArgv) {
|
|
|
1001
1022
|
process.exit(r.ok ? 0 : 3);
|
|
1002
1023
|
}
|
|
1003
1024
|
|
|
1004
|
-
console.log(JSON.stringify(deriveResumeState(cwd, args.slug)));
|
|
1025
|
+
console.log(JSON.stringify(deriveResumeState(cwd, args.slug, { pluginRoot: args.pluginRoot ?? null })));
|
|
1005
1026
|
process.exit(0);
|
|
1006
1027
|
}
|
|
1007
1028
|
|
|
@@ -0,0 +1,83 @@
|
|
|
1
|
+
// staged — is the orchestrator this launch will execute the one that is installed?
|
|
2
|
+
//
|
|
3
|
+
// A run stages its own copy of the workflow scripts into the local tier when it is OPENED, and
|
|
4
|
+
// keeps that copy for the run's life. That is deliberate and right: an upgrade must not swap the
|
|
5
|
+
// orchestrator under a run in flight. The consequence is not obvious from anywhere a person about
|
|
6
|
+
// to soak an upgrade would look — relaunching an existing run after installing a new version
|
|
7
|
+
// executes the OLD orchestrator, reports normally, closes normally, and every observation made of
|
|
8
|
+
// it is an observation of the previous release. That is worse than a failed soak: it is confident
|
|
9
|
+
// evidence about the wrong artifact.
|
|
10
|
+
//
|
|
11
|
+
// So the launch compares, and says so. A warning, never a block — the run keeping its copy is the
|
|
12
|
+
// correct behaviour, and the operator is the one who decides whether this run is the one they
|
|
13
|
+
// meant to measure.
|
|
14
|
+
import { existsSync, readFileSync, readdirSync } from "node:fs";
|
|
15
|
+
import { join } from "node:path";
|
|
16
|
+
import { createHash } from "node:crypto";
|
|
17
|
+
import { workflowsStage, receipt, readReceipt } from "../lib/paths.mjs";
|
|
18
|
+
|
|
19
|
+
const sha256 = (buf) => createHash("sha256").update(buf).digest("hex");
|
|
20
|
+
|
|
21
|
+
/** A file's digest, or null when it cannot be read. */
|
|
22
|
+
function digestOf(path) {
|
|
23
|
+
try { return sha256(readFileSync(path)); } catch { return null; }
|
|
24
|
+
}
|
|
25
|
+
|
|
26
|
+
/** The version the plugin at `pluginRoot` declares, or null. */
|
|
27
|
+
export function installedPluginVersion(pluginRoot) {
|
|
28
|
+
if (!pluginRoot) return null;
|
|
29
|
+
try { return JSON.parse(readFileSync(join(pluginRoot, ".claude-plugin", "plugin.json"), "utf8")).version ?? null; }
|
|
30
|
+
catch { return null; }
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
/**
|
|
34
|
+
* How the orchestrator this launch will run compares with the one installed.
|
|
35
|
+
*
|
|
36
|
+
* @param {string} cwd - Project root.
|
|
37
|
+
* @param {(string|null)} pluginRoot - The installed plugin's root, as the launch knows it.
|
|
38
|
+
* @param {(string|null)} [slug] - The run, when the caller knows it — for the version the run opened under.
|
|
39
|
+
* @returns {object} `{checked, drift, staged_dir, plugin_root, installed_version, run_version}`.
|
|
40
|
+
* `drift` lists each script whose staged copy differs from the installed one; `checked: false`
|
|
41
|
+
* means the comparison could not be made (no plugin root, or nothing staged), which is not the
|
|
42
|
+
* same fact as "they agree" and is reported as itself.
|
|
43
|
+
*/
|
|
44
|
+
export function stagedWorkflowDrift(cwd, pluginRoot, slug = null) {
|
|
45
|
+
const stagedDir = workflowsStage(cwd);
|
|
46
|
+
const srcDir = pluginRoot ? join(pluginRoot, "skills", "tech-lead", "workflows") : null;
|
|
47
|
+
const out = {
|
|
48
|
+
checked: false,
|
|
49
|
+
drift: [],
|
|
50
|
+
staged_dir: stagedDir,
|
|
51
|
+
plugin_root: pluginRoot ?? null,
|
|
52
|
+
installed_version: installedPluginVersion(pluginRoot),
|
|
53
|
+
run_version: slug ? (readReceipt(receipt(cwd, slug))?.plugin?.version ?? null) : null,
|
|
54
|
+
};
|
|
55
|
+
if (!srcDir || !existsSync(srcDir) || !existsSync(stagedDir)) return out;
|
|
56
|
+
let names;
|
|
57
|
+
try { names = readdirSync(stagedDir).filter((f) => f.endsWith(".js")).sort(); } catch { return out; }
|
|
58
|
+
if (!names.length) return out;
|
|
59
|
+
out.checked = true;
|
|
60
|
+
for (const f of names) {
|
|
61
|
+
const staged = digestOf(join(stagedDir, f));
|
|
62
|
+
const installed = digestOf(join(srcDir, f));
|
|
63
|
+
// A script the installed plugin no longer carries is drift too — the staged copy is running
|
|
64
|
+
// something that has no counterpart in what is installed.
|
|
65
|
+
if (staged !== installed) out.drift.push({ file: f, staged_sha256: staged, installed_sha256: installed });
|
|
66
|
+
}
|
|
67
|
+
return out;
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
/**
|
|
71
|
+
* The one-line warning a launch prints, or null when there is nothing to say.
|
|
72
|
+
* @param {object} d - A {@link stagedWorkflowDrift} result.
|
|
73
|
+
* @returns {(string|null)} Operator-facing text naming both versions.
|
|
74
|
+
*/
|
|
75
|
+
export function driftWarning(d) {
|
|
76
|
+
if (!d?.checked || !d.drift.length) return null;
|
|
77
|
+
const versions = d.run_version && d.installed_version && d.run_version !== d.installed_version
|
|
78
|
+
? ` The run opened under plugin ${d.run_version}; ${d.installed_version} is installed.`
|
|
79
|
+
: d.installed_version ? ` Installed plugin: ${d.installed_version}.` : "";
|
|
80
|
+
return `the orchestrator this launch runs is NOT the installed one — ${d.drift.length} staged script(s) differ `
|
|
81
|
+
+ `(${d.drift.map((x) => x.file).join(", ")}).${versions} A run keeps the copy it opened with, by design, so this `
|
|
82
|
+
+ `launch measures the release the run was opened under. Open a NEW run to soak an upgrade.`;
|
|
83
|
+
}
|
package/kernel/report/export.mjs
CHANGED
|
@@ -130,6 +130,11 @@ function t0Row(a, runId) {
|
|
|
130
130
|
fixtures_green: a?.fixtures_green ?? null,
|
|
131
131
|
db_probe_green: a?.db_probe_green ?? null,
|
|
132
132
|
seesaw_green: a?.seesaw_green ?? null,
|
|
133
|
+
// One field a reader compares, and the block itself stays in the artifact for a human to diff:
|
|
134
|
+
// two rows with the same tree and different env digests are two machines, not a regression.
|
|
135
|
+
env_sha256: a?.env?.env_sha256 ?? null,
|
|
136
|
+
tree_head: a?.env?.tree?.head ?? null,
|
|
137
|
+
tree_dirty: a?.env?.tree?.dirty ?? null,
|
|
133
138
|
fixtures_total: fixtures.length,
|
|
134
139
|
fixtures_passed: fixtures.filter((f) => f?.pass === true).length,
|
|
135
140
|
seesaw_ran: a?.seesaw?.ran ?? null,
|
|
@@ -542,7 +542,14 @@
|
|
|
542
542
|
"items": {
|
|
543
543
|
"type": "string"
|
|
544
544
|
},
|
|
545
|
-
"description": "Explicitly untouchable paths (spec core: domain-model, UC Steps, contracts, ux-behavior)."
|
|
545
|
+
"description": "Explicitly untouchable paths (spec core: domain-model, UC Steps, contracts, ux-behavior). A build leg's order freezes the whole run trace here and carves out what it authors through `own`: a freeze that lists the channels a defect happened to name grows one defect at a time and is not a boundary."
|
|
546
|
+
},
|
|
547
|
+
"own": {
|
|
548
|
+
"type": "array",
|
|
549
|
+
"items": {
|
|
550
|
+
"type": "string"
|
|
551
|
+
},
|
|
552
|
+
"description": "Paths this order may write DESPITE a broader freeze — the compiler's grant, derived from the order's own identity and never requested by a worker. For a build leg: its own WorkResult (no sibling's — a leg that can write another's can report work nobody did), its task files, the discovery ledger, its spikes. Checked BEFORE `frozen`, and only against the contract that declared it, so another live order's exception never licenses this write."
|
|
546
553
|
}
|
|
547
554
|
}
|
|
548
555
|
},
|
|
@@ -1354,6 +1361,21 @@
|
|
|
1354
1361
|
"$ref": "#/$defs/AegisTriple"
|
|
1355
1362
|
},
|
|
1356
1363
|
"description": "Populated only on red; feeds the next order."
|
|
1364
|
+
},
|
|
1365
|
+
"env": {
|
|
1366
|
+
"type": "object",
|
|
1367
|
+
"description": "Where the fixtures ran — host, cwd, git tree, resolved toolchain paths, lockfile digests, profile-declared cache dirs, and a digest over an allowlist of environment variable VALUES (names in the clear, values never stored). `env_sha256` digests the whole block, so a reader compares one field and a human diffs the rest: two verdicts with the same tree and different digests were measured on different machines, which is a fact about portability rather than a regression. Written by kernel/verify/env.mjs, which judges nothing — no field here makes a verdict green or red.",
|
|
1368
|
+
"properties": {
|
|
1369
|
+
"schema_version": { "const": 1 },
|
|
1370
|
+
"env_sha256": { "type": "string" },
|
|
1371
|
+
"host": { "type": "object", "description": "platform, arch, os_release, node, hostname_sha256 (hashed — equality is all a reader needs)." },
|
|
1372
|
+
"cwd": { "type": "string", "description": "Absolute: a path-keyed toolchain cache is identified by this." },
|
|
1373
|
+
"tree": { "type": "object", "description": "git head, branch, and a dirty flag — which says something differs, never what." },
|
|
1374
|
+
"toolchain": { "type": "array", "description": "Each invoked binary as written, and where it resolved on this machine (null when nothing resolved it)." },
|
|
1375
|
+
"lockfiles": { "type": "array", "description": "Digest per lockfile present at the root. Says what was declared, never what is installed." },
|
|
1376
|
+
"caches": { "type": ["array", "null"], "description": "Cache dirs the project profile declares, resolved. null means the profile declared none — NOT that there are none." },
|
|
1377
|
+
"env": { "type": "object", "description": "The allowlist of variable names, and one digest over their values." }
|
|
1378
|
+
}
|
|
1357
1379
|
}
|
|
1358
1380
|
}
|
|
1359
1381
|
},
|
|
@@ -2652,6 +2674,17 @@
|
|
|
2652
2674
|
"description": "ANALYZE finished: the spec folder's usecases/ carries at least one use case that is not _index.md. WIRE reads these — one wiring-map entry per use case — which is why ANALYZE precedes WIRE in the phase chain: dispatched against an empty spec folder, WIRE escalates on every launch."
|
|
2653
2675
|
},
|
|
2654
2676
|
"has_board": { "type": "boolean", "description": "The per-machine board (tasks/TASK-*.md) holds at least one task. ANALYZE is complete only with both the committed spec tree and this; a committed tree with no board resumes at analyze, where the board-only operation regenerates it." },
|
|
2677
|
+
"staged_workflow": {
|
|
2678
|
+
"type": "object",
|
|
2679
|
+
"description": "Whether the orchestrator this launch will execute is the installed one. A run stages its own copy of the workflow scripts when it is OPENED and keeps them for its life — right for a run in flight, and silent: relaunching after an upgrade executes the OLD orchestrator, reports normally, and every observation is of the previous release. `checked: false` means the comparison could not be made (no plugin root, nothing staged), which is not the same fact as agreement. A warning, never a block: opening a new run is how an upgrade is soaked.",
|
|
2680
|
+
"properties": {
|
|
2681
|
+
"checked": { "type": "boolean" },
|
|
2682
|
+
"drift": { "type": "array", "items": { "type": "string" }, "description": "Staged scripts whose bytes differ from the installed plugin's." },
|
|
2683
|
+
"installed_version": { "type": ["string", "null"] },
|
|
2684
|
+
"run_version": { "type": ["string", "null"], "description": "The plugin version the run's receipt records — what this run actually opened under." },
|
|
2685
|
+
"warning": { "type": "string" }
|
|
2686
|
+
}
|
|
2687
|
+
},
|
|
2655
2688
|
"has_requirements": {
|
|
2656
2689
|
"type": "boolean",
|
|
2657
2690
|
"description": "The requirements registry is on disk: shapeup/<slug>/requirements.md exists. A PLAIN FACT, not a phase — the orchestrator guards its single `coverage` dispatch on this boolean, and it is deliberately absent from kernel/probe/resume.mjs's PHASE_ARTIFACT map, which doubles as nextPhase()'s ordered list: an entry there would fast-forward every run recorded before the registry existed to the registry instead of to build."
|
|
@@ -0,0 +1,197 @@
|
|
|
1
|
+
// env — the machine a T0 verdict was measured on.
|
|
2
|
+
//
|
|
3
|
+
// A T0 artifact records `exit 0`, `pass: true` and a captured tail, and the harness treats that as
|
|
4
|
+
// the fact a generator cannot fabricate. What it did not record is that the command's outcome
|
|
5
|
+
// depended on state outside the tree. Measured on a consumer whose toolchain resolves its build
|
|
6
|
+
// plugins through a cache keyed by the project's ABSOLUTE PATH: the working tree built green; a
|
|
7
|
+
// clone of the same commit at a different path failed on a registry 404; a clone with a
|
|
8
|
+
// hand-seeded cache compiled a different plugin set and produced two errors the original never
|
|
9
|
+
// saw. Three environments, three outcomes, one tree — and three byte-identical verdicts apart from
|
|
10
|
+
// their captured output.
|
|
11
|
+
//
|
|
12
|
+
// The consequence is not that such a project is badly configured; that is its own problem. It is
|
|
13
|
+
// that a green verdict is portable evidence in appearance only, and nothing in it said so. This
|
|
14
|
+
// module records enough about where a command ran that two machines disagreeing can be told from a
|
|
15
|
+
// regression. It measures and never judges: no field here makes a verdict green or red.
|
|
16
|
+
//
|
|
17
|
+
// WHAT IS DELIBERATELY NOT HERE. No wall clock — a duration is not an environment fact and invites
|
|
18
|
+
// comparing speeds across machines. No dependency-tree manifest — the lockfile digest plus the
|
|
19
|
+
// cache paths answer the decision this record exists for, and a manifest is unbounded. No raw
|
|
20
|
+
// environment values: variables are hashed, never stored, and only from a declared allowlist.
|
|
21
|
+
import { existsSync, readFileSync, statSync } from "node:fs";
|
|
22
|
+
import { join, resolve } from "node:path";
|
|
23
|
+
import { createHash } from "node:crypto";
|
|
24
|
+
import { spawnSync } from "node:child_process";
|
|
25
|
+
import { platform, arch, release, hostname } from "node:os";
|
|
26
|
+
import { splitFrontmatter } from "../lib/contract.mjs";
|
|
27
|
+
|
|
28
|
+
const sha256 = (t) => createHash("sha256").update(t).digest("hex");
|
|
29
|
+
|
|
30
|
+
/** Lockfiles worth digesting, by ecosystem. Bounded on purpose — a glob would walk the tree. */
|
|
31
|
+
export const LOCKFILES = [
|
|
32
|
+
"package-lock.json", "npm-shrinkwrap.json", "yarn.lock", "pnpm-lock.yaml", "bun.lockb",
|
|
33
|
+
"oh-package-lock.json5", "Podfile.lock", "Gemfile.lock", "Cargo.lock", "go.sum",
|
|
34
|
+
"poetry.lock", "Pipfile.lock", "composer.lock", "gradle.lockfile", "pubspec.lock",
|
|
35
|
+
];
|
|
36
|
+
|
|
37
|
+
/**
|
|
38
|
+
* Environment variables whose VALUES are hashed into the fingerprint. Names are recorded in the
|
|
39
|
+
* clear; values never are. Kept short and reviewed — a wide allowlist is how a token ends up
|
|
40
|
+
* hashed into a record somebody later publishes.
|
|
41
|
+
*/
|
|
42
|
+
export const ENV_ALLOWLIST = ["PATH", "NODE_ENV", "CI", "LANG", "TZ"];
|
|
43
|
+
|
|
44
|
+
/** What an unset variable hashes as — distinct from a variable set to the empty string. */
|
|
45
|
+
const UNSET = "<unset>";
|
|
46
|
+
|
|
47
|
+
/**
|
|
48
|
+
* The tokens a shell command actually invokes — one per `&&`/`;`/`||` segment, with `cd`, `env` and
|
|
49
|
+
* `VAR=value` prefixes stripped. Returns the token AS WRITTEN (`./scripts/t0-assemble.sh`,
|
|
50
|
+
* `/Applications/…/hvigorw`), because the resolved path is the signal a basename loses.
|
|
51
|
+
*
|
|
52
|
+
* @param {string} cmd - A shell command line.
|
|
53
|
+
* @returns {string[]} Invoked tokens, in order, without duplicates.
|
|
54
|
+
*/
|
|
55
|
+
export function cdTargets(cmd) {
|
|
56
|
+
const out = [];
|
|
57
|
+
for (const seg of String(cmd || "").split(/&&|;|\|\|/).map((s) => s.trim()).filter(Boolean)) {
|
|
58
|
+
const m = seg.match(/^cd\s+(?:"([^"]+)"|'([^']+)'|(\S+))/);
|
|
59
|
+
const dir = m && (m[1] || m[2] || m[3]);
|
|
60
|
+
if (dir && !dir.startsWith("-") && !out.includes(dir)) out.push(dir);
|
|
61
|
+
}
|
|
62
|
+
return out;
|
|
63
|
+
}
|
|
64
|
+
|
|
65
|
+
export function invokedTokens(cmd) {
|
|
66
|
+
const out = [];
|
|
67
|
+
for (const seg of String(cmd || "").split(/&&|;|\|\|/).map((s) => s.trim()).filter(Boolean)) {
|
|
68
|
+
if (seg.startsWith("cd ")) continue;
|
|
69
|
+
const tokens = seg.split(/\s+/).filter((t) => t && !/^[A-Za-z_][A-Za-z0-9_]*=/.test(t) && t !== "env" && t !== "cd");
|
|
70
|
+
if (tokens.length && !out.includes(tokens[0])) out.push(tokens[0]);
|
|
71
|
+
}
|
|
72
|
+
return out;
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
/** Where a token resolves on this machine, or null when nothing resolves it. */
|
|
76
|
+
function resolveBin(token, cwd) {
|
|
77
|
+
try {
|
|
78
|
+
const r = spawnSync(`command -v ${JSON.stringify(token)}`, { shell: true, cwd, encoding: "utf8", timeout: 10_000 });
|
|
79
|
+
const p = (r.stdout || "").trim().split("\n")[0];
|
|
80
|
+
return p || null;
|
|
81
|
+
} catch { return null; }
|
|
82
|
+
}
|
|
83
|
+
|
|
84
|
+
/** `git rev-parse HEAD` plus whether the tree is dirty. All null outside a repository. */
|
|
85
|
+
function treeState(cwd) {
|
|
86
|
+
const git = (args) => {
|
|
87
|
+
try {
|
|
88
|
+
const r = spawnSync("git", args, { cwd, encoding: "utf8", timeout: 20_000 });
|
|
89
|
+
return r.status === 0 ? (r.stdout || "").trim() : null;
|
|
90
|
+
} catch { return null; }
|
|
91
|
+
};
|
|
92
|
+
const head = git(["rev-parse", "HEAD"]);
|
|
93
|
+
if (head === null) return { head: null, dirty: null, branch: null };
|
|
94
|
+
const porcelain = git(["status", "--porcelain"]);
|
|
95
|
+
return {
|
|
96
|
+
head,
|
|
97
|
+
// A dirty flag says "something differs", never what — the honest limit of one boolean, and the
|
|
98
|
+
// reason `head` alone cannot call two measurements the same measurement.
|
|
99
|
+
dirty: porcelain === null ? null : porcelain.length > 0,
|
|
100
|
+
branch: git(["rev-parse", "--abbrev-ref", "HEAD"]),
|
|
101
|
+
};
|
|
102
|
+
}
|
|
103
|
+
|
|
104
|
+
/**
|
|
105
|
+
* Cache directories the project profile declares, resolved. A path-keyed cache is identified by its
|
|
106
|
+
* path, which is exactly the mechanism that made one tree build three ways.
|
|
107
|
+
*
|
|
108
|
+
* `null` means the profile declared nothing — NOT that there are none. "Not asked" and "none" are
|
|
109
|
+
* different facts, and collapsing them is the mistake the seesaw arm already makes elsewhere.
|
|
110
|
+
*
|
|
111
|
+
* @param {(string|null)} profilePath - `shapeup/<slug>/project-profile.md`, when the caller knows it.
|
|
112
|
+
* @returns {(object[]|null)} One entry per declared cache, or null when none is declared.
|
|
113
|
+
*/
|
|
114
|
+
export function declaredCaches(profilePath) {
|
|
115
|
+
if (!profilePath || !existsSync(profilePath)) return null;
|
|
116
|
+
let meta;
|
|
117
|
+
try { meta = splitFrontmatter(readFileSync(profilePath, "utf8")).meta || {}; } catch { return null; }
|
|
118
|
+
const raw = meta.cache_dirs ?? meta.caches ?? null;
|
|
119
|
+
const list = Array.isArray(raw)
|
|
120
|
+
? raw
|
|
121
|
+
: typeof raw === "string" && raw.trim() && raw.trim() !== "~"
|
|
122
|
+
? raw.replace(/^\[|\]$/g, "").split(",").map((s) => s.trim().replace(/^["']|["']$/g, "")).filter(Boolean)
|
|
123
|
+
: null;
|
|
124
|
+
if (!list || !list.length) return null;
|
|
125
|
+
return list.map((p) => {
|
|
126
|
+
const path = p.startsWith("~") ? join(process.env.HOME || "", p.slice(1)) : p;
|
|
127
|
+
let mtime = null;
|
|
128
|
+
try { mtime = statSync(path).mtime.toISOString(); } catch { /* absent is a fact, not an error */ }
|
|
129
|
+
return { declared: p, path, exists: existsSync(path), mtime };
|
|
130
|
+
});
|
|
131
|
+
}
|
|
132
|
+
|
|
133
|
+
/**
|
|
134
|
+
* Stable JSON: object keys sorted at every depth, so a digest does not depend on insertion order.
|
|
135
|
+
* @param {*} value - Anything JSON-representable.
|
|
136
|
+
* @returns {string} The canonical form.
|
|
137
|
+
*/
|
|
138
|
+
export function canonical(value) {
|
|
139
|
+
if (Array.isArray(value)) return `[${value.map(canonical).join(",")}]`;
|
|
140
|
+
if (value && typeof value === "object") {
|
|
141
|
+
return `{${Object.keys(value).sort().map((k) => `${JSON.stringify(k)}:${canonical(value[k])}`).join(",")}}`;
|
|
142
|
+
}
|
|
143
|
+
return JSON.stringify(value ?? null);
|
|
144
|
+
}
|
|
145
|
+
|
|
146
|
+
/**
|
|
147
|
+
* The environment block a T0 verdict carries.
|
|
148
|
+
*
|
|
149
|
+
* @param {string} rawCwd - The directory the fixtures ran in; resolved to an absolute path here.
|
|
150
|
+
* @param {object} [opts] - `commands` (the fixture command lines) and `profilePath`.
|
|
151
|
+
* @returns {object} `{schema_version, host, cwd, tree, toolchain, lockfiles, caches, env, env_sha256}`.
|
|
152
|
+
* `env_sha256` digests the block itself, so a reader compares one field and a human diffs the
|
|
153
|
+
* rest. Without the digest every consumer re-implements the comparison and they disagree, which
|
|
154
|
+
* is three key spaces for one id waiting to happen on a new record.
|
|
155
|
+
*/
|
|
156
|
+
export function environmentFingerprint(rawCwd, { commands = [], profilePath = null } = {}) {
|
|
157
|
+
// ABSOLUTE, always. A caller that ran with `--cwd .` would otherwise record "." as the place —
|
|
158
|
+
// and the place is the whole point: the cache that made one tree build three ways is keyed by
|
|
159
|
+
// the project's absolute path.
|
|
160
|
+
const cwd = resolve(rawCwd || process.cwd());
|
|
161
|
+
const tokens = [...new Set(commands.flatMap((c) => invokedTokens(c)))];
|
|
162
|
+
const block = {
|
|
163
|
+
schema_version: 1,
|
|
164
|
+
host: {
|
|
165
|
+
platform: platform(),
|
|
166
|
+
arch: arch(),
|
|
167
|
+
os_release: release(),
|
|
168
|
+
node: process.version,
|
|
169
|
+
// Hashed: equality is all a reader needs, and a hostname names a person's laptop.
|
|
170
|
+
hostname_sha256: sha256(hostname()),
|
|
171
|
+
},
|
|
172
|
+
cwd,
|
|
173
|
+
tree: treeState(cwd),
|
|
174
|
+
toolchain: tokens.map((bin) => ({ bin, path: resolveBin(bin, cwd) })),
|
|
175
|
+
// The root AND wherever the commands actually run. Measured on a consumer whose fixtures are
|
|
176
|
+
// `cd app && …`: the lockfile that decides what the build resolves lives in `app/`, and a scan
|
|
177
|
+
// of the project root alone recorded an empty list beside a build whose dependencies were the
|
|
178
|
+
// whole question.
|
|
179
|
+
lockfiles: [...new Set(["", ...commands.flatMap((c) => cdTargets(c))])]
|
|
180
|
+
.flatMap((sub) => LOCKFILES
|
|
181
|
+
.map((f) => (sub ? `${sub.replace(/\/+$/, "")}/${f}` : f))
|
|
182
|
+
.filter((rel) => existsSync(join(cwd, rel)))
|
|
183
|
+
.map((rel) => {
|
|
184
|
+
try { return { file: rel, sha256: sha256(readFileSync(join(cwd, rel))) }; }
|
|
185
|
+
catch { return { file: rel, sha256: null }; }
|
|
186
|
+
}))
|
|
187
|
+
.filter((l, i, all) => all.findIndex((x) => x.file === l.file) === i),
|
|
188
|
+
caches: declaredCaches(profilePath),
|
|
189
|
+
env: {
|
|
190
|
+
allowlist: ENV_ALLOWLIST,
|
|
191
|
+
// One digest over the allowlisted names AND values: a PATH that changed shows up, and no
|
|
192
|
+
// value is stored.
|
|
193
|
+
sha256: sha256(ENV_ALLOWLIST.map((k) => `${k}=${process.env[k] ?? UNSET}`).join("\n")),
|
|
194
|
+
},
|
|
195
|
+
};
|
|
196
|
+
return { ...block, env_sha256: sha256(canonical(block)) };
|
|
197
|
+
}
|
package/kernel/verify/spec.mjs
CHANGED
|
@@ -422,6 +422,9 @@ export function lintScopeAnchors({ scopes, specDir: specRoot, reqIds = null, tas
|
|
|
422
422
|
return findings;
|
|
423
423
|
}
|
|
424
424
|
|
|
425
|
+
/** Registry sources that name the pitch's own out-of-scope section, in the spellings pitches use. */
|
|
426
|
+
const NOGO_SOURCE = /\bno[-\s]?gos?\b|\bnon[-\s]?goals?\b|\bout[-\s]of[-\s]scope\b|\bwill not build\b/i;
|
|
427
|
+
|
|
425
428
|
/**
|
|
426
429
|
* REQ-UNCOVERED — a live requirement that nothing in the plan reaches.
|
|
427
430
|
*
|
|
@@ -451,6 +454,57 @@ export function lintScopeAnchors({ scopes, specDir: specRoot, reqIds = null, tas
|
|
|
451
454
|
* @returns {Array<{rule:string, level:("red"|"warn"), scope:string, detail:string}>} One red per
|
|
452
455
|
* uncovered live requirement; [] when every one is graded, claimed or cut.
|
|
453
456
|
*/
|
|
457
|
+
/**
|
|
458
|
+
* REQ-NARRATED — a committed spec file stating the requirement-coverage verdict as fact.
|
|
459
|
+
*
|
|
460
|
+
* The Health Dashboard's `Coverage` row is about USE CASES and tasks, derived by inverting each
|
|
461
|
+
* task's `use_case_refs` over the local board. Measured on a consumer, a worker filled its Signal
|
|
462
|
+
* cell with a different claim entirely — *"every registered non-CUT REQ-id (REQ-1 … REQ-7) reaches
|
|
463
|
+
* an AC carrying `(covers: REQ-…)`"*, with a 🟢 beside it — while a grep for `covers:` across the
|
|
464
|
+
* whole spec folder returned that sentence and nothing else. Not one acceptance criterion carried
|
|
465
|
+
* the clause, and the run's own derived report said `0/11 PASS`.
|
|
466
|
+
*
|
|
467
|
+
* `AGENTS.md` names the invariant this breaks: the requirements matrix is a projection, never a
|
|
468
|
+
* verdict, derived from files for one named run and never narrated. The rule is the narrow,
|
|
469
|
+
* checkable form of it — a dashboard Coverage row in a committed file may not name a REQ id — and
|
|
470
|
+
* it cannot fire on the legitimate signal, which counts use cases and tasks.
|
|
471
|
+
*
|
|
472
|
+
* @param {{cwd:string, slug:string}} opts - Working root and feature slug.
|
|
473
|
+
* @returns {object[]} Findings, one per offending line.
|
|
474
|
+
*/
|
|
475
|
+
export function lintNarratedCoverage({ cwd, slug }) {
|
|
476
|
+
const findings = [];
|
|
477
|
+
const dir = join(sharedRoot(cwd, slug), "spec");
|
|
478
|
+
let files;
|
|
479
|
+
try { files = readdirSync(dir).filter((f) => f.endsWith(".md")); } catch { return findings; }
|
|
480
|
+
for (const f of files) {
|
|
481
|
+
let lines;
|
|
482
|
+
try { lines = readFileSync(join(dir, f), "utf8").split(/\r?\n/); } catch { continue; }
|
|
483
|
+
lines.forEach((line, i) => {
|
|
484
|
+
if (!/^\|\s*Coverage\s*\|/i.test(line.trim())) return;
|
|
485
|
+
const named = [...line.matchAll(/\bREQ-\d+/g)].map((m) => m[0]);
|
|
486
|
+
if (!named.length) return;
|
|
487
|
+
findings.push({ rule: "REQ-NARRATED", level: "red", scope: `${f}:${i + 1}`, detail:
|
|
488
|
+
`${f}:${i + 1} states the requirement-coverage verdict in a committed file, naming ${named.slice(0, 3).join(", ")}` +
|
|
489
|
+
`${named.length > 3 ? ` (+${named.length - 3})` : ""}. That row is the UC × Task indicator; the ` +
|
|
490
|
+
"REQ → AC → criterion → verdict state is a projection derived per run (probe requirements), " +
|
|
491
|
+
"never a claim a committed artifact may make — a reader who checks the file finds corroboration " +
|
|
492
|
+
"for something no run measured. Say what the use cases and tasks show, and leave the requirement " +
|
|
493
|
+
"matrix to the run that derives it." });
|
|
494
|
+
});
|
|
495
|
+
}
|
|
496
|
+
return findings;
|
|
497
|
+
}
|
|
498
|
+
|
|
499
|
+
/**
|
|
500
|
+
* REQ-UNCOVERED and REQ-NOGO — the registry's two ways of being wrong about what ships.
|
|
501
|
+
*
|
|
502
|
+
* @param {{clauses:object[], board:object[], scopes:object[]}} opts - The parsed registry, the
|
|
503
|
+
* board `readBoard` produced (its acceptance criteria carry the covers clauses), and the scope
|
|
504
|
+
* contracts.
|
|
505
|
+
* @returns {object[]} Findings, most specific first: a no-go registered as covered is reported as
|
|
506
|
+
* itself rather than as the coverage gap it inevitably becomes.
|
|
507
|
+
*/
|
|
454
508
|
export function lintRequirementCoverage({ clauses = [], board = [], scopes = [] }) {
|
|
455
509
|
const findings = [];
|
|
456
510
|
const graded = coveredReqIds(board);
|
|
@@ -459,6 +513,21 @@ export function lintRequirementCoverage({ clauses = [], board = [], scopes = []
|
|
|
459
513
|
const claimed = new Set();
|
|
460
514
|
for (const s of scopes) for (const r of s.covers || []) claimed.add(reqId(r).toUpperCase());
|
|
461
515
|
for (const c of clauses) {
|
|
516
|
+
// A NO-GO IS A CONSTRAINT, NOT A DELIVERABLE, and marking one `covered` asserts something that
|
|
517
|
+
// cannot be true: nothing grades "do not build a settings screen". Measured on a consumer — a
|
|
518
|
+
// coverage dispatch lifted seven clauses out of the pitch's No-gos section, registered each as
|
|
519
|
+
// covered, and L1b then refused the run with seven REQ-UNCOVERED findings, correctly and
|
|
520
|
+
// unavoidably. Reported here as itself, so the operator reads one cause instead of seven
|
|
521
|
+
// symptoms, and named before REQ-UNCOVERED can fire on the same row.
|
|
522
|
+
if (c.status === "covered" && NOGO_SOURCE.test(c.source || "")) {
|
|
523
|
+
findings.push({ rule: "REQ-NOGO", level: "red", scope: c.id, detail:
|
|
524
|
+
`${c.id} ← ${c.source} registers a NO-GO as a covered requirement — "${(c.clause || "").slice(0, 60)}". ` +
|
|
525
|
+
"A no-go is a constraint the shape deliberately does not build, so no acceptance criterion can " +
|
|
526
|
+
"grade it and nothing downstream can ever turn it green. Mark it CUT (PO-approved) in " +
|
|
527
|
+
"requirements.md — the family that already means deliberately-not-built — or drop the row and give " +
|
|
528
|
+
"the breach a Test Surface row (TS-NOGO-NN) instead, which is the channel that does grade one." });
|
|
529
|
+
continue;
|
|
530
|
+
}
|
|
462
531
|
if (c.status !== "covered") continue; // CUT (PO-approved) — an answer on the record, not a gap
|
|
463
532
|
const id = c.id.toUpperCase();
|
|
464
533
|
if (graded.has(c.id) || claimed.has(id)) continue;
|
|
@@ -851,6 +920,7 @@ export function lint({ cwd, slug }) {
|
|
|
851
920
|
// `readBoard`, not the `tasks` above: only the compile-order parser carries acceptance_criteria.
|
|
852
921
|
...lintRequirementCoverage({ clauses: reqClauses, board: readBoard(cwd, slug), scopes }),
|
|
853
922
|
...lintCommittedTier({ cwd, slug }),
|
|
923
|
+
...lintNarratedCoverage({ cwd, slug }),
|
|
854
924
|
...lintStructure({ specDir: specRoot, tasks, intakeContent }),
|
|
855
925
|
...(() => {
|
|
856
926
|
const bbText = runBreadboard(cwd, slug, intakeContent);
|
package/kernel/verify/t0.mjs
CHANGED
|
@@ -40,10 +40,11 @@ import { join, dirname } from "node:path";
|
|
|
40
40
|
import { spawnSync } from "node:child_process";
|
|
41
41
|
import { createHash } from "node:crypto";
|
|
42
42
|
import { digest } from "../probe/digest.mjs";
|
|
43
|
+
import { environmentFingerprint } from "./env.mjs";
|
|
43
44
|
import { runArgs } from "../lib/argv.mjs";
|
|
44
45
|
import { snapshot, restore, keptRef } from "./ratchet-tree.mjs";
|
|
45
46
|
import { readContract, SCOPE_CONTRACT } from "../lib/contract.mjs";
|
|
46
|
-
import { runIdFromRoot, localRoot, SHARED } from "../lib/paths.mjs";
|
|
47
|
+
import { runIdFromRoot, localRoot, SHARED, projectProfile } from "../lib/paths.mjs";
|
|
47
48
|
|
|
48
49
|
/**
|
|
49
50
|
* The feature slug a scope contract belongs to, from its path.
|
|
@@ -193,13 +194,17 @@ export function runDbProbe(dbProbeCmd, cwd) {
|
|
|
193
194
|
*/
|
|
194
195
|
export function seesawCheck(registryPath, cwd) {
|
|
195
196
|
if (!registryPath || !existsSync(registryPath)) {
|
|
196
|
-
|
|
197
|
+
// `pass: null`, not `pass: true`: a check that did not run has no result, and recording one as
|
|
198
|
+
// clean is how "not asked" came to read as "nothing regressed". The verdict below still treats
|
|
199
|
+
// an absent seesaw as non-blocking — that part is deliberate while the arm is unwired — but the
|
|
200
|
+
// artifact now says which of the two it was.
|
|
201
|
+
return { ran: false, pass: null, scopes_checked: [], failing: [] };
|
|
197
202
|
}
|
|
198
203
|
let registry;
|
|
199
204
|
try {
|
|
200
205
|
registry = JSON.parse(readFileSync(registryPath, "utf8"));
|
|
201
206
|
} catch {
|
|
202
|
-
return { ran: false, pass:
|
|
207
|
+
return { ran: false, pass: null, scopes_checked: [], failing: [], error: "registry unparsable" };
|
|
203
208
|
}
|
|
204
209
|
const scopes = registry.scopes || [];
|
|
205
210
|
const failing = [];
|
|
@@ -221,7 +226,12 @@ export function seesawCheck(registryPath, cwd) {
|
|
|
221
226
|
export function computeVerdict({ fixtures, dbProbe, seesaw }) {
|
|
222
227
|
const fixturesGreen = fixtures.pass;
|
|
223
228
|
const dbGreen = dbProbe === null || dbProbe.pass;
|
|
224
|
-
|
|
229
|
+
// A seesaw that did not run does not hold the verdict red — the arm is declared and unwired, and
|
|
230
|
+
// blocking every build on it would be a different defect. It does not make it green either: the
|
|
231
|
+
// hill requires `ran && pass` before a scope may reach FINISHED, and `seesaw_green` here means
|
|
232
|
+
// "nothing this check found is wrong", which is true of a check that found nothing because it
|
|
233
|
+
// never looked.
|
|
234
|
+
const seesawGreen = seesaw.ran ? seesaw.pass === true : true;
|
|
225
235
|
return {
|
|
226
236
|
fixtures_green: fixturesGreen,
|
|
227
237
|
db_probe_green: dbGreen,
|
|
@@ -576,6 +586,14 @@ export async function cli(rawArgv) {
|
|
|
576
586
|
const { path, sha256: hash, trial } = writeArtifact(outDir, round, attempt, {
|
|
577
587
|
...(runId ? { run_id: runId } : {}),
|
|
578
588
|
scope_id: contract.scope_id,
|
|
589
|
+
// WHERE IT RAN, beside what it measured. A verdict that records only the tree is portable
|
|
590
|
+
// evidence in appearance only: the same commit built three ways on three machines because the
|
|
591
|
+
// toolchain resolved through a path-keyed cache. This block does not judge — it is what lets a
|
|
592
|
+
// disagreeing re-run be told from a regression (see verify/env.mjs).
|
|
593
|
+
env: environmentFingerprint(cwd, {
|
|
594
|
+
commands: [...fixtures.results.map((r) => r.cmd), ...(dbProbe?.cmd ? [dbProbe.cmd] : [])],
|
|
595
|
+
profilePath: projectProfile(cwd, slugFromContractPath(contractPath)),
|
|
596
|
+
}),
|
|
579
597
|
// The evidence, not just the score — see `commandEvidence` for what the three-field record
|
|
580
598
|
// could not tell apart, and why `exit` still reads the way it always did.
|
|
581
599
|
fixtures: fixtures.results.map((r) => commandEvidence(r)),
|
package/package.json
CHANGED
|
@@ -26,6 +26,11 @@ depends_on:
|
|
|
26
26
|
|
|
27
27
|
| Indicator | Status | Signal |
|
|
28
28
|
|-----------|--------|--------|
|
|
29
|
+
<!-- Coverage here is USE CASES × TASKS, derived by inverting each task's use_case_refs over the
|
|
30
|
+
local board. It is NOT requirement coverage: whether every REQ-id reaches an acceptance
|
|
31
|
+
criterion that a judge graded is a projection derived per run (`probe requirements`), and a
|
|
32
|
+
committed file that states it is corroborating something no run measured. Name REQ ids in this
|
|
33
|
+
row and spec-lint reds it (REQ-NARRATED). Count use cases and tasks; say nothing about REQ. -->
|
|
29
34
|
| Coverage | COVERAGE_STATUS | COVERAGE_SIGNAL |
|
|
30
35
|
| Risk | RISK_STATUS | RISK_SIGNAL |
|
|
31
36
|
| Dependency | DEPENDENCY_STATUS | DEPENDENCY_SIGNAL |
|
|
@@ -414,6 +414,16 @@ const RESUME = {
|
|
|
414
414
|
has_orient_artifacts: { type: "boolean" },
|
|
415
415
|
has_spec_tree: { type: "boolean" },
|
|
416
416
|
has_board: { type: "boolean" },
|
|
417
|
+
staged_workflow: {
|
|
418
|
+
type: "object",
|
|
419
|
+
properties: {
|
|
420
|
+
checked: { type: "boolean" },
|
|
421
|
+
drift: { type: "array", items: { type: "string" } },
|
|
422
|
+
installed_version: nullable("string"),
|
|
423
|
+
run_version: nullable("string"),
|
|
424
|
+
warning: { type: "string" },
|
|
425
|
+
},
|
|
426
|
+
},
|
|
417
427
|
// The requirements registry — a fact, not a phase. See the COVERAGE block below for why it is
|
|
418
428
|
// guarded on this bare boolean and never asked about through `probe resume --require`.
|
|
419
429
|
has_requirements: { type: "boolean" },
|
|
@@ -1232,7 +1242,14 @@ if (launchRecordAbort) return await withWarnings(launchRecordAbort);
|
|
|
1232
1242
|
|
|
1233
1243
|
phase("Orient");
|
|
1234
1244
|
|
|
1235
|
-
const rs = await query(`probe resume --slug ${slug}`, RESUME, "Orient", "resume-state");
|
|
1245
|
+
const rs = await query(`probe resume --slug ${slug} --plugin-root "${args.pluginRoot}"`, RESUME, "Orient", "resume-state");
|
|
1246
|
+
// WHICH ORCHESTRATOR IS RUNNING. A run keeps the workflow copy it opened with — correct for a run
|
|
1247
|
+
// in flight, and silent: a relaunch after an upgrade executes the old script, reports normally and
|
|
1248
|
+
// closes normally, so every observation is of the previous release. Said out loud, never enforced.
|
|
1249
|
+
if (rs.staged_workflow?.warning) {
|
|
1250
|
+
log(`RUN STATE — ${rs.staged_workflow.warning}`);
|
|
1251
|
+
stateWarnings.push(rs.staged_workflow.warning);
|
|
1252
|
+
}
|
|
1236
1253
|
// A probe that produced nothing is not an EMPTY run — it is an unknown one. Treating it as empty
|
|
1237
1254
|
// would re-dispatch every phase from the top, over a run that may be in progress.
|
|
1238
1255
|
if (!rs) {
|