shapeup-sdlc 3.1.1 → 3.1.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude-plugin/plugin.json +1 -1
- package/AGENTS.md +1 -1
- package/kernel/compile.mjs +51 -2
- package/kernel/probe/eval.mjs +83 -12
- package/kernel/probe/resume.mjs +8 -1
- package/kernel/probe/t0.mjs +26 -3
- package/kernel/reduce/ingest.mjs +15 -0
- package/package.json +1 -1
- package/skills/spec-evaluator/SKILL.md +1 -1
- package/skills/tech-lead/references/protocol.md +5 -1
- package/skills/tech-lead/schemas/domain.schema.json +1 -1
- package/skills/tech-lead/workflows/shapeup-run.js +14 -2
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "shapeup-sdlc-plugin",
|
|
3
3
|
"displayName": "ShapeUp SDLC Plugin",
|
|
4
|
-
"version": "3.1.
|
|
4
|
+
"version": "3.1.2",
|
|
5
5
|
"description": "Shape Up SDLC harness for Claude Code: shaping, intake, orient, scope-mapping, building (T0-verified, sandboxed, scope-contracted), evaluation and QA skills orchestrated by a tech-lead.",
|
|
6
6
|
"author": {
|
|
7
7
|
"name": "Liberty Nguyen",
|
package/AGENTS.md
CHANGED
|
@@ -58,7 +58,7 @@ Everything discovered funnels into `.shapeup/<slug>/discovery/ledger.md` (Orient
|
|
|
58
58
|
- **Ledger = single source of truth** — every discovery flow writes only its own section.
|
|
59
59
|
- **QA is a level-up, not a gate** — `--no-qa` skips it; circuit breaker outranks the Hunter.
|
|
60
60
|
- **Role separation** — Evaluator grades, task-executor fixes, QA discovers.
|
|
61
|
-
- **Hill phase is mechanical ✦** — derived only from T0/T1/seesaw artifacts, never self-reported; the evaluator cites a T0 artifact it re-hashes itself.
|
|
61
|
+
- **Hill phase is mechanical ✦** — derived only from T0/T1/seesaw artifacts, never self-reported; the evaluator cites a T0 artifact it re-hashes itself, from the list its order carries. A scoped verdict citing none is refused: its round stays open and is evaluated again, never advanced.
|
|
62
62
|
- **Envelope port (v1.0)** — every dispatch is WorkOrder in / WorkResult out; shared state has exactly one writer (the ingest step); malformed envelopes are hook-denied. Workers: stateless, craft-only, pipeline-blind.
|
|
63
63
|
|
|
64
64
|
## Setup & Execution
|
package/kernel/compile.mjs
CHANGED
|
@@ -36,10 +36,11 @@ import { readRunId } from "./lib/paths.mjs";
|
|
|
36
36
|
// --spec-overridden directory, and the import is the convention-derived default.
|
|
37
37
|
import {
|
|
38
38
|
tasksDir, specDir as defaultSpecDir, roundLedger, trials, verdictsDir, ordersDir,
|
|
39
|
-
relShared, globLocal, globShared, relKnowledgeBase, resultsDir, scopesDir,
|
|
39
|
+
relShared, relLocal, globLocal, globShared, relKnowledgeBase, resultsDir, scopesDir,
|
|
40
40
|
} from "./lib/paths.mjs";
|
|
41
|
-
import { readContract, tasksForScope, SCOPE_CONTRACT } from "./lib/contract.mjs";
|
|
41
|
+
import { readContract, readAllContracts, tasksForScope, SCOPE_CONTRACT } from "./lib/contract.mjs";
|
|
42
42
|
import { writeActiveOrder } from "./probe/resume.mjs";
|
|
43
|
+
import { greenVerdict } from "./probe/t0.mjs";
|
|
43
44
|
// The SAME matcher the sandbox hook enforces with. "Is this cited file inside this scope's
|
|
44
45
|
// substrate" has to mean exactly what the guard means, or a bug is addressed to a scope that is
|
|
45
46
|
// then denied the write that fixes it.
|
|
@@ -482,6 +483,43 @@ export function scopeSubstrates(cwd, slug) {
|
|
|
482
483
|
return out;
|
|
483
484
|
}
|
|
484
485
|
|
|
486
|
+
// --- the T0 artifacts the judge must cite ----------------------------------------------------
|
|
487
|
+
//
|
|
488
|
+
// WHY THE KERNEL DERIVES THEM. spec-evaluator treats a scoped spec whose order lists no T0 artifact
|
|
489
|
+
// as NOT gradeable and returns `failed` without grading a criterion. The precondition is right —
|
|
490
|
+
// its verdict must cite a T0 artifact it re-hashed itself — and nothing met it: the list was once
|
|
491
|
+
// assembled by an orchestrator courier from the paths each scope reported, and was lost when the
|
|
492
|
+
// orchestrator became a workflow script that passes only `{dimensions, run_cmd, round}`. Evaluators
|
|
493
|
+
// that went looking on disk graded anyway; one that followed its contract refused, and the run
|
|
494
|
+
// aborted at L3 over a round whose every scope was green.
|
|
495
|
+
//
|
|
496
|
+
// Derived here for the reason `bugs` is: this is the one line every lane compiles through, and the
|
|
497
|
+
// evidence is already on disk. A caller could not rebuild the list from filenames in any case —
|
|
498
|
+
// verdict files are addressed by round, attempt and trial, never by scope, so the scope lives only
|
|
499
|
+
// inside each body, which is what `probe t0` reads.
|
|
500
|
+
|
|
501
|
+
/**
|
|
502
|
+
* The green T0 verdict each scope contract holds for a round — an evaluate order's `t0_artifacts`.
|
|
503
|
+
*
|
|
504
|
+
* @param {string} cwd - Project root.
|
|
505
|
+
* @param {string} slug - Feature slug.
|
|
506
|
+
* @param {number} [round] - The round being evaluated. Omitted, each scope's newest green verdict
|
|
507
|
+
* of any round — a standalone evaluation has no round.
|
|
508
|
+
* @returns {{artifacts: string[], missing: string[]}} Repo-relative verdict paths in scope-id
|
|
509
|
+
* order, one per scope that has one; and the scopes that have none. Both empty on an unscoped spec.
|
|
510
|
+
*/
|
|
511
|
+
export function t0ArtifactsFor(cwd, slug, round) {
|
|
512
|
+
const artifacts = [];
|
|
513
|
+
const missing = [];
|
|
514
|
+
for (const { contract, id } of readAllContracts(scopesDir(cwd, slug))) {
|
|
515
|
+
const scopeId = contract?.scope_id || id;
|
|
516
|
+
const { green, path } = greenVerdict(cwd, slug, scopeId, round);
|
|
517
|
+
if (green) artifacts.push(relLocal(slug, "t0", "verdicts", basename(path)));
|
|
518
|
+
else missing.push(scopeId);
|
|
519
|
+
}
|
|
520
|
+
return { artifacts, missing };
|
|
521
|
+
}
|
|
522
|
+
|
|
485
523
|
/**
|
|
486
524
|
* Assemble a WorkOrder envelope. Pure given its inputs — the CLI wrapper does the disk reads.
|
|
487
525
|
* @param {object} opts - The order inputs (destructured):
|
|
@@ -743,6 +781,17 @@ export async function cli(rawArgv) {
|
|
|
743
781
|
let payloadExtra = flag("payload") || {};
|
|
744
782
|
if (specDir && !payloadExtra.spec_folder) payloadExtra.spec_folder = specDir;
|
|
745
783
|
if (!payloadExtra.feature) payloadExtra.feature = slug;
|
|
784
|
+
// The judge's citations, for every lane (see t0ArtifactsFor). An explicit `--payload` list still
|
|
785
|
+
// wins, as it does for `bugs`: an operator naming the evidence outranks the derivation.
|
|
786
|
+
if (operation === "evaluate" && payloadExtra.t0_artifacts === undefined) {
|
|
787
|
+
const { artifacts, missing } = t0ArtifactsFor(cwd, slug, round);
|
|
788
|
+
if (artifacts.length) payloadExtra.t0_artifacts = artifacts;
|
|
789
|
+
// On stderr, never stdout: stdout is the order path the caller consumes.
|
|
790
|
+
if (missing.length) {
|
|
791
|
+
console.error(`compile-order: warning — no green T0 verdict${round ? ` in round ${round}` : ""} for ` +
|
|
792
|
+
`${missing.join(", ")}; the evaluator has nothing to cite for ${missing.length === 1 ? "that scope" : "those scopes"}`);
|
|
793
|
+
}
|
|
794
|
+
}
|
|
746
795
|
|
|
747
796
|
const order = compileOrder({
|
|
748
797
|
slug, worker, operation, round, attempt, scope, tasks, decisions, digestedErrors, trialHistory, bugs,
|
package/kernel/probe/eval.mjs
CHANGED
|
@@ -1,8 +1,10 @@
|
|
|
1
1
|
// probe eval — "what did round N's EVAL WorkResult actually say?"
|
|
2
2
|
//
|
|
3
3
|
// CONTRACT. A bounded, read-only query over the evaluate WorkResult ingest already wrote. Prints
|
|
4
|
-
// `{ok, overall, bug_count, report_path}` on stdout; exits 0 when
|
|
5
|
-
//
|
|
4
|
+
// `{ok, overall, bug_count, report_path, round, status, reason}` on stdout; exits 0 when the round
|
|
5
|
+
// holds a verdict the run may act on, 1 when it does not — nothing ran, ingest hasn't landed, the
|
|
6
|
+
// evaluator refused the round, or the verdict is structurally invalid; `reason` says which — and 2
|
|
7
|
+
// on a bad argv. Writes nothing.
|
|
6
8
|
//
|
|
7
9
|
// WHY THIS EXISTS. `shapeup-run.js` cannot read a file itself (a Workflow script has no filesystem
|
|
8
10
|
// of its own — see this repo's own note on why it may not call `Date.now()`), so every fact it
|
|
@@ -23,11 +25,65 @@
|
|
|
23
25
|
// shared state; the `.md` report is prose for a human. Reading the prose to re-derive a verdict a
|
|
24
26
|
// schema already carries structurally is the paraphrase channel this repo's hooks exist to close
|
|
25
27
|
// everywhere else.
|
|
28
|
+
//
|
|
29
|
+
// WHY "NO VERDICT" CARRIES A REASON. An evaluator that refuses a round — a structural precondition
|
|
30
|
+
// it cannot meet — still writes `evaluate-r<N>.json`, with `status: failed`, no verdict, and the
|
|
31
|
+
// cause as its first deviation. A bare `ok: false` reached the operator as a sub-agent that died
|
|
32
|
+
// after retries, while the one sentence naming the actual cause sat in a file nobody was pointed at.
|
|
26
33
|
|
|
27
|
-
import { existsSync, readFileSync } from "node:fs";
|
|
34
|
+
import { existsSync, readFileSync, readdirSync } from "node:fs";
|
|
28
35
|
import { join, resolve } from "node:path";
|
|
29
36
|
import { runArgs } from "../lib/argv.mjs";
|
|
30
|
-
import { resultsDir } from "../lib/paths.mjs";
|
|
37
|
+
import { resultsDir, scopesDir } from "../lib/paths.mjs";
|
|
38
|
+
|
|
39
|
+
/** Longest `reason` reported. A deviation is prose written by a worker and can run to paragraphs. */
|
|
40
|
+
const REASON_MAX = 400;
|
|
41
|
+
|
|
42
|
+
/**
|
|
43
|
+
* Bound a reason to {@link REASON_MAX} characters.
|
|
44
|
+
* @param {string} s - The reason.
|
|
45
|
+
* @returns {string} `s`, or its first REASON_MAX − 1 characters and an ellipsis.
|
|
46
|
+
*/
|
|
47
|
+
const clip = (s) => (s.length > REASON_MAX ? `${s.slice(0, REASON_MAX - 1)}…` : s);
|
|
48
|
+
|
|
49
|
+
/**
|
|
50
|
+
* Whether a feature's spec is SCOPED — has scope contracts, the case in which every verdict must
|
|
51
|
+
* cite the T0 artifacts it re-hashed.
|
|
52
|
+
*
|
|
53
|
+
* @param {string} cwd - Project root.
|
|
54
|
+
* @param {string} slug - Feature slug.
|
|
55
|
+
* @returns {boolean} True when `scopes/` holds at least one contract (`.md`, or a legacy `.json`).
|
|
56
|
+
*/
|
|
57
|
+
export function isScoped(cwd, slug) {
|
|
58
|
+
try { return readdirSync(scopesDir(cwd, slug)).some((f) => /\.(md|json)$/.test(f)); }
|
|
59
|
+
catch { return false; }
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
/**
|
|
63
|
+
* Why a verdict cannot stand as its round's judgement on T0 grounds, or null when it can.
|
|
64
|
+
*
|
|
65
|
+
* A PASS or FAIL on a scoped spec that cites no T0 artifact is structurally invalid — the
|
|
66
|
+
* evaluator's own contract says so, because T0 is the machine fact a generator cannot fabricate.
|
|
67
|
+
* That rule used to live only in the contract, so a verdict citing nothing was ingested, ledgered
|
|
68
|
+
* and branched on like any other. It is checked here so the round loop, the resume derivation, the
|
|
69
|
+
* hill and ingest all refuse the same verdict for the same reason.
|
|
70
|
+
*
|
|
71
|
+
* PRESENCE, NOT HASHES. The evaluator re-hashes what it cites; a slip transcribing a digest is not
|
|
72
|
+
* evidence the verdict is wrong, and refusing a round over one would cost a whole re-evaluation.
|
|
73
|
+
*
|
|
74
|
+
* @param {string} cwd - Project root.
|
|
75
|
+
* @param {string} slug - Feature slug.
|
|
76
|
+
* @param {object} verdict - The WorkResult's `verdict` block.
|
|
77
|
+
* @returns {(string|null)} The problem, phrased for an operator; null for a cited verdict, an
|
|
78
|
+
* unscoped spec, or a block with no PASS/FAIL in it (there is no judgement to invalidate).
|
|
79
|
+
*/
|
|
80
|
+
export function citationProblem(cwd, slug, verdict) {
|
|
81
|
+
if (verdict?.overall !== "PASS" && verdict?.overall !== "FAIL") return null;
|
|
82
|
+
if (Array.isArray(verdict.t0_citations) && verdict.t0_citations.length) return null;
|
|
83
|
+
if (!isScoped(cwd, slug)) return null;
|
|
84
|
+
return `the ${verdict.overall} verdict cites no T0 artifact, and a verdict on a scoped spec must ` +
|
|
85
|
+
"cite the T0 verdict it re-hashed (the order lists them under payload.t0_artifacts)";
|
|
86
|
+
}
|
|
31
87
|
|
|
32
88
|
/**
|
|
33
89
|
* Read one round's EVAL verdict straight from the WorkResult `reduce ingest` wrote.
|
|
@@ -35,20 +91,35 @@ import { resultsDir } from "../lib/paths.mjs";
|
|
|
35
91
|
* @param {string} cwd - Project root.
|
|
36
92
|
* @param {string} slug - Feature slug.
|
|
37
93
|
* @param {number} round - The EVAL round (`evaluate-r<N>.json`).
|
|
38
|
-
* @returns {{found: boolean, overall: (string|null),
|
|
39
|
-
* report_path: (string|null)}} `found
|
|
94
|
+
* @returns {{found: boolean, overall: (string|null), status: (string|null), reason: (string|null),
|
|
95
|
+
* bug_count: (number|null), report_path: (string|null)}} `found` is true only for a PASS/FAIL the
|
|
96
|
+
* round may act on; otherwise `reason` says why not — a fact, not a guess.
|
|
40
97
|
*/
|
|
41
98
|
export function evalVerdict(cwd, slug, round) {
|
|
42
99
|
const path = join(resultsDir(cwd, slug), `evaluate-r${round}.json`);
|
|
43
|
-
|
|
100
|
+
const unfit = (reason, status = null, overall = null) =>
|
|
101
|
+
({ found: false, overall, status, reason, bug_count: null, report_path: null });
|
|
102
|
+
if (!existsSync(path)) return unfit("no evaluate result for this round yet");
|
|
44
103
|
let doc;
|
|
45
104
|
try { doc = JSON.parse(readFileSync(path, "utf8")); }
|
|
46
|
-
catch { return
|
|
105
|
+
catch { return unfit("the evaluate result is not readable JSON"); }
|
|
106
|
+
const status = typeof doc?.status === "string" ? doc.status : null;
|
|
47
107
|
const v = doc?.verdict || {};
|
|
48
108
|
const overall = v.overall === "PASS" || v.overall === "FAIL" ? v.overall : null;
|
|
109
|
+
if (!overall) {
|
|
110
|
+
// A worker that refused to grade says why in its FIRST deviation — the only channel it has.
|
|
111
|
+
const first = Array.isArray(doc?.deviations) && typeof doc.deviations[0] === "string" ? doc.deviations[0] : "";
|
|
112
|
+
return unfit(clip(first
|
|
113
|
+
? `the evaluator returned ${status || "no status"}: ${first}`
|
|
114
|
+
: `status ${status || "unknown"} with no PASS/FAIL verdict`), status);
|
|
115
|
+
}
|
|
116
|
+
const problem = citationProblem(cwd, slug, v);
|
|
117
|
+
if (problem) return unfit(problem, status, overall);
|
|
49
118
|
return {
|
|
50
|
-
found:
|
|
119
|
+
found: true,
|
|
51
120
|
overall,
|
|
121
|
+
status,
|
|
122
|
+
reason: null,
|
|
52
123
|
bug_count: Array.isArray(v.bugs) ? v.bugs.length : null,
|
|
53
124
|
report_path: typeof v.report_path === "string" ? v.report_path : null,
|
|
54
125
|
};
|
|
@@ -66,12 +137,12 @@ export const ARGV_SPEC = {
|
|
|
66
137
|
* Report round N's EVAL verdict, mechanically, from the WorkResult on disk.
|
|
67
138
|
*
|
|
68
139
|
* @param {string[]} rawArgv - The subcommand's own arguments (harness.mjs strips the verb words).
|
|
69
|
-
* @returns {void} Exits 0 when a verdict
|
|
140
|
+
* @returns {void} Exits 0 when the round holds a verdict the run may act on, 1 when it does not.
|
|
70
141
|
*/
|
|
71
142
|
export function cli(rawArgv) {
|
|
72
143
|
const args = runArgs(ARGV_SPEC, rawArgv);
|
|
73
144
|
const cwd = resolve(args.cwd || process.cwd());
|
|
74
|
-
const { found, overall, bug_count, report_path } = evalVerdict(cwd, args.slug, args.round);
|
|
75
|
-
console.log(JSON.stringify({ ok: found, overall, bug_count, report_path, round: args.round }));
|
|
145
|
+
const { found, overall, status, reason, bug_count, report_path } = evalVerdict(cwd, args.slug, args.round);
|
|
146
|
+
console.log(JSON.stringify({ ok: found, overall, bug_count, report_path, round: args.round, status, reason }));
|
|
76
147
|
process.exit(found ? 0 : 1);
|
|
77
148
|
}
|
package/kernel/probe/resume.mjs
CHANGED
|
@@ -65,6 +65,7 @@ import {
|
|
|
65
65
|
intake, harnessRun, wiringMap, projectProfile, scopesDir, resultsDir, ordersDir,
|
|
66
66
|
orientDir, activeOrder, usecasesDir,
|
|
67
67
|
} from "../lib/paths.mjs";
|
|
68
|
+
import { evalVerdict } from "./eval.mjs";
|
|
68
69
|
|
|
69
70
|
/** The run-state values `references/protocol.md` (Part 4 — State) defines. A typo'd status is a rejection,
|
|
70
71
|
* not a write — the whole point of this file is that a write nobody validates is a write nobody
|
|
@@ -405,9 +406,15 @@ export function deriveResumeState(cwd, slug) {
|
|
|
405
406
|
// that permits the overlap is the same one that makes it invisible to the disjointness lint.
|
|
406
407
|
scope_exclusions: scopeExclusions(cwd, slug, scope_files),
|
|
407
408
|
pending_orders: orderFiles.filter((f) => f.endsWith(".json") && !resultFiles.includes(f)),
|
|
409
|
+
// A round is DONE when it was graded, not when its result file exists. An evaluator that
|
|
410
|
+
// refused the round — no PASS/FAIL, or a scoped verdict citing no T0 artifact — still writes
|
|
411
|
+
// `evaluate-r<N>.json`; counted, the relaunch opened round N+1 over a round nobody judged, with
|
|
412
|
+
// no bugs to route, and rebuilt every scope. Left open, it re-enters round N, skips the scopes
|
|
413
|
+
// already green there, and evaluates again.
|
|
408
414
|
eval_rounds_done: resultFiles
|
|
409
415
|
.filter((f) => /^evaluate-r\d+\.json$/.test(f))
|
|
410
|
-
.map((f) => Number(f.match(/\d+/)[0]))
|
|
416
|
+
.map((f) => Number(f.match(/\d+/)[0]))
|
|
417
|
+
.filter((n) => evalVerdict(cwd, slug, n).found),
|
|
411
418
|
};
|
|
412
419
|
return { ...facts, next_phase: nextPhase(facts) };
|
|
413
420
|
}
|
package/kernel/probe/t0.mjs
CHANGED
|
@@ -18,13 +18,36 @@ import { join, resolve } from "node:path";
|
|
|
18
18
|
import { runArgs } from "../lib/argv.mjs";
|
|
19
19
|
import { verdictsDir } from "../lib/paths.mjs";
|
|
20
20
|
|
|
21
|
+
/**
|
|
22
|
+
* Verdict filenames, newest first by their NUMERIC address.
|
|
23
|
+
*
|
|
24
|
+
* A string sort files `r1-a1-t10.json` before `r1-a1-t9.json`, and the trial ordinal is shared by
|
|
25
|
+
* every scope verified at one (round, attempt) — ten scopes on their first attempt are enough to
|
|
26
|
+
* make an older verdict read as the newest. Names that carry no address sort last.
|
|
27
|
+
*
|
|
28
|
+
* @param {string[]} names - Filenames from the verdicts directory.
|
|
29
|
+
* @returns {string[]} A new array, newest first: round, then attempt, then trial, descending.
|
|
30
|
+
*/
|
|
31
|
+
export function newestFirst(names) {
|
|
32
|
+
const key = (f) => {
|
|
33
|
+
const m = f.match(/^r(\d+)-a(\d+)(?:-t(\d+))?\.json$/);
|
|
34
|
+
return m ? [Number(m[1]), Number(m[2]), Number(m[3] ?? 0)] : [-1, -1, -1];
|
|
35
|
+
};
|
|
36
|
+
return [...names].sort((a, b) => {
|
|
37
|
+
const ka = key(a), kb = key(b);
|
|
38
|
+
for (let i = 0; i < 3; i++) if (ka[i] !== kb[i]) return kb[i] - ka[i];
|
|
39
|
+
return b.localeCompare(a);
|
|
40
|
+
});
|
|
41
|
+
}
|
|
42
|
+
|
|
21
43
|
/**
|
|
22
44
|
* The newest green T0 verdict for one scope in one round.
|
|
23
45
|
*
|
|
24
46
|
* @param {string} cwd - Project root.
|
|
25
47
|
* @param {string} slug - Feature slug.
|
|
26
48
|
* @param {string} scopeId - Scope contract id.
|
|
27
|
-
* @param {number} round - Build round.
|
|
49
|
+
* @param {number} [round] - Build round. Omitted, the newest green verdict of ANY round — what an
|
|
50
|
+
* evaluation with no round (a standalone single pass) has to cite.
|
|
28
51
|
* @returns {{green: boolean, path: (string|null)}} `path` is the artifact a later EVAL can cite.
|
|
29
52
|
*/
|
|
30
53
|
export function greenVerdict(cwd, slug, scopeId, round) {
|
|
@@ -32,11 +55,11 @@ export function greenVerdict(cwd, slug, scopeId, round) {
|
|
|
32
55
|
if (!existsSync(dir)) return { green: false, path: null };
|
|
33
56
|
// Newest first: an attempt retried after a red one writes a higher trial ordinal at the same
|
|
34
57
|
// (round, attempt) address, and the LAST verdict is the one that stands.
|
|
35
|
-
for (const f of readdirSync(dir).filter((x) => x.endsWith(".json"))
|
|
58
|
+
for (const f of newestFirst(readdirSync(dir).filter((x) => x.endsWith(".json")))) {
|
|
36
59
|
const p = join(dir, f);
|
|
37
60
|
try {
|
|
38
61
|
const b = JSON.parse(readFileSync(p, "utf8"));
|
|
39
|
-
if (b.scope_id === scopeId && b.round === round && b.overall === "green") return { green: true, path: p };
|
|
62
|
+
if (b.scope_id === scopeId && (round == null || b.round === round) && b.overall === "green") return { green: true, path: p };
|
|
40
63
|
} catch { /* a torn artifact proves nothing; keep looking */ }
|
|
41
64
|
}
|
|
42
65
|
return { green: false, path: null };
|
package/kernel/reduce/ingest.mjs
CHANGED
|
@@ -28,6 +28,7 @@ import { fileURLToPath } from "node:url";
|
|
|
28
28
|
import { validate } from "../verify/envelope.mjs";
|
|
29
29
|
import { runArgs } from "../lib/argv.mjs";
|
|
30
30
|
import { tasksDir, localRoot, dispatchReceipts, legLedger, readRunId } from "../lib/paths.mjs";
|
|
31
|
+
import { citationProblem } from "../probe/eval.mjs";
|
|
31
32
|
|
|
32
33
|
const HERE = dirname(fileURLToPath(import.meta.url));
|
|
33
34
|
const RESULT_SCHEMA = JSON.parse(readFileSync(resolve(HERE, "../../skills/tech-lead/schemas/work-result.schema.json"), "utf8"));
|
|
@@ -596,6 +597,20 @@ export async function cli(rawArgv) {
|
|
|
596
597
|
process.exit(1);
|
|
597
598
|
}
|
|
598
599
|
|
|
600
|
+
// --- T0 citation gate -----------------------------------------------------------------------
|
|
601
|
+
// A PASS or FAIL on a scoped spec that cites no T0 artifact is not a judgement this run may act
|
|
602
|
+
// on (see `citationProblem`). `probe eval` refuses it to the round loop; refusing it here as well
|
|
603
|
+
// keeps the verdict ledger from recording a verdict the loop will never branch on.
|
|
604
|
+
if (result.verdict) {
|
|
605
|
+
const problem = citationProblem(cwd, String(result.order_id).split("/")[0], result.verdict);
|
|
606
|
+
if (problem) {
|
|
607
|
+
console.error(`ingest-result: result refused — ${problem}.`);
|
|
608
|
+
console.error(` The round stays open: re-dispatch the evaluator against its order, which lists`);
|
|
609
|
+
console.error(` the T0 artifacts to cite. Nothing was written.`);
|
|
610
|
+
process.exit(1);
|
|
611
|
+
}
|
|
612
|
+
}
|
|
613
|
+
|
|
599
614
|
// Resolved for EVERY order, not only the gated ones: the attesting receipt is this leg's start,
|
|
600
615
|
// and a standalone or `--no-receipt-check` ingest still deserves a truthful timing row rather
|
|
601
616
|
// than one silently falling back to the order's re-writable `compiled_at`.
|
package/package.json
CHANGED
|
@@ -37,7 +37,7 @@ Invoked as `--order <path>`. Fields you may rely on (absent = unknown, never inf
|
|
|
37
37
|
| `payload.feature` | Feature slug — scopes the probe and names the report |
|
|
38
38
|
| `payload.dimensions[]` | The active dimension set (the caller resolved precedence). Absent → `[spec-conformance]` + the auto-enable rules below |
|
|
39
39
|
| `payload.run_cmd` | How to start the running app. Absent standalone → ask; absent orchestrated → ESCALATE, do not guess |
|
|
40
|
-
| `payload.t0_artifacts[]` | Per-scope T0 verdict paths for this round (scoped specs). An artifact listed but missing/red on disk, or a scoped spec with none listed → the round is NOT gradeable: return `status: failed` naming the scope — a structural precondition, not a criterion |
|
|
40
|
+
| `payload.t0_artifacts[]` | Per-scope T0 verdict paths for this round (scoped specs), compiled from each scope's green verdict. An artifact listed but missing/red on disk, or a scoped spec with none listed → the round is NOT gradeable: return `status: failed` with the reason, naming the scope, as your FIRST deviation — a structural precondition, not a criterion |
|
|
41
41
|
| `payload.browser` | `cli` (default, ~4x cheaper) \| `mcp` \| `none` |
|
|
42
42
|
| `payload.tasks[]` | Traceability only (which UCs a task claims): NEVER a grading source — the committed UC text is the criterion, a paraphrase mismatch is a finding |
|
|
43
43
|
| `substrate.allowed` | Your only write surface: `.shapeup/<slug>/evaluation/**` (the report + evidence) |
|
|
@@ -465,7 +465,9 @@ Read back: the stdout JSON — {path, sha256, trial, overall, regression, score,
|
|
|
465
465
|
## 4. EVAL → spec-evaluator (once per round)
|
|
466
466
|
```
|
|
467
467
|
compile-order --operation evaluate --slug <slug> --worker spec-evaluator --round <r>
|
|
468
|
-
--payload '{"dimensions": ["spec-conformance"], "run_cmd": "<cmd>"
|
|
468
|
+
--payload '{"dimensions": ["spec-conformance"], "run_cmd": "<cmd>"}'
|
|
469
|
+
t0_artifacts is compiled from each scope's green T0 verdict for round <r> — pass it only to
|
|
470
|
+
override. A scope with no green verdict is named on stderr: the judge has nothing to cite for it.
|
|
469
471
|
Invoke via Agent (model: eval), ONCE, after GATE L2:
|
|
470
472
|
Skill(shapeup-sdlc-plugin:spec-evaluator) --order <path>
|
|
471
473
|
Effect: one feature-level pass over the running app against all AC + Done-when; writes
|
|
@@ -473,6 +475,8 @@ Effect: one feature-level pass over the running app against all AC + Done-when;
|
|
|
473
475
|
verdicts, refuted boxes, T0 citations). It touches NO task file and NO board.
|
|
474
476
|
ingest-result <results/evaluate-r<r>.json>: appends the .verdicts JSONL ledger, un-ticks the
|
|
475
477
|
refuted AC boxes, sets eval_verdict frontmatter — the judge returns data, ingest writes.
|
|
478
|
+
A verdict on a scoped spec that cites no T0 artifact is refused and the round stays
|
|
479
|
+
open: re-dispatch the evaluator, do not advance the round.
|
|
476
480
|
Read back: EVAL-FEATURE-<slug>.md → verdict (pass|fail) + the bug list (each bug has
|
|
477
481
|
task ref, severity, file:line, expected vs actual).
|
|
478
482
|
```
|
|
@@ -2512,7 +2512,7 @@
|
|
|
2512
2512
|
"items": {
|
|
2513
2513
|
"type": "integer"
|
|
2514
2514
|
},
|
|
2515
|
-
"description": "Round numbers
|
|
2515
|
+
"description": "Round numbers whose evaluate-r<n>.json holds a verdict the run may act on (PASS/FAIL, citing T0 artifacts when the spec is scoped) — the resumed run's round counter starts one past the maximum. A refused or uncited round is not done: the relaunch re-enters it."
|
|
2516
2516
|
},
|
|
2517
2517
|
"next_phase": {
|
|
2518
2518
|
"type": "string",
|
|
@@ -560,6 +560,10 @@ const EVAL_VERDICT = {
|
|
|
560
560
|
bug_count: nullable("integer"),
|
|
561
561
|
report_path: nullable("string"),
|
|
562
562
|
round: { type: "integer" },
|
|
563
|
+
// Why the round holds no verdict it may act on — the evaluator's own first deviation when it
|
|
564
|
+
// refused, or what is structurally wrong with the verdict it returned. Null when `ok`.
|
|
565
|
+
status: nullable("string"),
|
|
566
|
+
reason: nullable("string"),
|
|
563
567
|
},
|
|
564
568
|
required: ["ok", "round"],
|
|
565
569
|
};
|
|
@@ -1358,14 +1362,22 @@ while (verdict !== "pass" && round <= maxRounds) {
|
|
|
1358
1362
|
const e = await worker({
|
|
1359
1363
|
skill: "spec-evaluator", operation: "evaluate", schema: EVAL, phase: "Eval", label: `eval:r${round}`,
|
|
1360
1364
|
model: evalModel, round,
|
|
1365
|
+
// No `t0_artifacts` here, deliberately: `harness compile` derives them from the round's green
|
|
1366
|
+
// T0 verdicts on disk, for every lane — this script could only name paths it was told about.
|
|
1361
1367
|
payload: { dimensions: evalDims, run_cmd: rs.run_cmd, round },
|
|
1362
|
-
extra: "Evaluate the running feature against every acceptance criterion and Done-when. One feature-level pass; cite the
|
|
1368
|
+
extra: "Evaluate the running feature against every acceptance criterion and Done-when. One feature-level pass; cite every artifact the order lists under t0_artifacts, re-hashing each yourself.",
|
|
1363
1369
|
});
|
|
1364
1370
|
if (e.__failed) return diedAt("L3", e);
|
|
1365
1371
|
// The pass/fail branch is decided from the WorkResult on disk, not from the dispatching
|
|
1366
1372
|
// agent's own summary of it (`e.overall`) — see EVAL_VERDICT's comment for why.
|
|
1367
1373
|
const ev = await query(`probe eval --slug ${slug} --round ${round}`, EVAL_VERDICT, "Eval", `verdict:r${round}`);
|
|
1368
|
-
if (!ev
|
|
1374
|
+
if (!ev) return diedAt("L3", nullFail(`verdict:r${round}`));
|
|
1375
|
+
// A round with no verdict to act on is NOT a dead worker. An evaluator that refused the round
|
|
1376
|
+
// wrote a result saying why, and `probe eval` carries it as `reason`; reported as "died after
|
|
1377
|
+
// retries", the one sentence naming the cause stayed in a file nobody was pointed at.
|
|
1378
|
+
if (!ev.ok || !ev.overall) {
|
|
1379
|
+
return diedAt("L3", { __failed: `verdict:r${round}: no verdict this round can act on — ${ev.reason || `status ${ev.status || "unknown"}`}` });
|
|
1380
|
+
}
|
|
1369
1381
|
verdict = ev.overall === "PASS" ? "pass" : "fail";
|
|
1370
1382
|
findings = e.findings || [];
|
|
1371
1383
|
}
|