shapeup-sdlc 3.1.1 → 3.1.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "name": "shapeup-sdlc-plugin",
3
3
  "displayName": "ShapeUp SDLC Plugin",
4
- "version": "3.1.1",
4
+ "version": "3.1.2",
5
5
  "description": "Shape Up SDLC harness for Claude Code: shaping, intake, orient, scope-mapping, building (T0-verified, sandboxed, scope-contracted), evaluation and QA skills orchestrated by a tech-lead.",
6
6
  "author": {
7
7
  "name": "Liberty Nguyen",
package/AGENTS.md CHANGED
@@ -58,7 +58,7 @@ Everything discovered funnels into `.shapeup/<slug>/discovery/ledger.md` (Orient
58
58
  - **Ledger = single source of truth** — every discovery flow writes only its own section.
59
59
  - **QA is a level-up, not a gate** — `--no-qa` skips it; circuit breaker outranks the Hunter.
60
60
  - **Role separation** — Evaluator grades, task-executor fixes, QA discovers.
61
- - **Hill phase is mechanical ✦** — derived only from T0/T1/seesaw artifacts, never self-reported; the evaluator cites a T0 artifact it re-hashes itself.
61
+ - **Hill phase is mechanical ✦** — derived only from T0/T1/seesaw artifacts, never self-reported; the evaluator cites a T0 artifact it re-hashes itself, from the list its order carries. A scoped verdict citing none is refused: its round stays open and is evaluated again, never advanced.
62
62
  - **Envelope port (v1.0)** — every dispatch is WorkOrder in / WorkResult out; shared state has exactly one writer (the ingest step); malformed envelopes are hook-denied. Workers: stateless, craft-only, pipeline-blind.
63
63
 
64
64
  ## Setup & Execution
@@ -36,10 +36,11 @@ import { readRunId } from "./lib/paths.mjs";
36
36
  // --spec-overridden directory, and the import is the convention-derived default.
37
37
  import {
38
38
  tasksDir, specDir as defaultSpecDir, roundLedger, trials, verdictsDir, ordersDir,
39
- relShared, globLocal, globShared, relKnowledgeBase, resultsDir, scopesDir,
39
+ relShared, relLocal, globLocal, globShared, relKnowledgeBase, resultsDir, scopesDir,
40
40
  } from "./lib/paths.mjs";
41
- import { readContract, tasksForScope, SCOPE_CONTRACT } from "./lib/contract.mjs";
41
+ import { readContract, readAllContracts, tasksForScope, SCOPE_CONTRACT } from "./lib/contract.mjs";
42
42
  import { writeActiveOrder } from "./probe/resume.mjs";
43
+ import { greenVerdict } from "./probe/t0.mjs";
43
44
  // The SAME matcher the sandbox hook enforces with. "Is this cited file inside this scope's
44
45
  // substrate" has to mean exactly what the guard means, or a bug is addressed to a scope that is
45
46
  // then denied the write that fixes it.
@@ -482,6 +483,43 @@ export function scopeSubstrates(cwd, slug) {
482
483
  return out;
483
484
  }
484
485
 
486
+ // --- the T0 artifacts the judge must cite ----------------------------------------------------
487
+ //
488
+ // WHY THE KERNEL DERIVES THEM. spec-evaluator treats a scoped spec whose order lists no T0 artifact
489
+ // as NOT gradeable and returns `failed` without grading a criterion. The precondition is right —
490
+ // its verdict must cite a T0 artifact it re-hashed itself — and nothing met it: the list was once
491
+ // assembled by an orchestrator courier from the paths each scope reported, and was lost when the
492
+ // orchestrator became a workflow script that passes only `{dimensions, run_cmd, round}`. Evaluators
493
+ // that went looking on disk graded anyway; one that followed its contract refused, and the run
494
+ // aborted at L3 over a round whose every scope was green.
495
+ //
496
+ // Derived here for the reason `bugs` is: this is the one line every lane compiles through, and the
497
+ // evidence is already on disk. A caller could not rebuild the list from filenames in any case —
498
+ // verdict files are addressed by round, attempt and trial, never by scope, so the scope lives only
499
+ // inside each body, which is what `probe t0` reads.
500
+
501
+ /**
502
+ * The green T0 verdict each scope contract holds for a round — an evaluate order's `t0_artifacts`.
503
+ *
504
+ * @param {string} cwd - Project root.
505
+ * @param {string} slug - Feature slug.
506
+ * @param {number} [round] - The round being evaluated. Omitted, each scope's newest green verdict
507
+ * of any round — a standalone evaluation has no round.
508
+ * @returns {{artifacts: string[], missing: string[]}} Repo-relative verdict paths in scope-id
509
+ * order, one per scope that has one; and the scopes that have none. Both empty on an unscoped spec.
510
+ */
511
+ export function t0ArtifactsFor(cwd, slug, round) {
512
+ const artifacts = [];
513
+ const missing = [];
514
+ for (const { contract, id } of readAllContracts(scopesDir(cwd, slug))) {
515
+ const scopeId = contract?.scope_id || id;
516
+ const { green, path } = greenVerdict(cwd, slug, scopeId, round);
517
+ if (green) artifacts.push(relLocal(slug, "t0", "verdicts", basename(path)));
518
+ else missing.push(scopeId);
519
+ }
520
+ return { artifacts, missing };
521
+ }
522
+
485
523
  /**
486
524
  * Assemble a WorkOrder envelope. Pure given its inputs — the CLI wrapper does the disk reads.
487
525
  * @param {object} opts - The order inputs (destructured):
@@ -743,6 +781,17 @@ export async function cli(rawArgv) {
743
781
  let payloadExtra = flag("payload") || {};
744
782
  if (specDir && !payloadExtra.spec_folder) payloadExtra.spec_folder = specDir;
745
783
  if (!payloadExtra.feature) payloadExtra.feature = slug;
784
+ // The judge's citations, for every lane (see t0ArtifactsFor). An explicit `--payload` list still
785
+ // wins, as it does for `bugs`: an operator naming the evidence outranks the derivation.
786
+ if (operation === "evaluate" && payloadExtra.t0_artifacts === undefined) {
787
+ const { artifacts, missing } = t0ArtifactsFor(cwd, slug, round);
788
+ if (artifacts.length) payloadExtra.t0_artifacts = artifacts;
789
+ // On stderr, never stdout: stdout is the order path the caller consumes.
790
+ if (missing.length) {
791
+ console.error(`compile-order: warning — no green T0 verdict${round ? ` in round ${round}` : ""} for ` +
792
+ `${missing.join(", ")}; the evaluator has nothing to cite for ${missing.length === 1 ? "that scope" : "those scopes"}`);
793
+ }
794
+ }
746
795
 
747
796
  const order = compileOrder({
748
797
  slug, worker, operation, round, attempt, scope, tasks, decisions, digestedErrors, trialHistory, bugs,
@@ -1,8 +1,10 @@
1
1
  // probe eval — "what did round N's EVAL WorkResult actually say?"
2
2
  //
3
3
  // CONTRACT. A bounded, read-only query over the evaluate WorkResult ingest already wrote. Prints
4
- // `{ok, overall, bug_count, report_path}` on stdout; exits 0 when found and readable, 1 when the
5
- // result is missing (nothing ran, or ingest hasn't landed yet), 2 on a bad argv. Writes nothing.
4
+ // `{ok, overall, bug_count, report_path, round, status, reason}` on stdout; exits 0 when the round
5
+ // holds a verdict the run may act on, 1 when it does not — nothing ran, ingest hasn't landed, the
6
+ // evaluator refused the round, or the verdict is structurally invalid; `reason` says which — and 2
7
+ // on a bad argv. Writes nothing.
6
8
  //
7
9
  // WHY THIS EXISTS. `shapeup-run.js` cannot read a file itself (a Workflow script has no filesystem
8
10
  // of its own — see this repo's own note on why it may not call `Date.now()`), so every fact it
@@ -23,11 +25,65 @@
23
25
  // shared state; the `.md` report is prose for a human. Reading the prose to re-derive a verdict a
24
26
  // schema already carries structurally is the paraphrase channel this repo's hooks exist to close
25
27
  // everywhere else.
28
+ //
29
+ // WHY "NO VERDICT" CARRIES A REASON. An evaluator that refuses a round — a structural precondition
30
+ // it cannot meet — still writes `evaluate-r<N>.json`, with `status: failed`, no verdict, and the
31
+ // cause as its first deviation. A bare `ok: false` reached the operator as a sub-agent that died
32
+ // after retries, while the one sentence naming the actual cause sat in a file nobody was pointed at.
26
33
 
27
- import { existsSync, readFileSync } from "node:fs";
34
+ import { existsSync, readFileSync, readdirSync } from "node:fs";
28
35
  import { join, resolve } from "node:path";
29
36
  import { runArgs } from "../lib/argv.mjs";
30
- import { resultsDir } from "../lib/paths.mjs";
37
+ import { resultsDir, scopesDir } from "../lib/paths.mjs";
38
+
39
+ /** Longest `reason` reported. A deviation is prose written by a worker and can run to paragraphs. */
40
+ const REASON_MAX = 400;
41
+
42
+ /**
43
+ * Bound a reason to {@link REASON_MAX} characters.
44
+ * @param {string} s - The reason.
45
+ * @returns {string} `s`, or its first REASON_MAX − 1 characters and an ellipsis.
46
+ */
47
+ const clip = (s) => (s.length > REASON_MAX ? `${s.slice(0, REASON_MAX - 1)}…` : s);
48
+
49
+ /**
50
+ * Whether a feature's spec is SCOPED — has scope contracts, the case in which every verdict must
51
+ * cite the T0 artifacts it re-hashed.
52
+ *
53
+ * @param {string} cwd - Project root.
54
+ * @param {string} slug - Feature slug.
55
+ * @returns {boolean} True when `scopes/` holds at least one contract (`.md`, or a legacy `.json`).
56
+ */
57
+ export function isScoped(cwd, slug) {
58
+ try { return readdirSync(scopesDir(cwd, slug)).some((f) => /\.(md|json)$/.test(f)); }
59
+ catch { return false; }
60
+ }
61
+
62
+ /**
63
+ * Why a verdict cannot stand as its round's judgement on T0 grounds, or null when it can.
64
+ *
65
+ * A PASS or FAIL on a scoped spec that cites no T0 artifact is structurally invalid — the
66
+ * evaluator's own contract says so, because T0 is the machine fact a generator cannot fabricate.
67
+ * That rule used to live only in the contract, so a verdict citing nothing was ingested, ledgered
68
+ * and branched on like any other. It is checked here so the round loop, the resume derivation, the
69
+ * hill and ingest all refuse the same verdict for the same reason.
70
+ *
71
+ * PRESENCE, NOT HASHES. The evaluator re-hashes what it cites; a slip transcribing a digest is not
72
+ * evidence the verdict is wrong, and refusing a round over one would cost a whole re-evaluation.
73
+ *
74
+ * @param {string} cwd - Project root.
75
+ * @param {string} slug - Feature slug.
76
+ * @param {object} verdict - The WorkResult's `verdict` block.
77
+ * @returns {(string|null)} The problem, phrased for an operator; null for a cited verdict, an
78
+ * unscoped spec, or a block with no PASS/FAIL in it (there is no judgement to invalidate).
79
+ */
80
+ export function citationProblem(cwd, slug, verdict) {
81
+ if (verdict?.overall !== "PASS" && verdict?.overall !== "FAIL") return null;
82
+ if (Array.isArray(verdict.t0_citations) && verdict.t0_citations.length) return null;
83
+ if (!isScoped(cwd, slug)) return null;
84
+ return `the ${verdict.overall} verdict cites no T0 artifact, and a verdict on a scoped spec must ` +
85
+ "cite the T0 verdict it re-hashed (the order lists them under payload.t0_artifacts)";
86
+ }
31
87
 
32
88
  /**
33
89
  * Read one round's EVAL verdict straight from the WorkResult `reduce ingest` wrote.
@@ -35,20 +91,35 @@ import { resultsDir } from "../lib/paths.mjs";
35
91
  * @param {string} cwd - Project root.
36
92
  * @param {string} slug - Feature slug.
37
93
  * @param {number} round - The EVAL round (`evaluate-r<N>.json`).
38
- * @returns {{found: boolean, overall: (string|null), bug_count: (number|null),
39
- * report_path: (string|null)}} `found: false` when no result exists yet — a fact, not a guess.
94
+ * @returns {{found: boolean, overall: (string|null), status: (string|null), reason: (string|null),
95
+ * bug_count: (number|null), report_path: (string|null)}} `found` is true only for a PASS/FAIL the
96
+ * round may act on; otherwise `reason` says why not — a fact, not a guess.
40
97
  */
41
98
  export function evalVerdict(cwd, slug, round) {
42
99
  const path = join(resultsDir(cwd, slug), `evaluate-r${round}.json`);
43
- if (!existsSync(path)) return { found: false, overall: null, bug_count: null, report_path: null };
100
+ const unfit = (reason, status = null, overall = null) =>
101
+ ({ found: false, overall, status, reason, bug_count: null, report_path: null });
102
+ if (!existsSync(path)) return unfit("no evaluate result for this round yet");
44
103
  let doc;
45
104
  try { doc = JSON.parse(readFileSync(path, "utf8")); }
46
- catch { return { found: false, overall: null, bug_count: null, report_path: null }; }
105
+ catch { return unfit("the evaluate result is not readable JSON"); }
106
+ const status = typeof doc?.status === "string" ? doc.status : null;
47
107
  const v = doc?.verdict || {};
48
108
  const overall = v.overall === "PASS" || v.overall === "FAIL" ? v.overall : null;
109
+ if (!overall) {
110
+ // A worker that refused to grade says why in its FIRST deviation — the only channel it has.
111
+ const first = Array.isArray(doc?.deviations) && typeof doc.deviations[0] === "string" ? doc.deviations[0] : "";
112
+ return unfit(clip(first
113
+ ? `the evaluator returned ${status || "no status"}: ${first}`
114
+ : `status ${status || "unknown"} with no PASS/FAIL verdict`), status);
115
+ }
116
+ const problem = citationProblem(cwd, slug, v);
117
+ if (problem) return unfit(problem, status, overall);
49
118
  return {
50
- found: overall !== null,
119
+ found: true,
51
120
  overall,
121
+ status,
122
+ reason: null,
52
123
  bug_count: Array.isArray(v.bugs) ? v.bugs.length : null,
53
124
  report_path: typeof v.report_path === "string" ? v.report_path : null,
54
125
  };
@@ -66,12 +137,12 @@ export const ARGV_SPEC = {
66
137
  * Report round N's EVAL verdict, mechanically, from the WorkResult on disk.
67
138
  *
68
139
  * @param {string[]} rawArgv - The subcommand's own arguments (harness.mjs strips the verb words).
69
- * @returns {void} Exits 0 when a verdict was found, 1 when none exists yet for this round.
140
+ * @returns {void} Exits 0 when the round holds a verdict the run may act on, 1 when it does not.
70
141
  */
71
142
  export function cli(rawArgv) {
72
143
  const args = runArgs(ARGV_SPEC, rawArgv);
73
144
  const cwd = resolve(args.cwd || process.cwd());
74
- const { found, overall, bug_count, report_path } = evalVerdict(cwd, args.slug, args.round);
75
- console.log(JSON.stringify({ ok: found, overall, bug_count, report_path, round: args.round }));
145
+ const { found, overall, status, reason, bug_count, report_path } = evalVerdict(cwd, args.slug, args.round);
146
+ console.log(JSON.stringify({ ok: found, overall, bug_count, report_path, round: args.round, status, reason }));
76
147
  process.exit(found ? 0 : 1);
77
148
  }
@@ -65,6 +65,7 @@ import {
65
65
  intake, harnessRun, wiringMap, projectProfile, scopesDir, resultsDir, ordersDir,
66
66
  orientDir, activeOrder, usecasesDir,
67
67
  } from "../lib/paths.mjs";
68
+ import { evalVerdict } from "./eval.mjs";
68
69
 
69
70
  /** The run-state values `references/protocol.md` (Part 4 — State) defines. A typo'd status is a rejection,
70
71
  * not a write — the whole point of this file is that a write nobody validates is a write nobody
@@ -405,9 +406,15 @@ export function deriveResumeState(cwd, slug) {
405
406
  // that permits the overlap is the same one that makes it invisible to the disjointness lint.
406
407
  scope_exclusions: scopeExclusions(cwd, slug, scope_files),
407
408
  pending_orders: orderFiles.filter((f) => f.endsWith(".json") && !resultFiles.includes(f)),
409
+ // A round is DONE when it was graded, not when its result file exists. An evaluator that
410
+ // refused the round — no PASS/FAIL, or a scoped verdict citing no T0 artifact — still writes
411
+ // `evaluate-r<N>.json`; counted, the relaunch opened round N+1 over a round nobody judged, with
412
+ // no bugs to route, and rebuilt every scope. Left open, it re-enters round N, skips the scopes
413
+ // already green there, and evaluates again.
408
414
  eval_rounds_done: resultFiles
409
415
  .filter((f) => /^evaluate-r\d+\.json$/.test(f))
410
- .map((f) => Number(f.match(/\d+/)[0])),
416
+ .map((f) => Number(f.match(/\d+/)[0]))
417
+ .filter((n) => evalVerdict(cwd, slug, n).found),
411
418
  };
412
419
  return { ...facts, next_phase: nextPhase(facts) };
413
420
  }
@@ -18,13 +18,36 @@ import { join, resolve } from "node:path";
18
18
  import { runArgs } from "../lib/argv.mjs";
19
19
  import { verdictsDir } from "../lib/paths.mjs";
20
20
 
21
+ /**
22
+ * Verdict filenames, newest first by their NUMERIC address.
23
+ *
24
+ * A string sort files `r1-a1-t10.json` before `r1-a1-t9.json`, and the trial ordinal is shared by
25
+ * every scope verified at one (round, attempt) — ten scopes on their first attempt are enough to
26
+ * make an older verdict read as the newest. Names that carry no address sort last.
27
+ *
28
+ * @param {string[]} names - Filenames from the verdicts directory.
29
+ * @returns {string[]} A new array, newest first: round, then attempt, then trial, descending.
30
+ */
31
+ export function newestFirst(names) {
32
+ const key = (f) => {
33
+ const m = f.match(/^r(\d+)-a(\d+)(?:-t(\d+))?\.json$/);
34
+ return m ? [Number(m[1]), Number(m[2]), Number(m[3] ?? 0)] : [-1, -1, -1];
35
+ };
36
+ return [...names].sort((a, b) => {
37
+ const ka = key(a), kb = key(b);
38
+ for (let i = 0; i < 3; i++) if (ka[i] !== kb[i]) return kb[i] - ka[i];
39
+ return b.localeCompare(a);
40
+ });
41
+ }
42
+
21
43
  /**
22
44
  * The newest green T0 verdict for one scope in one round.
23
45
  *
24
46
  * @param {string} cwd - Project root.
25
47
  * @param {string} slug - Feature slug.
26
48
  * @param {string} scopeId - Scope contract id.
27
- * @param {number} round - Build round.
49
+ * @param {number} [round] - Build round. Omitted, the newest green verdict of ANY round — what an
50
+ * evaluation with no round (a standalone single pass) has to cite.
28
51
  * @returns {{green: boolean, path: (string|null)}} `path` is the artifact a later EVAL can cite.
29
52
  */
30
53
  export function greenVerdict(cwd, slug, scopeId, round) {
@@ -32,11 +55,11 @@ export function greenVerdict(cwd, slug, scopeId, round) {
32
55
  if (!existsSync(dir)) return { green: false, path: null };
33
56
  // Newest first: an attempt retried after a red one writes a higher trial ordinal at the same
34
57
  // (round, attempt) address, and the LAST verdict is the one that stands.
35
- for (const f of readdirSync(dir).filter((x) => x.endsWith(".json")).sort().reverse()) {
58
+ for (const f of newestFirst(readdirSync(dir).filter((x) => x.endsWith(".json")))) {
36
59
  const p = join(dir, f);
37
60
  try {
38
61
  const b = JSON.parse(readFileSync(p, "utf8"));
39
- if (b.scope_id === scopeId && b.round === round && b.overall === "green") return { green: true, path: p };
62
+ if (b.scope_id === scopeId && (round == null || b.round === round) && b.overall === "green") return { green: true, path: p };
40
63
  } catch { /* a torn artifact proves nothing; keep looking */ }
41
64
  }
42
65
  return { green: false, path: null };
@@ -28,6 +28,7 @@ import { fileURLToPath } from "node:url";
28
28
  import { validate } from "../verify/envelope.mjs";
29
29
  import { runArgs } from "../lib/argv.mjs";
30
30
  import { tasksDir, localRoot, dispatchReceipts, legLedger, readRunId } from "../lib/paths.mjs";
31
+ import { citationProblem } from "../probe/eval.mjs";
31
32
 
32
33
  const HERE = dirname(fileURLToPath(import.meta.url));
33
34
  const RESULT_SCHEMA = JSON.parse(readFileSync(resolve(HERE, "../../skills/tech-lead/schemas/work-result.schema.json"), "utf8"));
@@ -596,6 +597,20 @@ export async function cli(rawArgv) {
596
597
  process.exit(1);
597
598
  }
598
599
 
600
+ // --- T0 citation gate -----------------------------------------------------------------------
601
+ // A PASS or FAIL on a scoped spec that cites no T0 artifact is not a judgement this run may act
602
+ // on (see `citationProblem`). `probe eval` refuses it to the round loop; refusing it here as well
603
+ // keeps the verdict ledger from recording a verdict the loop will never branch on.
604
+ if (result.verdict) {
605
+ const problem = citationProblem(cwd, String(result.order_id).split("/")[0], result.verdict);
606
+ if (problem) {
607
+ console.error(`ingest-result: result refused — ${problem}.`);
608
+ console.error(` The round stays open: re-dispatch the evaluator against its order, which lists`);
609
+ console.error(` the T0 artifacts to cite. Nothing was written.`);
610
+ process.exit(1);
611
+ }
612
+ }
613
+
599
614
  // Resolved for EVERY order, not only the gated ones: the attesting receipt is this leg's start,
600
615
  // and a standalone or `--no-receipt-check` ingest still deserves a truthful timing row rather
601
616
  // than one silently falling back to the order's re-writable `compiled_at`.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "shapeup-sdlc",
3
- "version": "3.1.1",
3
+ "version": "3.1.2",
4
4
  "description": "Shape Up for coding agents \u2014 with gates the agent can't talk its way past. Harness for Claude Code.",
5
5
  "bin": {
6
6
  "shapeup-sdlc": "bin/init.mjs"
@@ -37,7 +37,7 @@ Invoked as `--order <path>`. Fields you may rely on (absent = unknown, never inf
37
37
  | `payload.feature` | Feature slug — scopes the probe and names the report |
38
38
  | `payload.dimensions[]` | The active dimension set (the caller resolved precedence). Absent → `[spec-conformance]` + the auto-enable rules below |
39
39
  | `payload.run_cmd` | How to start the running app. Absent standalone → ask; absent orchestrated → ESCALATE, do not guess |
40
- | `payload.t0_artifacts[]` | Per-scope T0 verdict paths for this round (scoped specs). An artifact listed but missing/red on disk, or a scoped spec with none listed → the round is NOT gradeable: return `status: failed` naming the scope — a structural precondition, not a criterion |
40
+ | `payload.t0_artifacts[]` | Per-scope T0 verdict paths for this round (scoped specs), compiled from each scope's green verdict. An artifact listed but missing/red on disk, or a scoped spec with none listed → the round is NOT gradeable: return `status: failed` with the reason, naming the scope, as your FIRST deviation — a structural precondition, not a criterion |
41
41
  | `payload.browser` | `cli` (default, ~4x cheaper) \| `mcp` \| `none` |
42
42
  | `payload.tasks[]` | Traceability only (which UCs a task claims): NEVER a grading source — the committed UC text is the criterion, a paraphrase mismatch is a finding |
43
43
  | `substrate.allowed` | Your only write surface: `.shapeup/<slug>/evaluation/**` (the report + evidence) |
@@ -465,7 +465,9 @@ Read back: the stdout JSON — {path, sha256, trial, overall, regression, score,
465
465
  ## 4. EVAL → spec-evaluator (once per round)
466
466
  ```
467
467
  compile-order --operation evaluate --slug <slug> --worker spec-evaluator --round <r>
468
- --payload '{"dimensions": ["spec-conformance"], "run_cmd": "<cmd>", "t0_artifacts": [...]}'
468
+ --payload '{"dimensions": ["spec-conformance"], "run_cmd": "<cmd>"}'
469
+ t0_artifacts is compiled from each scope's green T0 verdict for round <r> — pass it only to
470
+ override. A scope with no green verdict is named on stderr: the judge has nothing to cite for it.
469
471
  Invoke via Agent (model: eval), ONCE, after GATE L2:
470
472
  Skill(shapeup-sdlc-plugin:spec-evaluator) --order <path>
471
473
  Effect: one feature-level pass over the running app against all AC + Done-when; writes
@@ -473,6 +475,8 @@ Effect: one feature-level pass over the running app against all AC + Done-when;
473
475
  verdicts, refuted boxes, T0 citations). It touches NO task file and NO board.
474
476
  ingest-result <results/evaluate-r<r>.json>: appends the .verdicts JSONL ledger, un-ticks the
475
477
  refuted AC boxes, sets eval_verdict frontmatter — the judge returns data, ingest writes.
478
+ A verdict on a scoped spec that cites no T0 artifact is refused and the round stays
479
+ open: re-dispatch the evaluator, do not advance the round.
476
480
  Read back: EVAL-FEATURE-<slug>.md → verdict (pass|fail) + the bug list (each bug has
477
481
  task ref, severity, file:line, expected vs actual).
478
482
  ```
@@ -2512,7 +2512,7 @@
2512
2512
  "items": {
2513
2513
  "type": "integer"
2514
2514
  },
2515
- "description": "Round numbers with an evaluate-r<n>.json result — the resumed run's round counter starts one past the maximum."
2515
+ "description": "Round numbers whose evaluate-r<n>.json holds a verdict the run may act on (PASS/FAIL, citing T0 artifacts when the spec is scoped) — the resumed run's round counter starts one past the maximum. A refused or uncited round is not done: the relaunch re-enters it."
2516
2516
  },
2517
2517
  "next_phase": {
2518
2518
  "type": "string",
@@ -560,6 +560,10 @@ const EVAL_VERDICT = {
560
560
  bug_count: nullable("integer"),
561
561
  report_path: nullable("string"),
562
562
  round: { type: "integer" },
563
+ // Why the round holds no verdict it may act on — the evaluator's own first deviation when it
564
+ // refused, or what is structurally wrong with the verdict it returned. Null when `ok`.
565
+ status: nullable("string"),
566
+ reason: nullable("string"),
563
567
  },
564
568
  required: ["ok", "round"],
565
569
  };
@@ -1358,14 +1362,22 @@ while (verdict !== "pass" && round <= maxRounds) {
1358
1362
  const e = await worker({
1359
1363
  skill: "spec-evaluator", operation: "evaluate", schema: EVAL, phase: "Eval", label: `eval:r${round}`,
1360
1364
  model: evalModel, round,
1365
+ // No `t0_artifacts` here, deliberately: `harness compile` derives them from the round's green
1366
+ // T0 verdicts on disk, for every lane — this script could only name paths it was told about.
1361
1367
  payload: { dimensions: evalDims, run_cmd: rs.run_cmd, round },
1362
- extra: "Evaluate the running feature against every acceptance criterion and Done-when. One feature-level pass; cite the T0 artifact you re-hash yourself.",
1368
+ extra: "Evaluate the running feature against every acceptance criterion and Done-when. One feature-level pass; cite every artifact the order lists under t0_artifacts, re-hashing each yourself.",
1363
1369
  });
1364
1370
  if (e.__failed) return diedAt("L3", e);
1365
1371
  // The pass/fail branch is decided from the WorkResult on disk, not from the dispatching
1366
1372
  // agent's own summary of it (`e.overall`) — see EVAL_VERDICT's comment for why.
1367
1373
  const ev = await query(`probe eval --slug ${slug} --round ${round}`, EVAL_VERDICT, "Eval", `verdict:r${round}`);
1368
- if (!ev || !ev.ok || !ev.overall) return diedAt("L3", nullFail(`verdict:r${round}`));
1374
+ if (!ev) return diedAt("L3", nullFail(`verdict:r${round}`));
1375
+ // A round with no verdict to act on is NOT a dead worker. An evaluator that refused the round
1376
+ // wrote a result saying why, and `probe eval` carries it as `reason`; reported as "died after
1377
+ // retries", the one sentence naming the cause stayed in a file nobody was pointed at.
1378
+ if (!ev.ok || !ev.overall) {
1379
+ return diedAt("L3", { __failed: `verdict:r${round}: no verdict this round can act on — ${ev.reason || `status ${ev.status || "unknown"}`}` });
1380
+ }
1369
1381
  verdict = ev.overall === "PASS" ? "pass" : "fail";
1370
1382
  findings = e.findings || [];
1371
1383
  }