shapeup-sdlc 3.2.0 → 3.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -68,7 +68,9 @@ approval of the new tasks and estimates before resuming the BUILD loop.
68
68
 
69
69
  ## The EVAL timing rule (the core constraint)
70
70
 
71
- EVAL fires **once** per round and **only** when GATE L2 has confirmed the board is 100% done.
71
+ EVAL fires **once** per round and **only** when GATE L2 has confirmed the board is 100% done
72
+ and the round build gate (§3d) is not red — a feature that does not build or launch has nothing
73
+ for a judge to grade, and the gate's failing step is what round r+1 fixes.
72
74
  It is never:
73
75
  - called per task,
74
76
  - called inside the BUILD loop,
@@ -463,6 +465,27 @@ Read back: the stdout JSON — {path, sha256, trial, overall, regression, score,
463
465
  (harness verify t0 calls its sibling harness probe digest internally on failure).
464
466
  ```
465
467
 
468
+ ## 3d. Round build gate → `harness verify build` (once per round, before EVAL)
469
+ ```
470
+ Invoke via Bash directly — deterministic tooling, not a worker:
471
+ node "${CLAUDE_PLUGIN_ROOT}/kernel/harness.mjs" verify build --slug <slug> --round <N>
472
+ Effect: runs, in order and stopping at the first failure, the ledger's `run_cmd` (the build), then
473
+ project-profile.md's `build_probe` (the built artifact covers what the run wrote — a green
474
+ exit code is not proof the feature compiled when the toolchain compiles only what an entry
475
+ point reaches) and `launch_probe` (install, start, assert the first screen, fail on fatal
476
+ logs). Writes .shapeup/<slug>/build/r<N>-t<T>.json, immutable per run of the gate.
477
+ Read back: stdout JSON — {overall: green|red, steps[], warnings[], failed_step?, stderr_tail?}.
478
+ Exit 0 green · 1 red · 3 nothing declared (no run_cmd, no probes — logged, never green).
479
+ A `mobile` profile with no launch_probe is warned about on stderr every round, and so is
480
+ every scope none of whose fixtures invoke the tool run_cmd builds with — advisory, because a
481
+ green T0 from such fixtures is not evidence the scope compiles.
482
+ Consequences, both mechanical and both read off the artifact, never off this prose:
483
+ red → EVAL is not dispatched this round; `harness compile` turns each failing step into a
484
+ `payload.bugs` entry for round N+1, addressed to the scope whose substrate holds the
485
+ files the tool's output names (unowned → every scope, marked).
486
+ red → `reduce hill` withholds DOWNHILL_EXECUTION from every T0-green verdict of round N.
487
+ ```
488
+
466
489
  ## 4. EVAL → spec-evaluator (once per round)
467
490
  ```
468
491
  compile-order --operation evaluate --slug <slug> --worker spec-evaluator --round <r>
@@ -527,6 +550,7 @@ Read back: the proposed cut list + verdict (SHIP now | SHIP after fixing ship-bl
527
550
  | `harness-run.md` | **tech lead (sole writer)** | tech lead (round ledger + Hill + run-state), PO (audit) |
528
551
  | `scopes/<scope-id>.md` | `scope-architect` (sole writer) | tech lead (substrate/sequence), sandbox hook (write-whitelist), compile-order (inlined into orders) |
529
552
  | `t0/verdicts/r<N>-a<M>-t<T>.json` | `harness verify t0` (skill-local, mechanical — not a worker) | spec-evaluator (required citation), tech lead (hill derivation), compile-order (digested errors) |
553
+ | `build/r<N>-t<T>.json` (the round build gate) | `harness verify build` (mechanical — run_cmd + build_probe + launch_probe, once per round before EVAL) | tech lead (GATE L2 block), `harness reduce hill` (a red round moves no dot), compile-order (`payload.bugs` for round N+1) |
530
554
  | `t0/trials.jsonl` (the ratchet ledger, append-only, `baseline_trial` as the parent link) | `harness verify t0` (one row per attempt: score, status, delta, tree_ref) | compile-order (`trial_history` into the next order), ship-report (T0 + Ratchet sections), `harness probe stats --ratchet` |
531
555
  | `round-ledger.md` | **tech lead (sole writer)** | compile-order (decisions into every order), PO (audit) |
532
556
  | `hill/<scope-id>.yml` + `hill-chart.md` | **tech lead (sole writer)** | PO ("status without asking"), scope-hammer (H0 census) |
@@ -296,6 +296,27 @@
296
296
  "cardinality": "1:N",
297
297
  "via": "wiring-map entries[].wiring_seam",
298
298
  "note": "projected by reduce graph as UseCase -DEPENDS_ON-> Seam. NOTE: the projection reads `row.seam`/`row.entry_point` while WiringEntry declares `wiring_seam`/`entry_call_site` — see harness-defects"
299
+ },
300
+ {
301
+ "from": "RoundBuildVerdict",
302
+ "to": "ProjectProfile",
303
+ "cardinality": "N:1",
304
+ "via": "build_probe + launch_probe",
305
+ "note": "the gate runs the probes the profile declares; run_cmd comes from the run ledger"
306
+ },
307
+ {
308
+ "from": "RoundBuildVerdict",
309
+ "to": "AegisTriple",
310
+ "cardinality": "1:N",
311
+ "via": "discovered_tasks[]",
312
+ "note": "a red gate's output, digested — surfaces as payload.bugs on the next round's orders"
313
+ },
314
+ {
315
+ "from": "RoundBuildVerdict",
316
+ "to": "HillShard",
317
+ "cardinality": "1:N",
318
+ "via": "round",
319
+ "note": "a red gate withholds DOWNHILL_EXECUTION from every T0-green verdict of that round"
299
320
  }
300
321
  ]
301
322
  },
@@ -326,13 +347,15 @@
326
347
  "feature",
327
348
  "spec_folder",
328
349
  "tasks",
329
- "breadboard"
350
+ "breadboard",
351
+ "kb_rules_path"
330
352
  ],
331
353
  "solution-architect": [
332
354
  "feature",
333
355
  "spec_folder",
334
356
  "project_profile",
335
- "breadboard"
357
+ "breadboard",
358
+ "kb_rules_path"
336
359
  ],
337
360
  "spec-evaluator": [
338
361
  "spec_folder",
@@ -348,7 +371,8 @@
348
371
  "breadboard",
349
372
  "stack",
350
373
  "spec_folder",
351
- "feature"
374
+ "feature",
375
+ "kb_rules_path"
352
376
  ],
353
377
  "qa-edge-hunter": [
354
378
  "feature",
@@ -369,7 +393,8 @@
369
393
  "scope_id"
370
394
  ],
371
395
  "coach": [
372
- "feedback"
396
+ "feedback",
397
+ "stack"
373
398
  ]
374
399
  },
375
400
  "x-result-by-worker": {
@@ -458,7 +483,7 @@
458
483
  ]
459
484
  },
460
485
  "Operation": {
461
- "description": "The full operation vocabulary. Replaces lifecycle flags (--tasks-only, --from-discovered, --map-scopes …): the caller knows the pipeline position; the worker never re-derives it. Ownership: execute/fix/spike = task-executor · analyze/reconcile/retrofit-surface/coverage = ba-pitch-analyzer · map-scopes = scope-architect · wire = solution-architect · evaluate = spec-evaluator · orient = orient · hunt = qa-edge-hunter · translate = translator · hammer = scope-hammer · coach = coach.",
486
+ "description": "The full operation vocabulary. Replaces lifecycle flags (--tasks-only, --from-discovered, --map-scopes …): the caller knows the pipeline position; the worker never re-derives it. Ownership: execute/fix/spike = task-executor · analyze/reconcile/retrofit-surface/coverage = ba-pitch-analyzer · map-scopes = scope-architect · wire = solution-architect · evaluate = spec-evaluator · orient = orient · hunt = qa-edge-hunter · translate = translator · hammer = scope-hammer · coach/scan/research = coach (scan seeds the knowledge base from the project on disk instead of from L4 feedback; research seeds it from the platform's official documentation, aimed by a stack hint, and cross-checks the scan's rules where a scan exists; same write surface, same categorization gate).",
462
487
  "type": "string",
463
488
  "enum": [
464
489
  "execute",
@@ -475,7 +500,9 @@
475
500
  "hunt",
476
501
  "translate",
477
502
  "hammer",
478
- "coach"
503
+ "coach",
504
+ "scan",
505
+ "research"
479
506
  ]
480
507
  },
481
508
  "Substrate": {
@@ -1442,6 +1469,116 @@
1442
1469
  "status"
1443
1470
  ]
1444
1471
  },
1472
+ "RoundBuildVerdict": {
1473
+ "description": "The round build gate's verdict — did the FEATURE build and launch this round, measured before the judge is asked. A T0Artifact is one scope's fixtures inside its own substrate; this is the whole feature's run_cmd (from the run ledger), then the profile's build_probe and launch_probe, run in that order and stopping at the first failure. Zero LLM tokens. Two readers act on it: reduce hill withholds DOWNHILL_EXECUTION from every T0-green verdict of a round whose gate is red (a green fixture in a round the app did not compile is evidence about the fixture, not the scope), and harness compile turns each failing step into a `bug` for the next round's fix orders, addressed by the files the tool's output names. Immutable per gate run (`wx`, next ordinal); readers take the latest artifact for a round.",
1474
+ "x-tier": "LOCAL",
1475
+ "x-location": ".shapeup/<slug>/build/r<N>-t<T>.json",
1476
+ "x-writer": "harness verify build",
1477
+ "x-readers": "harness reduce hill (red rounds move no dot), harness compile (next round's payload.bugs), tech-lead (GATE L2 block: build gate state)",
1478
+ "type": "object",
1479
+ "required": [
1480
+ "schema_version",
1481
+ "round",
1482
+ "trial",
1483
+ "at",
1484
+ "overall",
1485
+ "steps"
1486
+ ],
1487
+ "properties": {
1488
+ "schema_version": {
1489
+ "type": "integer",
1490
+ "enum": [
1491
+ 1
1492
+ ]
1493
+ },
1494
+ "round": {
1495
+ "type": "integer"
1496
+ },
1497
+ "trial": {
1498
+ "type": "integer",
1499
+ "description": "Ordinal of this gate run within the round — a re-run after a hand fix lands beside its predecessor, never over it."
1500
+ },
1501
+ "run_id": {
1502
+ "type": "string",
1503
+ "pattern": "^[a-z0-9][a-z0-9-]*-[0-9]{8}T[0-9]{6}Z-[0-9a-f]{8}$"
1504
+ },
1505
+ "at": {
1506
+ "type": "string",
1507
+ "description": "ISO timestamp."
1508
+ },
1509
+ "archetype": {
1510
+ "type": [
1511
+ "string",
1512
+ "null"
1513
+ ],
1514
+ "description": "The profile's archetype, for the reader deciding whether a missing launch_probe matters."
1515
+ },
1516
+ "overall": {
1517
+ "type": "string",
1518
+ "enum": [
1519
+ "green",
1520
+ "red"
1521
+ ]
1522
+ },
1523
+ "steps": {
1524
+ "type": "array",
1525
+ "items": {
1526
+ "type": "object",
1527
+ "properties": {
1528
+ "kind": {
1529
+ "type": "string",
1530
+ "enum": [
1531
+ "run_cmd",
1532
+ "build_probe",
1533
+ "launch_probe"
1534
+ ]
1535
+ },
1536
+ "cmd": {
1537
+ "type": "string"
1538
+ },
1539
+ "exit": {
1540
+ "type": "integer"
1541
+ },
1542
+ "pass": {
1543
+ "type": "boolean"
1544
+ },
1545
+ "skipped": {
1546
+ "type": "boolean",
1547
+ "description": "true when an earlier step failed and this one was not run."
1548
+ },
1549
+ "stdout_tail": {
1550
+ "type": "string"
1551
+ },
1552
+ "stderr_tail": {
1553
+ "type": "string"
1554
+ },
1555
+ "error": {
1556
+ "type": "string",
1557
+ "description": "Spawn failure or timeout — the command did not run to completion, which is not the same fact as a non-zero exit."
1558
+ }
1559
+ },
1560
+ "required": [
1561
+ "kind",
1562
+ "cmd"
1563
+ ]
1564
+ }
1565
+ },
1566
+ "warnings": {
1567
+ "type": "array",
1568
+ "items": {
1569
+ "type": "string"
1570
+ },
1571
+ "description": "What the gate could not check and why — a launch-required archetype with no launch_probe, a ledger with no run_cmd."
1572
+ },
1573
+ "discovered_tasks": {
1574
+ "type": "array",
1575
+ "items": {
1576
+ "$ref": "#/$defs/AegisTriple"
1577
+ },
1578
+ "description": "The failing steps' output digested; populated only on red."
1579
+ }
1580
+ }
1581
+ },
1445
1582
  "SeesawRegistry": {
1446
1583
  "description": "The fixture registry of every FINISHED scope — what seesawCheck re-runs on each later attempt so a new scope cannot silently break a shipped one — a regression mistaken for progress is the pathology the seesaw exists for.",
1447
1584
  "x-tier": "LOCAL",
@@ -2155,7 +2292,7 @@
2155
2292
  "x-tier": "SHARED",
2156
2293
  "x-location": "shapeup/<slug>/project-profile.md",
2157
2294
  "x-writer": "tech-lead (GATE L0 — not harness compile, which stays pipeline-blind)",
2158
- "x-readers": "harness verify trace (reachability entry_point), solution-architect (wire), tech-lead",
2295
+ "x-readers": "harness verify trace (reachability entry_point), harness verify build (build_probe + launch_probe, once per round before EVAL), solution-architect (wire), tech-lead",
2159
2296
  "type": "object",
2160
2297
  "required": [
2161
2298
  "schema_version",
@@ -2187,6 +2324,14 @@
2187
2324
  "note": {
2188
2325
  "type": "string",
2189
2326
  "description": "Optional context on the archetype/entry-point choice."
2327
+ },
2328
+ "build_probe": {
2329
+ "type": "string",
2330
+ "description": "OPTIONAL — a command that asserts the BUILT ARTIFACT, not the build's exit code, and exits 0 only when it holds. Exists because a green build is not proof the feature compiled: some toolchains compile only the files reachable from an entry point, so a scope's new files can sit outside the compiled set while the build stays green (measured: an app package holding 3 compiled files, 58 errors once the rest became reachable). Archetype-specific by construction — e.g. 'the compiled source map lists every file under each scope's substrate'. Run by harness verify build after run_cmd, once per round before EVAL; absent = no such step, never a failure."
2331
+ },
2332
+ "launch_probe": {
2333
+ "type": "string",
2334
+ "description": "OPTIONAL for most archetypes, EXPECTED for `mobile` — a command that installs the built artifact, starts it, asserts the first screen (ids, text, a screenshot diff) and fails on fatal runtime log patterns. Exists because nothing else in the loop launches the app: a blank first screen survived three EVAL rounds on a run whose every T0 fixture was green. Run by harness verify build after run_cmd and build_probe; a mobile profile without one is warned about on every round so the 'on-device install unverified' risk has a named owner. Absent = no such step."
2190
2335
  }
2191
2336
  }
2192
2337
  },
@@ -2230,7 +2375,7 @@
2230
2375
  },
2231
2376
  "kb_rules_path": {
2232
2377
  "type": "string",
2233
- "description": "Coachable workers (task-executor, ba-pitch-analyzer, qa-edge-hunter): shapeup/knowledge-base/<skill>.md — steering, never spec; conflict → the AC wins, noted in deviations."
2378
+ "description": "Coachable workers (task-executor, ba-pitch-analyzer, qa-edge-hunter, orient, scope-architect, solution-architect): shapeup/knowledge-base/<skill>.md — steering, never spec and never a gate; conflict → the AC, the contract or the gate wins, noted in deviations. The judge and the census are not coachable."
2234
2379
  },
2235
2380
  "verify": {
2236
2381
  "$ref": "#/$defs/VerifySpec",
@@ -2275,7 +2420,7 @@
2275
2420
  },
2276
2421
  "stack": {
2277
2422
  "type": "string",
2278
- "description": "orient: stack hint aiming the code-surface sweeps (e.g. \"pnpm, Next 16 web :3000\")."
2423
+ "description": "orient: stack hint aiming the code-surface sweeps (e.g. \"pnpm, Next 16 web :3000\"). coach (research): the platform and toolchain the official-documentation research is aimed at — required, since a project with nothing on disk names no stack by itself."
2279
2424
  },
2280
2425
  "discovered_ledger": {
2281
2426
  "type": "string",
@@ -2,7 +2,7 @@
2
2
  //
3
3
  // WHAT THIS FILE OWNS.
4
4
  // ORIENT → GATE L1a → ANALYZE → WIRE → GATE L1a.5 → MAP SCOPES → GATE L1b →
5
- // rounds of (BUILD → GATE L2 → EVAL → GATE L3) bounded by budgets.maxRounds →
5
+ // rounds of (BUILD → build gate → GATE L2 → EVAL → GATE L3) bounded by budgets.maxRounds →
6
6
  // QA → GATE H → ship report → RunReturn.
7
7
  //
8
8
  // THE THREE PLANES THIS FILE RESPECTS, because collapsing them is what the previous version cost:
@@ -1350,17 +1350,54 @@ while (verdict !== "pass" && round <= maxRounds) {
1350
1350
  return withWarnings({ status: "gate_h", breaker: "inner", hammer_proposals: allHammer, green_scopes: allGreen });
1351
1351
  }
1352
1352
 
1353
+ // ---- ROUND BUILD GATE — the feature builds and launches, measured before anyone is asked --------
1354
+ //
1355
+ // T0 is per scope and inside the scope's substrate; nothing in it proves the FEATURE compiles or
1356
+ // starts. Measured on a live mobile run: 30/30 T0 trials green on the first try (TypeScript
1357
+ // stand-ins, structural greps, a suite wrapper that never compiled its sources) while the
1358
+ // ledger's own `run_cmd` failed, and three rounds of EVAL then graded a blank screen because
1359
+ // nothing in the loop had ever installed or launched the app. `verify build` runs the ledger's
1360
+ // `run_cmd`, then the profile's `build_probe` and `launch_probe`, and writes one artifact
1361
+ // `reduce hill` and `harness compile` both read — so a red gate moves no dot downhill and becomes
1362
+ // the next round's bug list without this script carrying anything across the round boundary.
1363
+ //
1364
+ // Exit 3 is "nothing declared" — the honest state of a run whose L0 pinned no run command and
1365
+ // whose profile names no probe — and it is logged, not treated as green. Any other non-0/1 exit
1366
+ // means the gate itself did not run, and the run says so rather than inferring a verdict.
1367
+ const gate = await cmd(`verify build --slug ${slug} --round ${round}`, "Build", `build-gate:r${round}`);
1368
+ let buildGate = "green";
1369
+ if (gate.exit_code === 1) {
1370
+ buildGate = "red";
1371
+ log(`BUILD GATE r${round} — RED${gate.detail ? `: ${gate.detail}` : ""}. EVAL will not run over a feature ` +
1372
+ `that does not build or launch; the failing step is compiled into round ${round + 1}'s orders as bugs.`);
1373
+ } else if (gate.exit_code === 3) {
1374
+ buildGate = "undeclared";
1375
+ log(`BUILD GATE r${round} — nothing declared: no run_cmd in the run ledger and no build_probe or ` +
1376
+ `launch_probe in project-profile.md. EVAL runs over an unproven build; pin them at GATE L0.`);
1377
+ } else if (gate.exit_code !== 0) {
1378
+ buildGate = "unknown";
1379
+ log(`BUILD GATE r${round} — the gate itself did not run (exit ${gate.exit_code}` +
1380
+ `${gate.detail ? `: ${gate.detail}` : ""}). Treated as undeclared, not as green.`);
1381
+ }
1382
+
1353
1383
  await advisory(`reduce hill --slug ${slug}`, "Build", "hill-derive");
1354
1384
  {
1355
1385
  const g = await crossGate("L2", "Build", ["proceed", "ask", "abort"],
1356
- { round, green_scopes: roundGreen, hammer_proposals: roundHammer });
1386
+ { round, green_scopes: roundGreen, hammer_proposals: roundHammer, build_gate: buildGate });
1357
1387
  if (g.stop) return withWarnings(g.stop);
1358
1388
  }
1359
1389
 
1360
1390
  // ---- EVAL — exactly one feature-level pass per round (the single-judge invariant) ------------
1361
1391
  phase("Eval");
1362
1392
  await setRunStatus("evaluating", "Eval");
1363
- if (args.noEval) {
1393
+ if (buildGate === "red") {
1394
+ // A red gate outranks --no-eval: skipping the judge is the operator's call, but the build failing
1395
+ // is a measured fact, and a round that does not compile has nothing for anyone to pass.
1396
+ verdict = "fail";
1397
+ findings = [];
1398
+ log(`EVAL r${round} — not dispatched: the round build gate is red. The judge grades a running ` +
1399
+ `feature; this one does not build or launch. Round ${round + 1} fixes the gate's failing step.`);
1400
+ } else if (args.noEval) {
1364
1401
  log("EVAL — skipped (--no-eval)");
1365
1402
  verdict = "pass";
1366
1403
  } else {
@@ -1410,7 +1447,7 @@ while (verdict !== "pass" && round <= maxRounds) {
1410
1447
 
1411
1448
  await advisory(`reduce graph --slug ${slug}`, "Eval", `graph:eval-r${round}`);
1412
1449
  await advisory(`reduce hill --slug ${slug}`, "Eval", "hill-derive");
1413
- const g3 = await crossGate("L3", "Eval", ["loop", "stop", "ask"], { round, verdict });
1450
+ const g3 = await crossGate("L3", "Eval", ["loop", "stop", "ask"], { round, verdict, build_gate: buildGate });
1414
1451
  if (g3.stop) return withWarnings(g3.stop);
1415
1452
 
1416
1453
  if (verdict === "pass") break; // → QA → GATE H → ship