shapeup-sdlc 3.2.0 → 3.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude-plugin/plugin.json +1 -1
- package/AGENTS.md +6 -5
- package/README.md +1 -1
- package/SECURITY.md +4 -1
- package/bin/init.mjs +3 -0
- package/commands/retro.md +19 -2
- package/hooks/dispatch-receipt.mjs +6 -3
- package/hooks/gate-zerowork.mjs +5 -2
- package/hooks/lib/decision.mjs +49 -6
- package/hooks/safety-spine.mjs +8 -5
- package/hooks/sandbox-guard.mjs +13 -6
- package/kernel/compile.mjs +112 -6
- package/kernel/harness.mjs +11 -5
- package/kernel/lib/contract.mjs +20 -4
- package/kernel/lib/paths.mjs +12 -2
- package/kernel/probe/owner.mjs +139 -0
- package/kernel/probe/stats.mjs +49 -2
- package/kernel/reduce/board.mjs +26 -2
- package/kernel/reduce/graph.mjs +26 -12
- package/kernel/reduce/hill.mjs +12 -2
- package/kernel/verify/build.mjs +319 -0
- package/package.json +1 -1
- package/skills/coach/SKILL.md +232 -43
- package/skills/orient/SKILL.md +8 -1
- package/skills/qa-edge-hunter/SKILL.md +3 -2
- package/skills/scope-architect/SKILL.md +3 -0
- package/skills/scope-hammer/SKILL.md +13 -0
- package/skills/solution-architect/SKILL.md +3 -1
- package/skills/tech-lead/SKILL.md +1 -1
- package/skills/tech-lead/references/gates.md +43 -5
- package/skills/tech-lead/references/protocol.md +25 -1
- package/skills/tech-lead/schemas/domain.schema.json +154 -9
- package/skills/tech-lead/workflows/shapeup-run.js +85 -13
|
@@ -68,7 +68,9 @@ approval of the new tasks and estimates before resuming the BUILD loop.
|
|
|
68
68
|
|
|
69
69
|
## The EVAL timing rule (the core constraint)
|
|
70
70
|
|
|
71
|
-
EVAL fires **once** per round and **only** when GATE L2 has confirmed the board is 100% done
|
|
71
|
+
EVAL fires **once** per round and **only** when GATE L2 has confirmed the board is 100% done
|
|
72
|
+
and the round build gate (§3d) is not red — a feature that does not build or launch has nothing
|
|
73
|
+
for a judge to grade, and the gate's failing step is what round r+1 fixes.
|
|
72
74
|
It is never:
|
|
73
75
|
- called per task,
|
|
74
76
|
- called inside the BUILD loop,
|
|
@@ -463,6 +465,27 @@ Read back: the stdout JSON — {path, sha256, trial, overall, regression, score,
|
|
|
463
465
|
(harness verify t0 calls its sibling harness probe digest internally on failure).
|
|
464
466
|
```
|
|
465
467
|
|
|
468
|
+
## 3d. Round build gate → `harness verify build` (once per round, before EVAL)
|
|
469
|
+
```
|
|
470
|
+
Invoke via Bash directly — deterministic tooling, not a worker:
|
|
471
|
+
node "${CLAUDE_PLUGIN_ROOT}/kernel/harness.mjs" verify build --slug <slug> --round <N>
|
|
472
|
+
Effect: runs, in order and stopping at the first failure, the ledger's `run_cmd` (the build), then
|
|
473
|
+
project-profile.md's `build_probe` (the built artifact covers what the run wrote — a green
|
|
474
|
+
exit code is not proof the feature compiled when the toolchain compiles only what an entry
|
|
475
|
+
point reaches) and `launch_probe` (install, start, assert the first screen, fail on fatal
|
|
476
|
+
logs). Writes .shapeup/<slug>/build/r<N>-t<T>.json, immutable per run of the gate.
|
|
477
|
+
Read back: stdout JSON — {overall: green|red, steps[], warnings[], failed_step?, stderr_tail?}.
|
|
478
|
+
Exit 0 green · 1 red · 3 nothing declared (no run_cmd, no probes — logged, never green).
|
|
479
|
+
A `mobile` profile with no launch_probe is warned about on stderr every round, and so is
|
|
480
|
+
every scope none of whose fixtures invoke the tool run_cmd builds with — advisory, because a
|
|
481
|
+
green T0 from such fixtures is not evidence the scope compiles.
|
|
482
|
+
Consequences, both mechanical and both read off the artifact, never off this prose:
|
|
483
|
+
red → EVAL is not dispatched this round; `harness compile` turns each failing step into a
|
|
484
|
+
`payload.bugs` entry for round N+1, addressed to the scope whose substrate holds the
|
|
485
|
+
files the tool's output names (unowned → every scope, marked).
|
|
486
|
+
red → `reduce hill` withholds DOWNHILL_EXECUTION from every T0-green verdict of round N.
|
|
487
|
+
```
|
|
488
|
+
|
|
466
489
|
## 4. EVAL → spec-evaluator (once per round)
|
|
467
490
|
```
|
|
468
491
|
compile-order --operation evaluate --slug <slug> --worker spec-evaluator --round <r>
|
|
@@ -527,6 +550,7 @@ Read back: the proposed cut list + verdict (SHIP now | SHIP after fixing ship-bl
|
|
|
527
550
|
| `harness-run.md` | **tech lead (sole writer)** | tech lead (round ledger + Hill + run-state), PO (audit) |
|
|
528
551
|
| `scopes/<scope-id>.md` | `scope-architect` (sole writer) | tech lead (substrate/sequence), sandbox hook (write-whitelist), compile-order (inlined into orders) |
|
|
529
552
|
| `t0/verdicts/r<N>-a<M>-t<T>.json` | `harness verify t0` (skill-local, mechanical — not a worker) | spec-evaluator (required citation), tech lead (hill derivation), compile-order (digested errors) |
|
|
553
|
+
| `build/r<N>-t<T>.json` (the round build gate) | `harness verify build` (mechanical — run_cmd + build_probe + launch_probe, once per round before EVAL) | tech lead (GATE L2 block), `harness reduce hill` (a red round moves no dot), compile-order (`payload.bugs` for round N+1) |
|
|
530
554
|
| `t0/trials.jsonl` (the ratchet ledger, append-only, `baseline_trial` as the parent link) | `harness verify t0` (one row per attempt: score, status, delta, tree_ref) | compile-order (`trial_history` into the next order), ship-report (T0 + Ratchet sections), `harness probe stats --ratchet` |
|
|
531
555
|
| `round-ledger.md` | **tech lead (sole writer)** | compile-order (decisions into every order), PO (audit) |
|
|
532
556
|
| `hill/<scope-id>.yml` + `hill-chart.md` | **tech lead (sole writer)** | PO ("status without asking"), scope-hammer (H0 census) |
|
|
@@ -296,6 +296,27 @@
|
|
|
296
296
|
"cardinality": "1:N",
|
|
297
297
|
"via": "wiring-map entries[].wiring_seam",
|
|
298
298
|
"note": "projected by reduce graph as UseCase -DEPENDS_ON-> Seam. NOTE: the projection reads `row.seam`/`row.entry_point` while WiringEntry declares `wiring_seam`/`entry_call_site` — see harness-defects"
|
|
299
|
+
},
|
|
300
|
+
{
|
|
301
|
+
"from": "RoundBuildVerdict",
|
|
302
|
+
"to": "ProjectProfile",
|
|
303
|
+
"cardinality": "N:1",
|
|
304
|
+
"via": "build_probe + launch_probe",
|
|
305
|
+
"note": "the gate runs the probes the profile declares; run_cmd comes from the run ledger"
|
|
306
|
+
},
|
|
307
|
+
{
|
|
308
|
+
"from": "RoundBuildVerdict",
|
|
309
|
+
"to": "AegisTriple",
|
|
310
|
+
"cardinality": "1:N",
|
|
311
|
+
"via": "discovered_tasks[]",
|
|
312
|
+
"note": "a red gate's output, digested — surfaces as payload.bugs on the next round's orders"
|
|
313
|
+
},
|
|
314
|
+
{
|
|
315
|
+
"from": "RoundBuildVerdict",
|
|
316
|
+
"to": "HillShard",
|
|
317
|
+
"cardinality": "1:N",
|
|
318
|
+
"via": "round",
|
|
319
|
+
"note": "a red gate withholds DOWNHILL_EXECUTION from every T0-green verdict of that round"
|
|
299
320
|
}
|
|
300
321
|
]
|
|
301
322
|
},
|
|
@@ -326,13 +347,15 @@
|
|
|
326
347
|
"feature",
|
|
327
348
|
"spec_folder",
|
|
328
349
|
"tasks",
|
|
329
|
-
"breadboard"
|
|
350
|
+
"breadboard",
|
|
351
|
+
"kb_rules_path"
|
|
330
352
|
],
|
|
331
353
|
"solution-architect": [
|
|
332
354
|
"feature",
|
|
333
355
|
"spec_folder",
|
|
334
356
|
"project_profile",
|
|
335
|
-
"breadboard"
|
|
357
|
+
"breadboard",
|
|
358
|
+
"kb_rules_path"
|
|
336
359
|
],
|
|
337
360
|
"spec-evaluator": [
|
|
338
361
|
"spec_folder",
|
|
@@ -348,7 +371,8 @@
|
|
|
348
371
|
"breadboard",
|
|
349
372
|
"stack",
|
|
350
373
|
"spec_folder",
|
|
351
|
-
"feature"
|
|
374
|
+
"feature",
|
|
375
|
+
"kb_rules_path"
|
|
352
376
|
],
|
|
353
377
|
"qa-edge-hunter": [
|
|
354
378
|
"feature",
|
|
@@ -369,7 +393,8 @@
|
|
|
369
393
|
"scope_id"
|
|
370
394
|
],
|
|
371
395
|
"coach": [
|
|
372
|
-
"feedback"
|
|
396
|
+
"feedback",
|
|
397
|
+
"stack"
|
|
373
398
|
]
|
|
374
399
|
},
|
|
375
400
|
"x-result-by-worker": {
|
|
@@ -458,7 +483,7 @@
|
|
|
458
483
|
]
|
|
459
484
|
},
|
|
460
485
|
"Operation": {
|
|
461
|
-
"description": "The full operation vocabulary. Replaces lifecycle flags (--tasks-only, --from-discovered, --map-scopes …): the caller knows the pipeline position; the worker never re-derives it. Ownership: execute/fix/spike = task-executor · analyze/reconcile/retrofit-surface/coverage = ba-pitch-analyzer · map-scopes = scope-architect · wire = solution-architect · evaluate = spec-evaluator · orient = orient · hunt = qa-edge-hunter · translate = translator · hammer = scope-hammer · coach = coach.",
|
|
486
|
+
"description": "The full operation vocabulary. Replaces lifecycle flags (--tasks-only, --from-discovered, --map-scopes …): the caller knows the pipeline position; the worker never re-derives it. Ownership: execute/fix/spike = task-executor · analyze/reconcile/retrofit-surface/coverage = ba-pitch-analyzer · map-scopes = scope-architect · wire = solution-architect · evaluate = spec-evaluator · orient = orient · hunt = qa-edge-hunter · translate = translator · hammer = scope-hammer · coach/scan/research = coach (scan seeds the knowledge base from the project on disk instead of from L4 feedback; research seeds it from the platform's official documentation, aimed by a stack hint, and cross-checks the scan's rules where a scan exists; same write surface, same categorization gate).",
|
|
462
487
|
"type": "string",
|
|
463
488
|
"enum": [
|
|
464
489
|
"execute",
|
|
@@ -475,7 +500,9 @@
|
|
|
475
500
|
"hunt",
|
|
476
501
|
"translate",
|
|
477
502
|
"hammer",
|
|
478
|
-
"coach"
|
|
503
|
+
"coach",
|
|
504
|
+
"scan",
|
|
505
|
+
"research"
|
|
479
506
|
]
|
|
480
507
|
},
|
|
481
508
|
"Substrate": {
|
|
@@ -1442,6 +1469,116 @@
|
|
|
1442
1469
|
"status"
|
|
1443
1470
|
]
|
|
1444
1471
|
},
|
|
1472
|
+
"RoundBuildVerdict": {
|
|
1473
|
+
"description": "The round build gate's verdict — did the FEATURE build and launch this round, measured before the judge is asked. A T0Artifact is one scope's fixtures inside its own substrate; this is the whole feature's run_cmd (from the run ledger), then the profile's build_probe and launch_probe, run in that order and stopping at the first failure. Zero LLM tokens. Two readers act on it: reduce hill withholds DOWNHILL_EXECUTION from every T0-green verdict of a round whose gate is red (a green fixture in a round the app did not compile is evidence about the fixture, not the scope), and harness compile turns each failing step into a `bug` for the next round's fix orders, addressed by the files the tool's output names. Immutable per gate run (`wx`, next ordinal); readers take the latest artifact for a round.",
|
|
1474
|
+
"x-tier": "LOCAL",
|
|
1475
|
+
"x-location": ".shapeup/<slug>/build/r<N>-t<T>.json",
|
|
1476
|
+
"x-writer": "harness verify build",
|
|
1477
|
+
"x-readers": "harness reduce hill (red rounds move no dot), harness compile (next round's payload.bugs), tech-lead (GATE L2 block: build gate state)",
|
|
1478
|
+
"type": "object",
|
|
1479
|
+
"required": [
|
|
1480
|
+
"schema_version",
|
|
1481
|
+
"round",
|
|
1482
|
+
"trial",
|
|
1483
|
+
"at",
|
|
1484
|
+
"overall",
|
|
1485
|
+
"steps"
|
|
1486
|
+
],
|
|
1487
|
+
"properties": {
|
|
1488
|
+
"schema_version": {
|
|
1489
|
+
"type": "integer",
|
|
1490
|
+
"enum": [
|
|
1491
|
+
1
|
|
1492
|
+
]
|
|
1493
|
+
},
|
|
1494
|
+
"round": {
|
|
1495
|
+
"type": "integer"
|
|
1496
|
+
},
|
|
1497
|
+
"trial": {
|
|
1498
|
+
"type": "integer",
|
|
1499
|
+
"description": "Ordinal of this gate run within the round — a re-run after a hand fix lands beside its predecessor, never over it."
|
|
1500
|
+
},
|
|
1501
|
+
"run_id": {
|
|
1502
|
+
"type": "string",
|
|
1503
|
+
"pattern": "^[a-z0-9][a-z0-9-]*-[0-9]{8}T[0-9]{6}Z-[0-9a-f]{8}$"
|
|
1504
|
+
},
|
|
1505
|
+
"at": {
|
|
1506
|
+
"type": "string",
|
|
1507
|
+
"description": "ISO timestamp."
|
|
1508
|
+
},
|
|
1509
|
+
"archetype": {
|
|
1510
|
+
"type": [
|
|
1511
|
+
"string",
|
|
1512
|
+
"null"
|
|
1513
|
+
],
|
|
1514
|
+
"description": "The profile's archetype, for the reader deciding whether a missing launch_probe matters."
|
|
1515
|
+
},
|
|
1516
|
+
"overall": {
|
|
1517
|
+
"type": "string",
|
|
1518
|
+
"enum": [
|
|
1519
|
+
"green",
|
|
1520
|
+
"red"
|
|
1521
|
+
]
|
|
1522
|
+
},
|
|
1523
|
+
"steps": {
|
|
1524
|
+
"type": "array",
|
|
1525
|
+
"items": {
|
|
1526
|
+
"type": "object",
|
|
1527
|
+
"properties": {
|
|
1528
|
+
"kind": {
|
|
1529
|
+
"type": "string",
|
|
1530
|
+
"enum": [
|
|
1531
|
+
"run_cmd",
|
|
1532
|
+
"build_probe",
|
|
1533
|
+
"launch_probe"
|
|
1534
|
+
]
|
|
1535
|
+
},
|
|
1536
|
+
"cmd": {
|
|
1537
|
+
"type": "string"
|
|
1538
|
+
},
|
|
1539
|
+
"exit": {
|
|
1540
|
+
"type": "integer"
|
|
1541
|
+
},
|
|
1542
|
+
"pass": {
|
|
1543
|
+
"type": "boolean"
|
|
1544
|
+
},
|
|
1545
|
+
"skipped": {
|
|
1546
|
+
"type": "boolean",
|
|
1547
|
+
"description": "true when an earlier step failed and this one was not run."
|
|
1548
|
+
},
|
|
1549
|
+
"stdout_tail": {
|
|
1550
|
+
"type": "string"
|
|
1551
|
+
},
|
|
1552
|
+
"stderr_tail": {
|
|
1553
|
+
"type": "string"
|
|
1554
|
+
},
|
|
1555
|
+
"error": {
|
|
1556
|
+
"type": "string",
|
|
1557
|
+
"description": "Spawn failure or timeout — the command did not run to completion, which is not the same fact as a non-zero exit."
|
|
1558
|
+
}
|
|
1559
|
+
},
|
|
1560
|
+
"required": [
|
|
1561
|
+
"kind",
|
|
1562
|
+
"cmd"
|
|
1563
|
+
]
|
|
1564
|
+
}
|
|
1565
|
+
},
|
|
1566
|
+
"warnings": {
|
|
1567
|
+
"type": "array",
|
|
1568
|
+
"items": {
|
|
1569
|
+
"type": "string"
|
|
1570
|
+
},
|
|
1571
|
+
"description": "What the gate could not check and why — a launch-required archetype with no launch_probe, a ledger with no run_cmd."
|
|
1572
|
+
},
|
|
1573
|
+
"discovered_tasks": {
|
|
1574
|
+
"type": "array",
|
|
1575
|
+
"items": {
|
|
1576
|
+
"$ref": "#/$defs/AegisTriple"
|
|
1577
|
+
},
|
|
1578
|
+
"description": "The failing steps' output digested; populated only on red."
|
|
1579
|
+
}
|
|
1580
|
+
}
|
|
1581
|
+
},
|
|
1445
1582
|
"SeesawRegistry": {
|
|
1446
1583
|
"description": "The fixture registry of every FINISHED scope — what seesawCheck re-runs on each later attempt so a new scope cannot silently break a shipped one — a regression mistaken for progress is the pathology the seesaw exists for.",
|
|
1447
1584
|
"x-tier": "LOCAL",
|
|
@@ -2155,7 +2292,7 @@
|
|
|
2155
2292
|
"x-tier": "SHARED",
|
|
2156
2293
|
"x-location": "shapeup/<slug>/project-profile.md",
|
|
2157
2294
|
"x-writer": "tech-lead (GATE L0 — not harness compile, which stays pipeline-blind)",
|
|
2158
|
-
"x-readers": "harness verify trace (reachability entry_point), solution-architect (wire), tech-lead",
|
|
2295
|
+
"x-readers": "harness verify trace (reachability entry_point), harness verify build (build_probe + launch_probe, once per round before EVAL), solution-architect (wire), tech-lead",
|
|
2159
2296
|
"type": "object",
|
|
2160
2297
|
"required": [
|
|
2161
2298
|
"schema_version",
|
|
@@ -2187,6 +2324,14 @@
|
|
|
2187
2324
|
"note": {
|
|
2188
2325
|
"type": "string",
|
|
2189
2326
|
"description": "Optional context on the archetype/entry-point choice."
|
|
2327
|
+
},
|
|
2328
|
+
"build_probe": {
|
|
2329
|
+
"type": "string",
|
|
2330
|
+
"description": "OPTIONAL — a command that asserts the BUILT ARTIFACT, not the build's exit code, and exits 0 only when it holds. Exists because a green build is not proof the feature compiled: some toolchains compile only the files reachable from an entry point, so a scope's new files can sit outside the compiled set while the build stays green (measured: an app package holding 3 compiled files, 58 errors once the rest became reachable). Archetype-specific by construction — e.g. 'the compiled source map lists every file under each scope's substrate'. Run by harness verify build after run_cmd, once per round before EVAL; absent = no such step, never a failure."
|
|
2331
|
+
},
|
|
2332
|
+
"launch_probe": {
|
|
2333
|
+
"type": "string",
|
|
2334
|
+
"description": "OPTIONAL for most archetypes, EXPECTED for `mobile` — a command that installs the built artifact, starts it, asserts the first screen (ids, text, a screenshot diff) and fails on fatal runtime log patterns. Exists because nothing else in the loop launches the app: a blank first screen survived three EVAL rounds on a run whose every T0 fixture was green. Run by harness verify build after run_cmd and build_probe; a mobile profile without one is warned about on every round so the 'on-device install unverified' risk has a named owner. Absent = no such step."
|
|
2190
2335
|
}
|
|
2191
2336
|
}
|
|
2192
2337
|
},
|
|
@@ -2230,7 +2375,7 @@
|
|
|
2230
2375
|
},
|
|
2231
2376
|
"kb_rules_path": {
|
|
2232
2377
|
"type": "string",
|
|
2233
|
-
"description": "Coachable workers (task-executor, ba-pitch-analyzer, qa-edge-hunter): shapeup/knowledge-base/<skill>.md — steering, never spec; conflict → the AC wins, noted in deviations."
|
|
2378
|
+
"description": "Coachable workers (task-executor, ba-pitch-analyzer, qa-edge-hunter, orient, scope-architect, solution-architect): shapeup/knowledge-base/<skill>.md — steering, never spec and never a gate; conflict → the AC, the contract or the gate wins, noted in deviations. The judge and the census are not coachable."
|
|
2234
2379
|
},
|
|
2235
2380
|
"verify": {
|
|
2236
2381
|
"$ref": "#/$defs/VerifySpec",
|
|
@@ -2275,7 +2420,7 @@
|
|
|
2275
2420
|
},
|
|
2276
2421
|
"stack": {
|
|
2277
2422
|
"type": "string",
|
|
2278
|
-
"description": "orient: stack hint aiming the code-surface sweeps (e.g. \"pnpm, Next 16 web :3000\")."
|
|
2423
|
+
"description": "orient: stack hint aiming the code-surface sweeps (e.g. \"pnpm, Next 16 web :3000\"). coach (research): the platform and toolchain the official-documentation research is aimed at — required, since a project with nothing on disk names no stack by itself."
|
|
2279
2424
|
},
|
|
2280
2425
|
"discovered_ledger": {
|
|
2281
2426
|
"type": "string",
|
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
//
|
|
3
3
|
// WHAT THIS FILE OWNS.
|
|
4
4
|
// ORIENT → GATE L1a → ANALYZE → WIRE → GATE L1a.5 → MAP SCOPES → GATE L1b →
|
|
5
|
-
// rounds of (BUILD → GATE L2 → EVAL → GATE L3) bounded by budgets.maxRounds →
|
|
5
|
+
// rounds of (BUILD → build gate → GATE L2 → EVAL → GATE L3) bounded by budgets.maxRounds →
|
|
6
6
|
// QA → GATE H → ship report → RunReturn.
|
|
7
7
|
//
|
|
8
8
|
// THE THREE PLANES THIS FILE RESPECTS, because collapsing them is what the previous version cost:
|
|
@@ -1193,6 +1193,17 @@ await advisory(`reduce hill --slug ${slug}`, "MapScopes", "hill-derive");
|
|
|
1193
1193
|
const lastEval = rs.eval_rounds_done?.length ? Math.max(...rs.eval_rounds_done) : 0;
|
|
1194
1194
|
let round = lastEval + 1;
|
|
1195
1195
|
let verdict = null;
|
|
1196
|
+
// DERIVED FROM DISK AT EVERY LAUNCH, never accumulated only in memory. These two lists ARE GATE H's
|
|
1197
|
+
// census: `scope-hammer` is dispatched with `hammer_proposals`, and AGENTS.md makes a scope that
|
|
1198
|
+
// exhausted its attempt budget a queued GATE H proposal. As bare `const []` they were emptied by the
|
|
1199
|
+
// one event that most needs them intact — a gate PAUSE is a `return`, so the PO's answer is followed
|
|
1200
|
+
// by a FRESH LAUNCH whose accumulators start empty and whose round loop is then fast-forwarded past
|
|
1201
|
+
// the rounds that filled them. The PO was handed an empty cut list for a run that had genuinely
|
|
1202
|
+
// exhausted scopes. Unattended (`ci`) runs never pause, which is why no archived trace shows it.
|
|
1203
|
+
//
|
|
1204
|
+
// This file has now paid for the same class three times — `findings` in the temporal dead zone and
|
|
1205
|
+
// `payload.bugs` "lived in a variable, which a relaunch between two rounds resets to empty" are the
|
|
1206
|
+
// other two. The cure is the same each time and it is not a bigger variable: re-derive the fact.
|
|
1196
1207
|
const allGreen = [];
|
|
1197
1208
|
const allHammer = [];
|
|
1198
1209
|
// OUTSIDE the loop, because its whole purpose is to cross a round boundary: round r's verdict is
|
|
@@ -1248,6 +1259,16 @@ while (verdict !== "pass" && round <= maxRounds) {
|
|
|
1248
1259
|
const alreadyGreen = new Set(g?.green_scopes_by_round?.[String(round)] || []);
|
|
1249
1260
|
if (alreadyGreen.size) log(`BUILD r${round} — ${alreadyGreen.size} scope(s) already green in the graph, skipping them`);
|
|
1250
1261
|
|
|
1262
|
+
// REBUILD THE CENSUS FROM THE GRAPH THIS QUERY JUST RETURNED. Everything a prior launch learned is
|
|
1263
|
+
// already on disk: a scope green in ANY round is green work, and a scope the run has touched but
|
|
1264
|
+
// never got green is what GATE H has to be shown. One query, already made, re-read — so a relaunch
|
|
1265
|
+
// after a paused gate carries the same census the launch that paused it would have.
|
|
1266
|
+
const greenEver = new Set(Object.values(g?.green_scopes_by_round || {}).flat());
|
|
1267
|
+
for (const sid of greenEver) if (!allGreen.includes(sid)) allGreen.push(sid);
|
|
1268
|
+
for (const sid of (g?.scopes || [])) {
|
|
1269
|
+
if (!greenEver.has(sid) && !allHammer.includes(sid) && scopes.some((x) => x.scope_id === sid)) allHammer.push(sid);
|
|
1270
|
+
}
|
|
1271
|
+
|
|
1251
1272
|
// SCOPES FAN OUT. A scope contract is the definition of an independent subtask — disjoint
|
|
1252
1273
|
// substrate, own fixtures, own ratchet — so the loop that ran them one at a time was leaving the
|
|
1253
1274
|
// whole point of the contract on the floor. `pipeline()` has NO barrier between its stages: a
|
|
@@ -1281,13 +1302,25 @@ while (verdict !== "pass" && round <= maxRounds) {
|
|
|
1281
1302
|
async (pre, s) => (pre?.pending ? buildScope(s, round) : pre),
|
|
1282
1303
|
async (res, s) => {
|
|
1283
1304
|
if (!res || res.__failed) return res;
|
|
1284
|
-
if (
|
|
1285
|
-
|
|
1286
|
-
|
|
1287
|
-
|
|
1288
|
-
|
|
1289
|
-
|
|
1290
|
-
|
|
1305
|
+
if (!res.green) return res;
|
|
1306
|
+
// THE T0 RE-READ IS SKIPPED FOR A RESUMED SCOPE; THE LEG CHECK BELOW IS NOT.
|
|
1307
|
+
//
|
|
1308
|
+
// `resumed` means the graph already reported this scope green for this round, so re-reading
|
|
1309
|
+
// its T0 artifact would only confirm what the resume derivation just read. But these two
|
|
1310
|
+
// stages answer DIFFERENT questions, and the second one is precisely the question a resumed
|
|
1311
|
+
// scope is most likely to fail: a relaunch happens because the previous launch DIED, and a
|
|
1312
|
+
// leg that died between writing its result and running `reduce ingest` leaves exactly this
|
|
1313
|
+
// state — green T0 on disk, result never applied, board still `pending`. The short-circuit
|
|
1314
|
+
// used to cover both stages, so the one scope class known to be at risk was the one class
|
|
1315
|
+
// nobody asked, and the late-ingest repair nine lines below could never fire for it.
|
|
1316
|
+
if (!res.resumed) {
|
|
1317
|
+
const confirmed = await query(`probe t0 --slug ${slug} --scope ${s.scope_id} --round ${round}`,
|
|
1318
|
+
T0CHECK, "Build", `t0confirm:${s.scope_id}-r${round}`);
|
|
1319
|
+
if (!confirmed?.green) {
|
|
1320
|
+
log(`BUILD r${round} — ${s.scope_id} reported green but no T0 verdict is on disk for this ` +
|
|
1321
|
+
`round; treating it as not green (the evaluator cites that artifact, and it is not there).`);
|
|
1322
|
+
return { ...res, green: false, reason: "reported green with no T0 verdict artifact on disk" };
|
|
1323
|
+
}
|
|
1291
1324
|
}
|
|
1292
1325
|
// AND ITS RESULT HAS TO HAVE REACHED THE BOARD. A green T0 says the worker's fixtures ran and
|
|
1293
1326
|
// passed; it says nothing about whether the WorkResult was applied, and this stage used to ask
|
|
@@ -1341,8 +1374,10 @@ while (verdict !== "pass" && round <= maxRounds) {
|
|
|
1341
1374
|
}
|
|
1342
1375
|
}
|
|
1343
1376
|
|
|
1344
|
-
allGreen.push(
|
|
1345
|
-
|
|
1377
|
+
for (const sid of roundGreen) if (!allGreen.includes(sid)) allGreen.push(sid);
|
|
1378
|
+
// A scope that went green this round is no longer a cut candidate, however it was queued earlier.
|
|
1379
|
+
for (const sid of roundGreen) { const i = allHammer.indexOf(sid); if (i !== -1) allHammer.splice(i, 1); }
|
|
1380
|
+
for (const sid of roundHammer) if (!allHammer.includes(sid) && !allGreen.includes(sid)) allHammer.push(sid);
|
|
1346
1381
|
|
|
1347
1382
|
// INNER breaker: nothing green and something queued → GATE H. The census is scope-hammer's job.
|
|
1348
1383
|
if (roundGreen.length === 0 && roundHammer.length > 0) {
|
|
@@ -1350,17 +1385,54 @@ while (verdict !== "pass" && round <= maxRounds) {
|
|
|
1350
1385
|
return withWarnings({ status: "gate_h", breaker: "inner", hammer_proposals: allHammer, green_scopes: allGreen });
|
|
1351
1386
|
}
|
|
1352
1387
|
|
|
1388
|
+
// ---- ROUND BUILD GATE — the feature builds and launches, measured before anyone is asked --------
|
|
1389
|
+
//
|
|
1390
|
+
// T0 is per scope and inside the scope's substrate; nothing in it proves the FEATURE compiles or
|
|
1391
|
+
// starts. Measured on a live mobile run: 30/30 T0 trials green on the first try (TypeScript
|
|
1392
|
+
// stand-ins, structural greps, a suite wrapper that never compiled its sources) while the
|
|
1393
|
+
// ledger's own `run_cmd` failed, and three rounds of EVAL then graded a blank screen because
|
|
1394
|
+
// nothing in the loop had ever installed or launched the app. `verify build` runs the ledger's
|
|
1395
|
+
// `run_cmd`, then the profile's `build_probe` and `launch_probe`, and writes one artifact
|
|
1396
|
+
// `reduce hill` and `harness compile` both read — so a red gate moves no dot downhill and becomes
|
|
1397
|
+
// the next round's bug list without this script carrying anything across the round boundary.
|
|
1398
|
+
//
|
|
1399
|
+
// Exit 3 is "nothing declared" — the honest state of a run whose L0 pinned no run command and
|
|
1400
|
+
// whose profile names no probe — and it is logged, not treated as green. Any other non-0/1 exit
|
|
1401
|
+
// means the gate itself did not run, and the run says so rather than inferring a verdict.
|
|
1402
|
+
const gate = await cmd(`verify build --slug ${slug} --round ${round}`, "Build", `build-gate:r${round}`);
|
|
1403
|
+
let buildGate = "green";
|
|
1404
|
+
if (gate.exit_code === 1) {
|
|
1405
|
+
buildGate = "red";
|
|
1406
|
+
log(`BUILD GATE r${round} — RED${gate.detail ? `: ${gate.detail}` : ""}. EVAL will not run over a feature ` +
|
|
1407
|
+
`that does not build or launch; the failing step is compiled into round ${round + 1}'s orders as bugs.`);
|
|
1408
|
+
} else if (gate.exit_code === 3) {
|
|
1409
|
+
buildGate = "undeclared";
|
|
1410
|
+
log(`BUILD GATE r${round} — nothing declared: no run_cmd in the run ledger and no build_probe or ` +
|
|
1411
|
+
`launch_probe in project-profile.md. EVAL runs over an unproven build; pin them at GATE L0.`);
|
|
1412
|
+
} else if (gate.exit_code !== 0) {
|
|
1413
|
+
buildGate = "unknown";
|
|
1414
|
+
log(`BUILD GATE r${round} — the gate itself did not run (exit ${gate.exit_code}` +
|
|
1415
|
+
`${gate.detail ? `: ${gate.detail}` : ""}). Treated as undeclared, not as green.`);
|
|
1416
|
+
}
|
|
1417
|
+
|
|
1353
1418
|
await advisory(`reduce hill --slug ${slug}`, "Build", "hill-derive");
|
|
1354
1419
|
{
|
|
1355
1420
|
const g = await crossGate("L2", "Build", ["proceed", "ask", "abort"],
|
|
1356
|
-
{ round, green_scopes: roundGreen, hammer_proposals: roundHammer });
|
|
1421
|
+
{ round, green_scopes: roundGreen, hammer_proposals: roundHammer, build_gate: buildGate });
|
|
1357
1422
|
if (g.stop) return withWarnings(g.stop);
|
|
1358
1423
|
}
|
|
1359
1424
|
|
|
1360
1425
|
// ---- EVAL — exactly one feature-level pass per round (the single-judge invariant) ------------
|
|
1361
1426
|
phase("Eval");
|
|
1362
1427
|
await setRunStatus("evaluating", "Eval");
|
|
1363
|
-
if (
|
|
1428
|
+
if (buildGate === "red") {
|
|
1429
|
+
// A red gate outranks --no-eval: skipping the judge is the operator's call, but the build failing
|
|
1430
|
+
// is a measured fact, and a round that does not compile has nothing for anyone to pass.
|
|
1431
|
+
verdict = "fail";
|
|
1432
|
+
findings = [];
|
|
1433
|
+
log(`EVAL r${round} — not dispatched: the round build gate is red. The judge grades a running ` +
|
|
1434
|
+
`feature; this one does not build or launch. Round ${round + 1} fixes the gate's failing step.`);
|
|
1435
|
+
} else if (args.noEval) {
|
|
1364
1436
|
log("EVAL — skipped (--no-eval)");
|
|
1365
1437
|
verdict = "pass";
|
|
1366
1438
|
} else {
|
|
@@ -1410,7 +1482,7 @@ while (verdict !== "pass" && round <= maxRounds) {
|
|
|
1410
1482
|
|
|
1411
1483
|
await advisory(`reduce graph --slug ${slug}`, "Eval", `graph:eval-r${round}`);
|
|
1412
1484
|
await advisory(`reduce hill --slug ${slug}`, "Eval", "hill-derive");
|
|
1413
|
-
const g3 = await crossGate("L3", "Eval", ["loop", "stop", "ask"], { round, verdict });
|
|
1485
|
+
const g3 = await crossGate("L3", "Eval", ["loop", "stop", "ask"], { round, verdict, build_gate: buildGate });
|
|
1414
1486
|
if (g3.stop) return withWarnings(g3.stop);
|
|
1415
1487
|
|
|
1416
1488
|
if (verdict === "pass") break; // → QA → GATE H → ship
|