@smartmemory/compose 0.3.7 → 0.3.8

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (215) hide show
  1. package/.compose-deps.json +1 -13
  2. package/README.md +72 -5
  3. package/bin/compose.js +470 -351
  4. package/bin/judgment-migrate.js +387 -0
  5. package/contracts/comp-obs-contract.schema.json +9 -3
  6. package/contracts/fluid-record.schema.json +209 -0
  7. package/contracts/lifecycle-backfill.schema.json +322 -0
  8. package/dist/assets/App-Z4MU-H_F.js +916 -0
  9. package/dist/assets/{_baseUniq-Bo837sRJ.js → _baseUniq-ClWoCPFl.js} +1 -1
  10. package/dist/assets/{arc-BafGpyqE.js → arc-DY26UIVo.js} +1 -1
  11. package/dist/assets/{architectureDiagram-Q4EWVU46-BOBfUsqL.js → architectureDiagram-Q4EWVU46-6Ggq4DqJ.js} +1 -1
  12. package/dist/assets/{blockDiagram-DXYQGD6D-Dwodev1a.js → blockDiagram-DXYQGD6D-CH3Ked0l.js} +1 -1
  13. package/dist/assets/{browser-1ntj1-x_.js → browser-BWkrenen.js} +1 -1
  14. package/dist/assets/{c4Diagram-AHTNJAMY-CU_bhYag.js → c4Diagram-AHTNJAMY-Bk8dYilu.js} +1 -1
  15. package/dist/assets/channel-SnZzzh7k.js +1 -0
  16. package/dist/assets/{chunk-4BX2VUAB-p8WsDwnO.js → chunk-4BX2VUAB-BMR0XaAQ.js} +1 -1
  17. package/dist/assets/{chunk-4TB4RGXK-B8h7-eR0.js → chunk-4TB4RGXK-JytR14a9.js} +1 -1
  18. package/dist/assets/{chunk-55IACEB6-DxeEr98s.js → chunk-55IACEB6-B4Q97BCP.js} +1 -1
  19. package/dist/assets/{chunk-EDXVE4YY-BYt8F151.js → chunk-EDXVE4YY-R_qarkSf.js} +1 -1
  20. package/dist/assets/{chunk-FMBD7UC4-DGSOVeie.js → chunk-FMBD7UC4-C9s7KR9m.js} +1 -1
  21. package/dist/assets/{chunk-OYMX7WX6-B-QdgYR2.js → chunk-OYMX7WX6-BySQzVxc.js} +1 -1
  22. package/dist/assets/{chunk-QZHKN3VN-Du5UAZLs.js → chunk-QZHKN3VN-DdpSYZsW.js} +1 -1
  23. package/dist/assets/{chunk-YZCP3GAM-C8JbNBSk.js → chunk-YZCP3GAM-iE_tzriw.js} +1 -1
  24. package/dist/assets/classDiagram-6PBFFD2Q-CBu92dSH.js +1 -0
  25. package/dist/assets/classDiagram-v2-HSJHXN6E-CBu92dSH.js +1 -0
  26. package/dist/assets/clone-DgklGjHm.js +1 -0
  27. package/dist/assets/{cose-bilkent-S5V4N54A-O1ESaqge.js → cose-bilkent-S5V4N54A-BdlU6ZX_.js} +1 -1
  28. package/dist/assets/{dagre-KV5264BT-CPTmFPHw.js → dagre-KV5264BT-Cp3F5KTn.js} +1 -1
  29. package/dist/assets/{diagram-5BDNPKRD-B3PNrWs5.js → diagram-5BDNPKRD-DiR6_2q_.js} +1 -1
  30. package/dist/assets/{diagram-G4DWMVQ6-Cscfr6vc.js → diagram-G4DWMVQ6-w0i-p5HX.js} +1 -1
  31. package/dist/assets/{diagram-MMDJMWI5-CSfqZ-TM.js → diagram-MMDJMWI5-tIHhwUv3.js} +1 -1
  32. package/dist/assets/{diagram-TYMM5635-Cg4aYS7W.js → diagram-TYMM5635-BAeY3B19.js} +1 -1
  33. package/dist/assets/{erDiagram-SMLLAGMA-_ZqwG5pl.js → erDiagram-SMLLAGMA-Ckx_Knko.js} +1 -1
  34. package/dist/assets/{flowDiagram-DWJPFMVM-C83boxFT.js → flowDiagram-DWJPFMVM-DeoNka6J.js} +1 -1
  35. package/dist/assets/{ganttDiagram-T4ZO3ILL-CWnIjuEi.js → ganttDiagram-T4ZO3ILL-BmGnFbEg.js} +1 -1
  36. package/dist/assets/{gitGraphDiagram-UUTBAWPF-DrMdxZfH.js → gitGraphDiagram-UUTBAWPF-Dk48IHsx.js} +1 -1
  37. package/dist/assets/{graph-RE4I7Ty7.js → graph-BNzKGvoy.js} +1 -1
  38. package/dist/assets/{graph-Bi99_6Yf.js → graph-CI_1htl0.js} +1 -1
  39. package/dist/assets/{index-Rm2RE-c0.js → index-BEfrNBp8.js} +3 -3
  40. package/dist/assets/index-yyrA5OZd.css +1 -0
  41. package/dist/assets/{infoDiagram-42DDH7IO-BLmP4Epr.js → infoDiagram-42DDH7IO-BRf827i0.js} +1 -1
  42. package/dist/assets/{ishikawaDiagram-UXIWVN3A-yuWWshKN.js → ishikawaDiagram-UXIWVN3A-0kCZaeCM.js} +1 -1
  43. package/dist/assets/{journeyDiagram-VCZTEJTY-BOfhaJov.js → journeyDiagram-VCZTEJTY-rvU7ayRt.js} +1 -1
  44. package/dist/assets/{kanban-definition-6JOO6SKY-Bbolde15.js → kanban-definition-6JOO6SKY-DpQwX1C5.js} +1 -1
  45. package/dist/assets/{layout-BSf33zm8.js → layout-BI8cXFPI.js} +1 -1
  46. package/dist/assets/{linear-AvSTWMqx.js → linear-a0glcDiw.js} +1 -1
  47. package/dist/assets/{min-QBM8H4xN.js → min-vPHfnXcC.js} +1 -1
  48. package/dist/assets/{mindmap-definition-QFDTVHPH-BuvgtqIc.js → mindmap-definition-QFDTVHPH-D14eF-7C.js} +1 -1
  49. package/dist/assets/mobile-B7m9EO9D.js +17 -0
  50. package/dist/assets/{pieDiagram-DEJITSTG-DIzF16vh.js → pieDiagram-DEJITSTG-Cno-gETh.js} +1 -1
  51. package/dist/assets/{quadrantDiagram-34T5L4WZ-D-mbUIjS.js → quadrantDiagram-34T5L4WZ-BUQM1Hfm.js} +1 -1
  52. package/dist/assets/{requirementDiagram-MS252O5E-CEs4kCLd.js → requirementDiagram-MS252O5E-pOXlN2-q.js} +1 -1
  53. package/dist/assets/{sankeyDiagram-XADWPNL6-DFsnCr9n.js → sankeyDiagram-XADWPNL6-Crynd3_b.js} +1 -1
  54. package/dist/assets/{sequenceDiagram-FGHM5R23-BEJYdTjQ.js → sequenceDiagram-FGHM5R23-D9fZdCM8.js} +1 -1
  55. package/dist/assets/{stateDiagram-FHFEXIEX-BBXs57uY.js → stateDiagram-FHFEXIEX-CW9qVec8.js} +1 -1
  56. package/dist/assets/stateDiagram-v2-QKLJ7IA2-DkVLzHbY.js +1 -0
  57. package/dist/assets/{timeline-definition-GMOUNBTQ-BGvLoVAY.js → timeline-definition-GMOUNBTQ-BcHzhm_8.js} +1 -1
  58. package/dist/assets/{vennDiagram-DHZGUBPP-9LaBTMe0.js → vennDiagram-DHZGUBPP-BfytJcWk.js} +1 -1
  59. package/dist/assets/{wardley-RL74JXVD-P4MEqMTP.js → wardley-RL74JXVD-DLj-IjyB.js} +1 -1
  60. package/dist/assets/{wardleyDiagram-NUSXRM2D-o-tmxnlC.js → wardleyDiagram-NUSXRM2D-Ds0Ue68c.js} +1 -1
  61. package/dist/assets/{xychartDiagram-5P7HB3ND-Dpn7V6qk.js → xychartDiagram-5P7HB3ND-vjWDXFL6.js} +1 -1
  62. package/dist/index.html +3 -3
  63. package/lib/agent-string.js +7 -5
  64. package/lib/append-integrity.js +81 -0
  65. package/lib/backfill-evidence.js +109 -0
  66. package/lib/bug-escalation.js +9 -0
  67. package/lib/build-stream-schema.js +3 -1
  68. package/lib/build-stream-writer.js +25 -0
  69. package/lib/build.js +874 -170
  70. package/lib/canon-guard.js +28 -6
  71. package/lib/canon-override.js +196 -0
  72. package/lib/canon-registry.js +104 -0
  73. package/lib/cli-commands.js +144 -0
  74. package/lib/codex-preflight.js +26 -13
  75. package/lib/colleague/context.js +215 -0
  76. package/lib/colleague/writeback.js +95 -0
  77. package/lib/completion-gate.js +1421 -0
  78. package/lib/completion-writer.js +47 -47
  79. package/lib/consumer-fanout.js +105 -11
  80. package/lib/coverage-gate.js +200 -0
  81. package/lib/dir-lock.js +170 -0
  82. package/lib/dispatch-ledger.js +3 -3
  83. package/lib/feature-json.js +1 -1
  84. package/lib/feature-reconciler.js +8 -0
  85. package/lib/feature-validator.js +64 -1
  86. package/lib/feature-writer.js +57 -2
  87. package/lib/fluid/factory.js +167 -0
  88. package/lib/fluid/ideabox-dates.js +73 -0
  89. package/lib/fluid/ideabox-migrate.js +154 -0
  90. package/lib/fluid/ideabox-ops.js +585 -0
  91. package/lib/fluid/ideabox-view.js +146 -0
  92. package/lib/fluid/import-ideabox.js +186 -0
  93. package/lib/fluid/local-provider.js +606 -0
  94. package/lib/fluid/provider.js +684 -0
  95. package/lib/fluid/record-shape.js +214 -0
  96. package/lib/fluid/record-store.js +328 -0
  97. package/lib/fluid/render-ideabox.js +261 -0
  98. package/lib/fluid/schema.js +40 -0
  99. package/lib/fluid/smartmemory-provider.js +1695 -0
  100. package/lib/gsd.js +63 -23
  101. package/lib/guard-cli.js +175 -0
  102. package/lib/guard-custody.js +141 -0
  103. package/lib/guard-descriptors.js +530 -0
  104. package/lib/guard-enrol.js +254 -0
  105. package/lib/health-score.js +1 -1
  106. package/lib/ideabox-cli.js +315 -0
  107. package/lib/ideabox.js +121 -21
  108. package/lib/judgment/store/index.js +9 -1
  109. package/lib/judgment/store/records.js +1 -1
  110. package/lib/judgment/trace.js +380 -0
  111. package/lib/judgment-decision-write.js +277 -0
  112. package/lib/judgment-decisions.js +466 -0
  113. package/lib/judgment-gen.js +5 -1
  114. package/lib/judgment-writer.js +56 -2
  115. package/lib/lifecycle-modes.js +4 -4
  116. package/lib/lineage.js +400 -0
  117. package/lib/local-claude-connector.js +52 -1
  118. package/lib/maya-client.js +302 -0
  119. package/lib/maya-config.js +53 -0
  120. package/lib/maya-identity.js +283 -0
  121. package/lib/migrate-anon.js +5 -0
  122. package/lib/migrate-roadmap.js +15 -0
  123. package/lib/new.js +13 -1
  124. package/lib/pipeline-compat.js +104 -0
  125. package/lib/policy-catalog.js +295 -0
  126. package/lib/policy-check.js +0 -0
  127. package/lib/process-termination.js +98 -0
  128. package/lib/resolve-workspace.js +5 -1
  129. package/lib/result-normalizer.js +396 -199
  130. package/lib/roadmap-errors.js +65 -0
  131. package/lib/roadmap-preservers.js +24 -4
  132. package/lib/roadmap-residue.js +299 -0
  133. package/lib/smartmemory-client.js +614 -78
  134. package/lib/smartmemory-config.js +54 -0
  135. package/lib/smartmemory-ingest.js +19 -2
  136. package/lib/step-prompt.js +7 -6
  137. package/lib/stratum-engine.js +53 -4
  138. package/lib/stratum-mcp-client.js +271 -36
  139. package/lib/test-bootstrap.js +31 -0
  140. package/lib/tool-inventory.js +122 -0
  141. package/lib/version-check.js +91 -19
  142. package/lib/vision-writer.js +88 -1
  143. package/package.json +7 -6
  144. package/pipelines/bug-fix.stratum.yaml +205 -211
  145. package/pipelines/build-quick.profiles.json +12 -0
  146. package/pipelines/build-quick.stratum.yaml +263 -350
  147. package/pipelines/content.stratum.yaml +81 -77
  148. package/pipelines/coverage-sweep.stratum.yaml +49 -30
  149. package/pipelines/plan.stratum.yaml +76 -86
  150. package/pipelines/refactor.stratum.yaml +125 -125
  151. package/pipelines/research.stratum.yaml +56 -58
  152. package/pipelines/review-fix.profiles.json +6 -0
  153. package/pipelines/review-fix.stratum.yaml +110 -83
  154. package/presets/team-feature.profiles.json +6 -0
  155. package/presets/team-feature.stratum.yaml +93 -66
  156. package/presets/team-research.profiles.json +6 -0
  157. package/presets/team-research.stratum.yaml +89 -80
  158. package/presets/team-review.profiles.json +8 -0
  159. package/presets/team-review.stratum.yaml +98 -80
  160. package/scripts/cost-census.mjs +70 -0
  161. package/scripts/guard-sign/compose-guard-sign.sh +62 -0
  162. package/server/agent-health.js +22 -0
  163. package/server/agent-hooks.js +14 -1
  164. package/server/agent-server.js +5 -248
  165. package/server/agent-spawn.js +3 -4
  166. package/server/agent-workspace.js +294 -0
  167. package/server/build-routes.js +6 -5
  168. package/server/build-stream-bridge.js +53 -0
  169. package/server/cc-session-watcher.js +4 -1
  170. package/server/coalescing-buffer.js +7 -1
  171. package/server/completion-projection.js +228 -0
  172. package/server/compose-mcp-tools.js +109 -23
  173. package/server/compose-mcp.js +88 -882
  174. package/server/decision-event-emit.js +41 -2
  175. package/server/decision-event-id.js +17 -0
  176. package/server/decision-events-snapshot.js +3 -0
  177. package/server/design-routes.js +14 -8
  178. package/server/feature-scan.js +76 -2
  179. package/server/file-watcher.js +170 -21
  180. package/server/ideabox-routes.js +166 -224
  181. package/server/index.js +70 -100
  182. package/server/lifecycle-guard.js +240 -10
  183. package/server/lifecycle-phase-history.js +276 -0
  184. package/server/maya-routes.js +507 -0
  185. package/server/mcp-tool-defs.js +940 -0
  186. package/server/mcp-tool-policy.js +34 -2
  187. package/server/model-tiers.js +22 -5
  188. package/server/pipeline-routes.js +21 -11
  189. package/server/project-root.js +58 -19
  190. package/server/remote-utils.js +3 -1
  191. package/server/schema-validator.js +7 -1
  192. package/server/session-manager.js +5 -6
  193. package/server/session-routes.js +3 -1
  194. package/server/stratum-client.js +57 -10
  195. package/server/stratum-sync.js +6 -3
  196. package/server/summarizer.js +3 -4
  197. package/server/supervisor.js +0 -1
  198. package/server/vision-routes.js +208 -98
  199. package/server/vision-server.js +86 -23
  200. package/server/vision-store.js +60 -6
  201. package/server/vision-utils.js +3 -4
  202. package/server/workspace-activity.js +18 -0
  203. package/server/workspace-middleware.js +2 -2
  204. package/server/workspace-runtime.js +243 -0
  205. package/server/worktree-gc.js +1 -0
  206. package/dist/assets/App-PkZzHeMj.js +0 -894
  207. package/dist/assets/channel-qVK_qn4E.js +0 -1
  208. package/dist/assets/classDiagram-6PBFFD2Q-B8UcfC1q.js +0 -1
  209. package/dist/assets/classDiagram-v2-HSJHXN6E-B8UcfC1q.js +0 -1
  210. package/dist/assets/clone-Pu3RyLUh.js +0 -1
  211. package/dist/assets/index-LIwREYgH.css +0 -1
  212. package/dist/assets/mobile-BnXEOE3U.js +0 -17
  213. package/dist/assets/stateDiagram-v2-QKLJ7IA2-BqKuX4rj.js +0 -1
  214. package/lib/staleness.js +0 -87
  215. package/server/ideabox-cache.js +0 -77
package/lib/build.js CHANGED
@@ -16,8 +16,12 @@ import { createHash, randomUUID } from 'node:crypto';
16
16
 
17
17
  import { StratumMcpClient, StratumError, resolvePlanSpecValues, resolveStepProfile } from './stratum-mcp-client.js';
18
18
  import { resolveStratumMcpConnection } from './stratum-engine.js';
19
- import { runAndNormalize, AgentTimeoutError, AgentAbortedError, UserInterruptError, AgentError } from './result-normalizer.js';
19
+ import { runAndNormalize, mergeUsage, AgentTimeoutError, AgentAbortedError, UserInterruptError, AgentError } from './result-normalizer.js';
20
20
  import { checkCapabilityViolation } from './capability-checker.js';
21
+ import { getCatalog as getPolicyCatalog, getPolicyCheckConfig } from './policy-catalog.js';
22
+ import {
23
+ resolveBuildUserMode, scanResponse, toViolationStrings, buildRevisionNotice, attachPolicyCount,
24
+ } from './policy-check.js';
21
25
  import { preflightCodexWorktreeProbe, codexProbeAbortMessage } from './codex-preflight.js';
22
26
  import { buildStepPrompt, buildGateContext, clearAmbientContextCache } from './step-prompt.js';
23
27
  import { promptGate } from './gate-prompt.js';
@@ -33,11 +37,12 @@ import { resolveAgentConfig, parseAgentString } from './agent-string.js';
33
37
  import { emitSections as emitPlanSections, appendTrailers as appendSectionTrailers, analyzeRollup, writeRollup } from './sections.js';
34
38
  import { SECTIONS_DIR } from './constants.js';
35
39
  import { rtkPrefix } from './rtk.js';
40
+ import { tsCompatibilityOf, quarantineMessage, INIT_PROVISIONED_SPECS } from './pipeline-compat.js';
36
41
 
37
42
  import YAML from 'yaml';
38
43
  // feature-json direct imports removed — mutations now go through TrackerProvider (T9)
39
44
  import { loadFeaturesDir, resolveContextPath, resolveRoadmapPath, resolveFeaturesPath } from './project-paths.js';
40
- import { getMode } from './lifecycle-modes.js';
45
+ import { getMode, resolveMode } from './lifecycle-modes.js';
41
46
  import { vocabularyEnabled, tagVocabularyViolations, VOCABULARY_FILE } from './vocabulary-inject.js';
42
47
  import { vocabularyCompliance } from './vocabulary-compliance.js';
43
48
 
@@ -54,7 +59,7 @@ import { applyFrontTriage, maybeEscalateLane } from './lane-gate.js';
54
59
  import { LENS_DEFINITIONS } from './review-lenses.js';
55
60
  import { injectCertInstructions } from './cert-inject.js';
56
61
  import { buildReviewPrompt } from './review-prompt.js';
57
- import { detectTestFramework, scaffoldTestFramework, parseTestSummary, deriveTestsPass, isTestFile } from './test-bootstrap.js';
62
+ import { detectTestFramework, scaffoldTestFramework, parseTestSummary, deriveTestsPass, deriveTestsAttested, isTestFile } from './test-bootstrap.js';
58
63
  import { classifyStepAsTier, evaluateTiers } from './gate-tiers.js';
59
64
  import { mapFilesToRoutes, classifyRoutes, isDocsOnlyDiff } from './qa-scoping.js';
60
65
  import { computeCompositeScore } from './health-score.js';
@@ -76,6 +81,7 @@ import {
76
81
  verifyConsumerRunRevision,
77
82
  } from './consumer-fanout.js';
78
83
  import { appendEvent as appendDispatchEvent, readEvents as readDispatchEvents } from './dispatch-ledger.js';
84
+ import { appendEvent as appendFeatureEvent } from './feature-events.js';
79
85
 
80
86
  // ---------------------------------------------------------------------------
81
87
  // COMP-ROADMAP-PLAN S8: gate the `ship` interception by mode.
@@ -97,6 +103,85 @@ export function shouldInterceptShip(stepId, mode) {
97
103
  return stepId === 'ship' && mode !== 'plan';
98
104
  }
99
105
 
106
+ // ---------------------------------------------------------------------------
107
+ // COMP-POLICY-CHECK: pre-response policy check (adherence enforcement).
108
+ // ---------------------------------------------------------------------------
109
+
110
+ /**
111
+ * Does this step declare a gate? A gate step is SKILL_GATED by construction —
112
+ * asking the user for a decision is the point of the step, so policy matches on
113
+ * its response are suppressed rather than flagged.
114
+ *
115
+ * @param {object} spec the local pipeline spec
116
+ * @param {string} flowName active flow
117
+ * @param {string} stepId ready-step id (scoped ids resolve to their bare tail)
118
+ * @returns {boolean}
119
+ */
120
+ export function isGateStep(spec, flowName, stepId) {
121
+ const bare = String(stepId ?? '').split('/').pop();
122
+ const steps = spec?.flows?.[flowName]?.steps;
123
+ if (!Array.isArray(steps)) return false;
124
+ return steps.some(st => st?.id === bare && !!st.gate);
125
+ }
126
+
127
+ /**
128
+ * COMP-POLICY-CHECK-2/3: scan one step response against the local catalog.
129
+ * Total — a broken catalog or scan degrades to "no findings" with a WARNING and
130
+ * never fails the step.
131
+ *
132
+ * The user mode is EXPLICIT here, never inferred: a build has no user turns to
133
+ * classify (see `resolveBuildUserMode`). Config override → gate step → default.
134
+ *
135
+ * @param {{cwd: string, text: string, skillGated: boolean}} args
136
+ * @returns {{records: object[], violations: string[], userMode: string}}
137
+ */
138
+ export function policyScanForStep({ cwd, text, skillGated }) {
139
+ const empty = { records: [], violations: [], userMode: 'AUTONOMOUS' };
140
+ try {
141
+ const config = getPolicyCheckConfig(cwd);
142
+ const catalog = getPolicyCatalog(cwd);
143
+ if (catalog.length === 0) return empty;
144
+ const userMode = resolveBuildUserMode(config.userMode, { skillGated });
145
+ const records = scanResponse(text ?? '', catalog, userMode);
146
+ return { records, violations: toViolationStrings(records), userMode };
147
+ } catch (err) {
148
+ // eslint-disable-next-line no-console
149
+ console.warn(`[policy-check] scan skipped: ${err.message}`);
150
+ return empty;
151
+ }
152
+ }
153
+
154
+ /**
155
+ * COMP-POLICY-CHECK-5: trace every match (flagged AND suppressed) to the
156
+ * append-only feature-events bus — which syncs into SmartMemory, closing the
157
+ * measurement loop — plus the build stream for live cockpit visibility.
158
+ *
159
+ * @param {object} args
160
+ * @param {string} args.pass 'initial' | 'policy_revision'
161
+ */
162
+ export function recordPolicyScan({ cwd, streamWriter, stepId, records, userMode, featureCode, buildId, pass = 'initial' }) {
163
+ for (const record of records ?? []) {
164
+ try {
165
+ appendFeatureEvent(cwd, {
166
+ tool: 'policy_check',
167
+ build_id: buildId ?? null,
168
+ step_id: stepId,
169
+ rule: record.rule,
170
+ matched: record.matched,
171
+ suppressed: record.suppressed,
172
+ user_mode: userMode,
173
+ pass,
174
+ });
175
+ } catch (err) {
176
+ // eslint-disable-next-line no-console
177
+ console.warn(`[policy-check] trace append failed: ${err.message}`);
178
+ }
179
+ try {
180
+ streamWriter?.writePolicyViolation(stepId, record, userMode, featureCode ?? null, buildId ?? null);
181
+ } catch { /* stream emit is best-effort */ }
182
+ }
183
+ }
184
+
100
185
  // ---------------------------------------------------------------------------
101
186
  // COMP-ROADMAP-PLAN S5: ratify a plan-authored design instead of clobbering it.
102
187
  // ---------------------------------------------------------------------------
@@ -445,6 +530,39 @@ export function deriveOrdinaryReviewScaffold({ contractName = null, stepId = '',
445
530
  return { isReviewMain, isReduceMain, isReviewScaffoldMain: isReviewMain && !isReduceMain };
446
531
  }
447
532
 
533
+ // COMP-AGENT-LANES: one lane per parallel worker slot. Identity is
534
+ // flowId:stepId:itemIndex (stepId/itemIndex RECUR across builds, so flowId is
535
+ // load-bearing); version is the ordered tuple (generation, attempt) — the UI
536
+ // resets a lane on a higher version and rejects lower (stale) events. The
537
+ // label is the human mandate: the review lens id when the item is a review,
538
+ // else the step intent truncated.
539
+ const LANE_LABEL_MAX = 80;
540
+
541
+ export function buildLaneEnvelope(descriptor, flowId, { lens = null } = {}) {
542
+ const rawLabel = (typeof lens === 'string' && lens)
543
+ || (typeof descriptor?.do === 'string' && descriptor.do)
544
+ || String(descriptor?.id ?? '');
545
+ const label = rawLabel.length > LANE_LABEL_MAX
546
+ ? `${rawLabel.slice(0, LANE_LABEL_MAX - 1)}…`
547
+ : rawLabel;
548
+ return {
549
+ flowId,
550
+ stepId: descriptor.id,
551
+ itemIndex: descriptor.itemIndex,
552
+ generation: descriptor.generation ?? 0,
553
+ attempt: descriptor.attempt ?? 1,
554
+ label,
555
+ agent: descriptor.agent ?? 'claude',
556
+ };
557
+ }
558
+
559
+ function deriveConsumerLane(descriptor, flowId) {
560
+ const reviewOpts = deriveConsumerReviewOptions(descriptor);
561
+ return buildLaneEnvelope(descriptor, flowId, {
562
+ lens: reviewOpts.reviewMode ? reviewOpts.lens : null,
563
+ });
564
+ }
565
+
448
566
  export function deriveConsumerReviewOptions(descriptor) {
449
567
  const reviewMode = descriptor?.contract?.root === 'ReviewResult';
450
568
  const item = (descriptor?.item && typeof descriptor.item === 'object') ? descriptor.item : {};
@@ -620,6 +738,7 @@ async function reportConsumerStepDone({
620
738
  itemIndex: descriptor.itemIndex,
621
739
  stage: descriptor.stage,
622
740
  generation: descriptor.generation,
741
+ lane: deriveConsumerLane(descriptor, flowId),
623
742
  });
624
743
  return { response, skipped: true };
625
744
  }
@@ -786,6 +905,12 @@ export async function runConsumerIssuance({
786
905
  // contract the python parallel-dispatch path emitted — otherwise the fanout runs
787
906
  // invisibly and the parallel progress bar never appears.
788
907
  const parallelStepNum = `∥${descriptor.itemIndex}`;
908
+ // COMP-AGENT-LANES: the same lane envelope rides every lifecycle write for
909
+ // this item AND (via runAndNormalize opts) every relayed output write, so the
910
+ // cockpit can attribute each event to its worker slot.
911
+ const lane = buildLaneEnvelope(descriptor, flowId, {
912
+ lens: reviewOpts.reviewMode ? reviewOpts.lens : null,
913
+ });
789
914
  progress.stepStart(parallelStepNum, '?', descriptor.id);
790
915
  streamWriter.write({
791
916
  type: 'build_step_start',
@@ -800,6 +925,7 @@ export async function runConsumerIssuance({
800
925
  itemIndex: descriptor.itemIndex,
801
926
  stage: descriptor.stage,
802
927
  generation: descriptor.generation,
928
+ lane,
803
929
  });
804
930
 
805
931
  let mainResult;
@@ -809,7 +935,9 @@ export async function runConsumerIssuance({
809
935
  streamWriter,
810
936
  maxDurationMs,
811
937
  stratum,
938
+ lane,
812
939
  cwd: recovery.worktree,
940
+ sandboxMode: descriptor.policy?.isolation === 'worktree' ? 'workspace-write' : 'read-only',
813
941
  onAgentEvent,
814
942
  profile,
815
943
  reviewMode: reviewOpts.reviewMode,
@@ -823,21 +951,24 @@ export async function runConsumerIssuance({
823
951
  step_id: descriptor.id,
824
952
  ...(typeof descriptor.attempt === 'number' ? { attempt: descriptor.attempt } : {}),
825
953
  },
826
- // V2/V3: the isolation:none review fanout is the safety-critical controlled
827
- // execution a claude review item runs via the compose-local connector so
828
- // its read-only tool restrictions actually BIND (the engine's sync
829
- // agent_run can't carry claude allowlists and its sandboxMode binds only
830
- // codex) and its per-item timeout / stuck abort truly INTERRUPTS it. Write
831
- // (worktree) items keep the engine seam: their per-item timeout still fails
832
- // the item, but interrupting an in-flight workspace-write run needs a
833
- // stratum follow-up (background mode is codex+read-only-only). Codex review
834
- // items also fall back to the sync seam (compose has no codex SDK).
954
+ // Local Claude owns its SDK process group for review fanout and drains
955
+ // graceful teardown before timeout/interrupt returns, as MCP does.
835
956
  localExecution: descriptor.policy?.isolation === 'none',
836
957
  });
837
958
  } catch (error) {
838
- // User interrupts and injected crashes are control-flow signals, not item
839
- // failures they must propagate and abort the pump.
840
- if (error instanceof UserInterruptError || error?.code === 'INJECTED_CONSUMER_CRASH') {
959
+ const failedUsage = failureUsageFields(error);
960
+ // Control failures do not settle the item, so record known dispatch usage
961
+ // before aborting the pump. Never retry work with uncertain termination.
962
+ if (error instanceof UserInterruptError || ['INJECTED_CONSUMER_CRASH', 'CANCELLATION_UNCONFIRMED', 'CANCELLATION_TEARDOWN_TIMEOUT'].includes(error?.code)) {
963
+ if (failedUsage.usage && typeof context?.onUsage === 'function') {
964
+ try {
965
+ await context.onUsage(usagePayload(failedUsage.usage, failedUsage.usages), {
966
+ dispatchId: error.dispatchId, stepId: descriptor.step ?? descriptor.id, source: 'consumer',
967
+ });
968
+ } catch (usageError) {
969
+ console.warn(`[consumer] Could not record cancelled usage: ${usageError?.message ?? usageError}`);
970
+ }
971
+ }
841
972
  throw error;
842
973
  }
843
974
  // D3: a stuck verdict halts the whole GSD run (not a per-item retry) — the
@@ -846,8 +977,12 @@ export async function runConsumerIssuance({
846
977
  // G3: no step_done envelope is sent on the stuck/abort path (the run halts),
847
978
  // so the billable usage the aborted run consumed would be lost. Record it
848
979
  // into compose's cumulative ledger before converting to the stuck signal.
849
- if (error.usage && typeof context?.onUsage === 'function') {
850
- context.onUsage(error.usage, descriptor);
980
+ if (failedUsage.usage && typeof context?.onUsage === 'function') {
981
+ await context.onUsage(usagePayload(failedUsage.usage, failedUsage.usages), {
982
+ dispatchId: error.dispatchId,
983
+ stepId: descriptor.step ?? descriptor.id,
984
+ source: 'consumer',
985
+ });
851
986
  }
852
987
  throw new ConsumerStuckError(stuckTaskId, error.reason);
853
988
  }
@@ -859,7 +994,7 @@ export async function runConsumerIssuance({
859
994
  settlementFailureClass: 'agent',
860
995
  // G3: a timed-out run still consumed billable usage — forward it so the
861
996
  // failure envelope debits the engine ledger (same mechanism as F3).
862
- ...(error && typeof error === 'object' && error.usage ? { usage: error.usage } : {}),
997
+ ...failedUsage,
863
998
  };
864
999
  } else {
865
1000
  // A non-timeout agent/connector error must fail ONLY this item, not abort
@@ -877,7 +1012,7 @@ export async function runConsumerIssuance({
877
1012
  // it to the error. Forward it so the failure envelope (and compose's
878
1013
  // cumulative ledger) debit the attempt instead of letting failures evade
879
1014
  // budget exhaustion.
880
- ...(error && typeof error === 'object' && error.usage ? { usage: error.usage } : {}),
1015
+ ...failedUsage,
881
1016
  };
882
1017
  }
883
1018
  }
@@ -886,7 +1021,11 @@ export async function runConsumerIssuance({
886
1021
  // D2(b): forward the item's agent usage so GSD can debit the cumulative
887
1022
  // budget ledger. Build mode passes no onUsage sink → byte-identical no-op.
888
1023
  if (typeof context?.onUsage === 'function' && mainResult?.usage) {
889
- context.onUsage(mainResult.usage, descriptor);
1024
+ await context.onUsage(usagePayload(mainResult.usage, mainResult.usages), {
1025
+ dispatchId: mainResult.dispatchIds?.primary,
1026
+ stepId: descriptor.step ?? descriptor.id,
1027
+ source: 'fanout',
1028
+ });
890
1029
  }
891
1030
  const finalStage = isFinalConsumerStage(localSpec, descriptor);
892
1031
  let localFailure = normalizationFailure
@@ -917,7 +1056,7 @@ export async function runConsumerIssuance({
917
1056
  // succeeded, so usage rides both the success and failure envelope. Compose's
918
1057
  // cumulative ledger (context.onUsage) is separate, compose-side accounting.
919
1058
  const engineUsage = toEngineUsage(mainResult?.usage);
920
- if (engineUsage) envelope.usage = engineUsage;
1059
+ if (engineUsage && !context?.receiptsMode) envelope.usage = engineUsage;
921
1060
 
922
1061
  if (typeof artifacts.hooks.afterAgentMutationBeforePrepared === 'function') {
923
1062
  await artifacts.hooks.afterAgentMutationBeforePrepared({
@@ -981,6 +1120,14 @@ export async function runConsumerIssuance({
981
1120
  // H6: matches the item's start stepId so the UI decrements the same task
982
1121
  // (AgentStream keys the per-task done on parallel:true + a known stepId).
983
1122
  parallel: true,
1123
+ // COMP-AGENT-LANES (C4): terminal status is explicit at source — the UI
1124
+ // must not infer "complete" from the done event's existence.
1125
+ status: localFailure ? 'failed' : 'succeeded',
1126
+ outcome: result?.outcome ?? (localFailure ? 'failed' : 'succeeded'),
1127
+ itemIndex: descriptor.itemIndex,
1128
+ stage: descriptor.stage,
1129
+ generation: descriptor.generation,
1130
+ lane,
984
1131
  });
985
1132
  return response;
986
1133
  }
@@ -1066,6 +1213,91 @@ export function toEngineUsage(usage) {
1066
1213
  return Object.keys(out).length > 0 ? out : null;
1067
1214
  }
1068
1215
 
1216
+ // A repair failure can carry two dispatch records. Keep those records for
1217
+ // receipts and aggregate both for the legacy step_done budget envelope.
1218
+ function failureUsageFields(error) {
1219
+ const usages = Array.isArray(error?.usages) ? error.usages : null;
1220
+ if (!usages?.length) return error?.usage ? { usage: error.usage } : {};
1221
+ if (usages.length === 1 && error?.usage) return { usage: error.usage, usages };
1222
+ const usage = {
1223
+ input_tokens: 0, output_tokens: 0, cache_creation_input_tokens: 0,
1224
+ cache_read_input_tokens: 0, cost_usd: 0, duration_ms: 0, model: null,
1225
+ };
1226
+ for (const entry of usages) {
1227
+ if (!entry || typeof entry !== 'object') continue;
1228
+ usage.input_tokens += entry.input_tokens ?? 0;
1229
+ usage.output_tokens += entry.output_tokens ?? 0;
1230
+ usage.cache_creation_input_tokens += entry.cache_creation ?? entry.cache_creation_input_tokens ?? 0;
1231
+ usage.cache_read_input_tokens += entry.cache_read ?? entry.cache_read_input_tokens ?? 0;
1232
+ usage.cost_usd += entry.cost_usd ?? 0;
1233
+ usage.duration_ms += entry.duration_ms ?? 0;
1234
+ usage.model = entry.model ?? usage.model;
1235
+ }
1236
+ return { usage, usages };
1237
+ }
1238
+
1239
+ function usagePayload(usage, usages) {
1240
+ if (!usage || typeof usage !== 'object') return usage;
1241
+ return Array.isArray(usages) ? { ...usage, usages } : usage;
1242
+ }
1243
+
1244
+ /** Send one surface-15 receipt per underlying model dispatch. */
1245
+ export async function reportUsageReceipts(context, usage, meta = {}) {
1246
+ if (!context?.receiptsMode || !context.flowId || typeof context.stratum?.usageReport !== 'function') {
1247
+ return [];
1248
+ }
1249
+ const entries = Array.isArray(usage)
1250
+ ? usage
1251
+ : (Array.isArray(usage?.usages) ? usage.usages : (usage ? [usage] : []));
1252
+ const responses = [];
1253
+ for (const entry of entries) {
1254
+ if (!entry || typeof entry !== 'object') continue;
1255
+ const engineUsage = toEngineUsage(entry);
1256
+ if (!engineUsage) continue;
1257
+ // Surface 15 requires explicit USD provenance. Normalized UsageRecords carry
1258
+ // `usd_source`; raw engine usage ({tokens, usd, ms}) does not. Preserve raw
1259
+ // token/time usage, but fail closed on an unlabelled dollar value instead of
1260
+ // manufacturing "reported" provenance.
1261
+ const usdSource = ['reported', 'estimated'].includes(entry.usd_source)
1262
+ ? entry.usd_source
1263
+ : null;
1264
+ if (Object.hasOwn(engineUsage, 'usd') && !usdSource) delete engineUsage.usd;
1265
+ if (Object.keys(engineUsage).length === 0) continue;
1266
+ const input = entry.input_tokens;
1267
+ const output = entry.output_tokens;
1268
+ const receipt = {
1269
+ dispatchId: entry.dispatch_id ?? meta.dispatchId ?? randomUUID(),
1270
+ ...(meta.stepId ? { stepId: meta.stepId } : {}),
1271
+ source: meta.source ?? 'main',
1272
+ usage: engineUsage,
1273
+ telemetry: {
1274
+ model: typeof entry.model === 'string' && entry.model.length > 0 ? entry.model : 'unknown',
1275
+ ...(typeof entry.effort === 'string' && entry.effort.length > 0 ? { effort: entry.effort } : {}),
1276
+ durationMs: entry.duration_ms ?? entry.ms ?? 0,
1277
+ },
1278
+ ...(typeof input === 'number' || typeof output === 'number'
1279
+ ? { split: {
1280
+ input: input ?? 0,
1281
+ output: output ?? 0,
1282
+ ...(typeof (entry.cache_read ?? entry.cache_read_input_tokens) === 'number'
1283
+ ? { cacheRead: entry.cache_read ?? entry.cache_read_input_tokens }
1284
+ : {}),
1285
+ ...(typeof (entry.cache_creation ?? entry.cache_creation_input_tokens) === 'number'
1286
+ ? { cacheCreation: entry.cache_creation ?? entry.cache_creation_input_tokens }
1287
+ : {}),
1288
+ } }
1289
+ : {}),
1290
+ ...(Object.hasOwn(engineUsage, 'usd') ? { usdSource } : {}),
1291
+ };
1292
+ try {
1293
+ responses.push(await context.stratum.usageReport(context.flowId, receipt));
1294
+ } catch (error) {
1295
+ console.warn(`[usage-receipt] failed for ${receipt.dispatchId}: ${error?.message ?? error}`);
1296
+ }
1297
+ }
1298
+ return responses;
1299
+ }
1300
+
1069
1301
  /**
1070
1302
  * F5: deterministic v1 vocabulary enforcement, evaluated compose-side at the
1071
1303
  * review_merge step (the step the now-dropped judged ensure was attached to). The
@@ -1244,7 +1476,20 @@ function writeActiveBuild(dataDir, state) {
1244
1476
  renameSync(tmp, target);
1245
1477
  }
1246
1478
 
1247
- const BUILD_ACCUMULATOR_VERSION = 1;
1479
+ // COMP-COMPLETION-GATE slice 2: v2 adds the completion evidence the gate needs
1480
+ // at terminalization — `tests_attested` (tri-state) and `evidence_root`.
1481
+ //
1482
+ // `test_count`/`pass_rate` were already here but are metrics, not attestation:
1483
+ // they are only populated when the output PARSED, so their absence is ambiguous
1484
+ // between "no tests" and "could not read the output". The gate cannot act on an
1485
+ // ambiguous signal, hence an explicit tri-state.
1486
+ //
1487
+ // `evidence_root` is persisted because a cross-repo build runs git and tests in
1488
+ // the agent's tree while feature metadata lives in the project tree — and
1489
+ // `runBuild` reconstructs that root from the CURRENT invocation, so a resumed
1490
+ // cross-repo build would otherwise fall back to the project root and verify the
1491
+ // wrong repository's HEAD.
1492
+ const BUILD_ACCUMULATOR_VERSION = 2;
1248
1493
  const BUILD_ACCUMULATOR_FIELDS = new Set([
1249
1494
  'v',
1250
1495
  'build_id',
@@ -1256,9 +1501,13 @@ const BUILD_ACCUMULATOR_FIELDS = new Set([
1256
1501
  'ship_files_changed',
1257
1502
  'test_count',
1258
1503
  'pass_rate',
1504
+ 'tests_attested',
1505
+ 'evidence_root',
1259
1506
  'tokens_total',
1260
1507
  'usd',
1261
1508
  ]);
1509
+ /** The only values `tests_attested` may hold. See deriveTestsAttested. */
1510
+ const TESTS_ATTESTED_VALUES = new Set(['passed', 'failed', 'no-signal']);
1262
1511
  const UUID_RE = /^[0-9a-f]{8}-[0-9a-f]{4}-[1-8][0-9a-f]{3}-[89ab][0-9a-f]{3}-[0-9a-f]{12}$/i;
1263
1512
 
1264
1513
  function assertFeatureCodeForAccumulator(featureCode) {
@@ -1326,9 +1575,35 @@ function validateBuildAccumulator(value, expectedFeatureCode = null) {
1326
1575
  throw new Error(`Build accumulator is corrupt: ${key} must be a non-negative finite number`);
1327
1576
  }
1328
1577
  }
1578
+ if (!TESTS_ATTESTED_VALUES.has(value.tests_attested)) {
1579
+ throw new Error(
1580
+ `Build accumulator is corrupt: tests_attested must be one of ${[...TESTS_ATTESTED_VALUES].join('|')}`,
1581
+ );
1582
+ }
1583
+ if (value.evidence_root !== null && typeof value.evidence_root !== 'string') {
1584
+ throw new Error('Build accumulator is corrupt: evidence_root must be null or a string');
1585
+ }
1329
1586
  return value;
1330
1587
  }
1331
1588
 
1589
+ /**
1590
+ * Bring a v1 accumulator forward. A v1 record predates completion evidence, so
1591
+ * it cannot say anything about whether tests were attested — and the honest value
1592
+ * for "we do not know" is `no-signal`, which the completion gate REFUSES. A build
1593
+ * resumed across this upgrade therefore has to re-attest rather than inheriting a
1594
+ * pass it never recorded. That is the intended direction: absence of signal is
1595
+ * never attestation.
1596
+ */
1597
+ function migrateBuildAccumulator(value) {
1598
+ if (!value || typeof value !== 'object' || value.v !== 1) return value;
1599
+ return {
1600
+ ...value,
1601
+ v: BUILD_ACCUMULATOR_VERSION,
1602
+ tests_attested: 'no-signal',
1603
+ evidence_root: null,
1604
+ };
1605
+ }
1606
+
1332
1607
  export function readBuildAccumulator(projectCwd, featureCode) {
1333
1608
  const path = buildAccumulatorPath(projectCwd, featureCode);
1334
1609
  if (!existsSync(path)) return null;
@@ -1338,7 +1613,7 @@ export function readBuildAccumulator(projectCwd, featureCode) {
1338
1613
  } catch (error) {
1339
1614
  throw new Error(`Build accumulator is corrupt at ${path}: ${error.message}`);
1340
1615
  }
1341
- return validateBuildAccumulator(parsed, featureCode);
1616
+ return validateBuildAccumulator(migrateBuildAccumulator(parsed), featureCode);
1342
1617
  }
1343
1618
 
1344
1619
  export function writeBuildAccumulator(projectCwd, accumulator) {
@@ -1368,6 +1643,8 @@ export function newBuildAccumulatorRecord(featureCode) {
1368
1643
  ship_files_changed: null,
1369
1644
  test_count: null,
1370
1645
  pass_rate: null,
1646
+ tests_attested: 'no-signal',
1647
+ evidence_root: null,
1371
1648
  tokens_total: 0,
1372
1649
  usd: 0,
1373
1650
  };
@@ -1473,7 +1750,11 @@ export function settleDispatches(projectCwd, buildId, stepId, {
1473
1750
  if (gsd) return [];
1474
1751
  const primary = dispatchIds?.primary;
1475
1752
  const repair = dispatchIds?.repair;
1476
- if (!primary && !repair) return [];
1753
+ // COMP-POLICY-CHECK-4: a policy revision is a second dispatch whose output
1754
+ // replaced the primary's. It settles on the same verdict as the run it
1755
+ // replaced — same path, one more id, no parallel settlement loop.
1756
+ const revision = dispatchIds?.revision;
1757
+ if (!primary && !repair && !revision) return [];
1477
1758
  const rows = [];
1478
1759
  const appendSettlement = (dispatchId, isAccepted, rejectedClass) => {
1479
1760
  if (typeof dispatchId !== 'string' || dispatchId.length === 0) return;
@@ -1501,6 +1782,14 @@ export function settleDispatches(projectCwd, buildId, stepId, {
1501
1782
  isEnsureRetry ? 'ensure-retry' : failureClass,
1502
1783
  );
1503
1784
  }
1785
+
1786
+ if (revision) {
1787
+ appendSettlement(
1788
+ revision,
1789
+ isEnsureRetry ? false : accepted === true,
1790
+ isEnsureRetry ? 'ensure-retry' : failureClass,
1791
+ );
1792
+ }
1504
1793
  return rows;
1505
1794
  }
1506
1795
 
@@ -1668,10 +1957,12 @@ function isProcessAlive(pid) {
1668
1957
  * @param {object} gateDispatch - Stratum gate dispatch (step_id, on_approve, on_revise, on_kill)
1669
1958
  * @param {object} [gateExtras] - Optional enrichment (fromPhase, toPhase, summary)
1670
1959
  */
1671
- function makeAskAgent(stratum, context, gateDispatch, gateExtras) {
1960
+ export function makeAskAgent(stratum, context, gateDispatch, gateExtras) {
1672
1961
  const preamble = buildGateContext(gateDispatch, context, gateExtras);
1962
+ let budgetExhausted = false;
1673
1963
 
1674
1964
  return async function askAgent(question, artifactPath) {
1965
+ if (budgetExhausted) return '(budget exhausted)';
1675
1966
  const fileRef = artifactPath && !artifactPath.endsWith('/')
1676
1967
  ? `Read the file "${artifactPath}" and answer`
1677
1968
  : `Look at the project files in the working directory and answer`;
@@ -1690,6 +1981,15 @@ function makeAskAgent(stratum, context, gateDispatch, gateExtras) {
1690
1981
  step_id: gateDispatch.step_id ?? gateDispatch.id,
1691
1982
  ...(typeof gateDispatch.attempt === 'number' ? { attempt: gateDispatch.attempt } : {}),
1692
1983
  },
1984
+ onUsage: async (usages) => {
1985
+ const results = await context.recordBuildUsage?.(usages, {
1986
+ stepId: gateDispatch.step_id ?? gateDispatch.id,
1987
+ source: 'gate_qa',
1988
+ });
1989
+ if (results?.some((result) => ['flow_exhausted', 'flow_exhausted_after_terminal'].includes(result?.budget))) {
1990
+ budgetExhausted = true;
1991
+ }
1992
+ },
1693
1993
  });
1694
1994
  return text || '(no answer)';
1695
1995
  };
@@ -1740,6 +2040,22 @@ export function resolveTemplatePath(name, cwd) {
1740
2040
  const presetsPath = join(packageDir, '..', 'presets', `${templateName}.stratum.yaml`);
1741
2041
  if (existsSync(presetsPath)) return presetsPath;
1742
2042
 
2043
+ // COMP-PIPELINE-QUARANTINE follow-up: fall back to the BUNDLED pipelines too,
2044
+ // not just presets. `compose init` seeds a curated few specs, so every other
2045
+ // shipped pipeline (content, coverage-sweep, refactor, research, review-fix)
2046
+ // was unreachable from a workspace no matter how it was invoked — the resolver
2047
+ // simply had no path to them. Project-local still wins, so a workspace that
2048
+ // customizes a spec keeps its own copy.
2049
+ //
2050
+ // NOT for the init-provisioned specs: if `build` is missing, the workspace was
2051
+ // never initialized, and answering with our bundled copy would silently run
2052
+ // Compose's own pipeline against an uninitialized project instead of raising
2053
+ // "Lifecycle spec not found".
2054
+ if (!INIT_PROVISIONED_SPECS.includes(templateName)) {
2055
+ const bundledPath = join(packageDir, '..', 'pipelines', `${templateName}.stratum.yaml`);
2056
+ if (existsSync(bundledPath)) return bundledPath;
2057
+ }
2058
+
1743
2059
  return projectPath;
1744
2060
  }
1745
2061
 
@@ -2160,6 +2476,15 @@ export async function runBuild(featureCode, opts = {}) {
2160
2476
  const stepProfiles = loadPipelineProfiles(specPath);
2161
2477
  let specYaml = readFileSync(specPath, 'utf-8');
2162
2478
 
2479
+ // COMP-PIPELINE-QUARANTINE: refuse a retired-dialect spec HERE, at the one
2480
+ // seam every template passes through (build, fix, plan, --quick, --template,
2481
+ // bundled presets), rather than letting the engine answer with a bare
2482
+ // `-32602: spec validation failed` that names neither the file nor the cause.
2483
+ const specCompat = tsCompatibilityOf(specYaml);
2484
+ if (!specCompat.compatible) {
2485
+ throw new Error(quarantineMessage(specPath, specCompat));
2486
+ }
2487
+
2163
2488
  // STRAT-IMMUTABLE: hash the on-disk spec BEFORE triage mutation for tamper detection.
2164
2489
  // verifyPipelineIntegrity() re-reads from disk, so we must compare against the original file content.
2165
2490
  const specFileHash = _sha256(specYaml);
@@ -2266,6 +2591,9 @@ export async function runBuild(featureCode, opts = {}) {
2266
2591
  // Stratum MCP client (test override permitted via opts.stratum)
2267
2592
  stratum = opts.stratum ?? new StratumMcpClient();
2268
2593
  if (!opts.stratum) await stratum.connect(resolveStratumMcpConnection(cwd));
2594
+ const receiptsMode = typeof stratum.hasTool === 'function'
2595
+ ? await stratum.hasTool('stratum_usage_report')
2596
+ : false;
2269
2597
 
2270
2598
  // Update feature.json status to IN_PROGRESS (only modes that track
2271
2599
  // feature.json lifecycle status; bug AND plan do not).
@@ -2396,6 +2724,26 @@ export async function runBuild(featureCode, opts = {}) {
2396
2724
  }
2397
2725
  reviewerAgent = opts.reviewer;
2398
2726
  }
2727
+ // COMP-PIPELINE-QUARANTINE round 3: keep the two roles on DIFFERENT providers
2728
+ // unless the caller asked for both explicitly. `--implementer codex` alone
2729
+ // leaves the reviewer at its codex default, so cross-model review silently
2730
+ // becomes Codex reviewing its own work — which is exactly the defect the
2731
+ // review-fix pipeline was corrected for, reintroduced one layer down. When
2732
+ // only one role is overridden, the other flips to the opposite provider.
2733
+ // Setting both to the same provider stays possible, but only deliberately.
2734
+ if (parseAgentString(implementerAgent).provider === parseAgentString(reviewerAgent).provider) {
2735
+ const bothExplicit = opts.implementer != null && opts.reviewer != null;
2736
+ if (!bothExplicit) {
2737
+ const flipped = parseAgentString(implementerAgent).provider === 'codex' ? 'claude' : 'codex';
2738
+ if (opts.reviewer == null) reviewerAgent = flipped;
2739
+ else implementerAgent = flipped;
2740
+ } else {
2741
+ console.warn(
2742
+ `⚠ implementer and reviewer are both ${parseAgentString(implementerAgent).provider}; ` +
2743
+ 'cross-model review is disabled for this run.'
2744
+ );
2745
+ }
2746
+ }
2399
2747
  const roles = { implementerAgent, reviewerAgent };
2400
2748
  // Restore persisted roles when (and only when) a resume actually happens.
2401
2749
  const restoreRolesFromActive = (src) => {
@@ -2562,7 +2910,10 @@ export async function runBuild(featureCode, opts = {}) {
2562
2910
  // The plan/resume above only created the flow object; no agent/worktree work has
2563
2911
  // happened yet, so aborting here still means we never reach `execute` (Codex
2564
2912
  // review finding #2). Cached + skippable via COMPOSE_SKIP_CODEX_PROBE.
2565
- if (implementerAgent === 'codex') {
2913
+ // Compare the PROVIDER, not the raw string: `codex:orchestrator` and
2914
+ // `codex::critical` are Codex implementers too, and an exact-string check let
2915
+ // them skip the mandatory worktree probe entirely (round 4 review).
2916
+ if (parseAgentString(implementerAgent).provider === 'codex') {
2566
2917
  const probe = await preflightCodexWorktreeProbe({
2567
2918
  cwd: agentCwd,
2568
2919
  projectCwd: cwd,
@@ -2667,6 +3018,9 @@ export async function runBuild(featureCode, opts = {}) {
2667
3018
  })();
2668
3019
 
2669
3020
  const context = {
3021
+ stratum,
3022
+ flowId: response.runId,
3023
+ receiptsMode,
2670
3024
  cwd: agentCwd,
2671
3025
  projectCwd: cwd,
2672
3026
  featureCode,
@@ -2691,24 +3045,33 @@ export async function runBuild(featureCode, opts = {}) {
2691
3045
  filesChanged: [...activeAccumulator.files_changed],
2692
3046
  ...(isBugMode ? { bug_code: featureCode } : {}),
2693
3047
  };
2694
- context.recordBuildUsage = (usage) => {
3048
+ context.recordBuildUsage = async (usage, meta) => {
2695
3049
  if (!usage || typeof usage !== 'object') return;
2696
- const componentTokens = (typeof usage.input_tokens === 'number' ? usage.input_tokens : 0)
2697
- + (typeof usage.output_tokens === 'number' ? usage.output_tokens : 0);
3050
+ const accumulatorUsage = Array.isArray(usage)
3051
+ ? usage.reduce((sum, entry) => ({
3052
+ input_tokens: sum.input_tokens + (entry?.input_tokens ?? 0),
3053
+ output_tokens: sum.output_tokens + (entry?.output_tokens ?? entry?.tokens ?? 0),
3054
+ cost_usd: sum.cost_usd + (entry?.cost_usd ?? entry?.usd ?? 0),
3055
+ }), { input_tokens: 0, output_tokens: 0, cost_usd: 0 })
3056
+ : usage;
3057
+ const componentTokens = (typeof accumulatorUsage.input_tokens === 'number' ? accumulatorUsage.input_tokens : 0)
3058
+ + (typeof accumulatorUsage.output_tokens === 'number' ? accumulatorUsage.output_tokens : 0);
2698
3059
  const tokens = componentTokens > 0
2699
3060
  ? componentTokens
2700
- : (typeof usage.tokens_total === 'number'
2701
- ? usage.tokens_total
2702
- : (typeof usage.tokens === 'number' ? usage.tokens : 0));
2703
- const usd = typeof usage.cost_usd === 'number'
2704
- ? usage.cost_usd
2705
- : (typeof usage.usd === 'number' ? usage.usd : 0);
2706
- if (tokens === 0 && usd === 0) return;
2707
- updateBuildAccumulator(cwd, featureCode, (accumulator) => ({
2708
- ...accumulator,
2709
- tokens_total: accumulator.tokens_total + tokens,
2710
- usd: accumulator.usd + usd,
2711
- }));
3061
+ : (typeof accumulatorUsage.tokens_total === 'number'
3062
+ ? accumulatorUsage.tokens_total
3063
+ : (typeof accumulatorUsage.tokens === 'number' ? accumulatorUsage.tokens : 0));
3064
+ const usd = typeof accumulatorUsage.cost_usd === 'number'
3065
+ ? accumulatorUsage.cost_usd
3066
+ : (typeof accumulatorUsage.usd === 'number' ? accumulatorUsage.usd : 0);
3067
+ if (tokens !== 0 || usd !== 0) {
3068
+ updateBuildAccumulator(cwd, featureCode, (accumulator) => ({
3069
+ ...accumulator,
3070
+ tokens_total: accumulator.tokens_total + tokens,
3071
+ usd: accumulator.usd + usd,
3072
+ }));
3073
+ }
3074
+ return reportUsageReceipts(context, usage, meta);
2712
3075
  };
2713
3076
  context.recordFilesChanged = (paths, { authoritativeShip = false } = {}) => {
2714
3077
  const normalized = Array.isArray(paths)
@@ -2721,6 +3084,27 @@ export async function runBuild(featureCode, opts = {}) {
2721
3084
  : { files_changed: [...new Set([...accumulator.files_changed, ...normalized])] }),
2722
3085
  }));
2723
3086
  };
3087
+ // COMP-COMPLETION-GATE slice 2: the ship step's completion evidence, carried
3088
+ // to terminalization (where completion now happens, after the health gate).
3089
+ //
3090
+ // `tests_attested` and `evidence_root` are PERSISTED because they cannot be
3091
+ // re-derived later: re-running the suite at terminalization would be a second
3092
+ // run with a different result, and `runBuild` rebuilds the agent cwd from the
3093
+ // current invocation — so a resumed cross-repo build would otherwise verify
3094
+ // the wrong repository. The commit SHA is deliberately NOT persisted; it is
3095
+ // resolved from HEAD at terminalization, where it is verifiable.
3096
+ context.completionEvidence = null;
3097
+ context.recordCompletionEvidence = (evidence = {}) => {
3098
+ context.completionEvidence = { ...(context.completionEvidence || {}), ...evidence };
3099
+ const attested = evidence.testsAttested;
3100
+ if (attested !== undefined) {
3101
+ updateBuildAccumulator(cwd, featureCode, (accumulator) => ({
3102
+ ...accumulator,
3103
+ tests_attested: attested,
3104
+ evidence_root: agentCwd,
3105
+ }));
3106
+ }
3107
+ };
2724
3108
  context.recordShipTestMetrics = (metrics) => {
2725
3109
  if (!metrics || typeof metrics.test_count !== 'number') return;
2726
3110
  updateBuildAccumulator(cwd, featureCode, (accumulator) => ({
@@ -2779,6 +3163,11 @@ export async function runBuild(featureCode, opts = {}) {
2779
3163
  // is the backstop — if the round can't be threaded for any reason, it trips
2780
3164
  // instead of letting the gate spin unbounded (the 52-round loop).
2781
3165
  const gateReentries = new Map();
3166
+ // Last consumer-merge preparation/apply failure per merge gate. A failure
3167
+ // that repeats byte-identically means the fan-out re-produced the same
3168
+ // conflict; revising again only re-dispatches every lane for the same result
3169
+ // (observed 2026-08-30: 4 paid rounds on one MERGE_WITNESS_PRECOMPUTE_FAILED).
3170
+ const consumerMergeFailures = new Map();
2782
3171
 
2783
3172
  // The run's effective-spec digest, carried on plan/resume responses only
2784
3173
  // (step_done responses omit it). Captured so the merge-gate path can pin the
@@ -3080,14 +3469,12 @@ export async function runBuild(featureCode, opts = {}) {
3080
3469
  };
3081
3470
  }
3082
3471
  verifyPipelineIntegrity(specPath, specFileHash);
3472
+ // `plan_items` is declared by build.stratum.yaml's PhaseResult but NOT
3473
+ // by gsd's, so it stays at this call site rather than in the shared
3474
+ // narrowing helper.
3083
3475
  const tsShipOutput = {
3084
- phase: shipResult.phase,
3085
- artifact: shipResult.artifact,
3086
- outcome: shipResult.outcome,
3087
- summary: shipResult.summary,
3476
+ ...toPhaseResultOutput(shipResult),
3088
3477
  ...(Array.isArray(shipResult.plan_items) ? { plan_items: shipResult.plan_items } : {}),
3089
- ...(Array.isArray(shipResult.filesChanged) ? { files_changed: shipResult.filesChanged } : {}),
3090
- ...(typeof shipResult.commit === 'string' ? { commit_hash: shipResult.commit } : {}),
3091
3478
  };
3092
3479
  const shipStepResult = response.status === 'ready'
3093
3480
  ? { output: tsShipOutput }
@@ -3124,7 +3511,14 @@ export async function runBuild(featureCode, opts = {}) {
3124
3511
  maxDurationMs: STEP_TIMEOUT_MS[stepId] ?? DEFAULT_TIMEOUT_MS,
3125
3512
  stratum,
3126
3513
  cwd: agentCwd,
3127
- profile: resolveStepProfile(context.stepProfiles, stepId),
3514
+ // C9: the sidecar's `fix` profile carries the fixer's tool
3515
+ // restrictions and model tier. Passing the bare agent literal
3516
+ // here handed the fixer an unrestricted profile; the sibling
3517
+ // review-repair site keys off `fix` the same way. Identity still
3518
+ // comes from the dispatch (`agent: fixAgent`).
3519
+ profile: resolveStepProfile(context.stepProfiles, 'fix')
3520
+ ?? resolveStepProfile(context.stepProfiles, stepId),
3521
+ sandboxMode: 'workspace-write',
3128
3522
  telemetry: {
3129
3523
  site: 'review-repair',
3130
3524
  project_cwd: cwd,
@@ -3135,11 +3529,23 @@ export async function runBuild(featureCode, opts = {}) {
3135
3529
  },
3136
3530
  });
3137
3531
  if (fixResult?.usage && typeof context.recordBuildUsage === 'function') {
3138
- try { context.recordBuildUsage(fixResult.usage); } catch { /* fail-open */ }
3532
+ try {
3533
+ await context.recordBuildUsage(usagePayload(fixResult.usage, fixResult.usages), {
3534
+ stepId,
3535
+ source: 'fixer',
3536
+ dispatchId: fixResult.dispatchIds?.primary,
3537
+ });
3538
+ } catch { /* fail-open */ }
3139
3539
  }
3140
3540
  } catch (err) {
3141
3541
  if (err?.usage && typeof context.recordBuildUsage === 'function') {
3142
- try { context.recordBuildUsage(err.usage); } catch { /* fail-open */ }
3542
+ try {
3543
+ await context.recordBuildUsage(usagePayload(err.usage, err.usages), {
3544
+ stepId,
3545
+ source: 'fixer',
3546
+ dispatchId: err.dispatchId,
3547
+ });
3548
+ } catch { /* fail-open */ }
3143
3549
  }
3144
3550
  if (err instanceof AgentTimeoutError) {
3145
3551
  console.warn(`\n⚠ Fix agent timed out on "${stepId}"`);
@@ -3154,7 +3560,9 @@ export async function runBuild(featureCode, opts = {}) {
3154
3560
  const stepStartMs = Date.now();
3155
3561
  const agentType = readyStep?.agent ?? response.agent ?? 'claude';
3156
3562
  const basePrompt = buildStepPrompt(stepDispatch, context);
3157
- const maxDurationMs = STEP_TIMEOUT_MS[stepId] ?? DEFAULT_TIMEOUT_MS;
3563
+ const maxDurationMs = process.env.NODE_ENV === 'test' && Number.isFinite(opts.stepTimeoutMs)
3564
+ ? opts.stepTimeoutMs
3565
+ : (STEP_TIMEOUT_MS[stepId] ?? DEFAULT_TIMEOUT_MS);
3158
3566
 
3159
3567
  // MF-1/SF-4: Prepend shared review scaffold when this is a review step.
3160
3568
  // Also covers a ReviewResult merge step so its output is normalized via
@@ -3209,14 +3617,19 @@ export async function runBuild(featureCode, opts = {}) {
3209
3617
  profile: resolveStepProfile(effectiveProfiles, stepId),
3210
3618
  });
3211
3619
  } catch (err) {
3620
+ const failedUsage = failureUsageFields(err);
3212
3621
  if (err instanceof UserInterruptError) {
3213
3622
  if (err.action === 'skip') {
3214
3623
  if (progress) progress.info(` ⏭ Skipped step "${stepId}"`);
3215
3624
  mainResult = {
3216
3625
  text: '',
3217
- result: { outcome: 'skipped', summary: 'Skipped by user' },
3626
+ // `phase` is carried because every pipeline's result contract
3627
+ // declares it and engine contracts are strict — without it a
3628
+ // user-initiated skip fails the step it was meant to bypass.
3629
+ result: { phase: stepId, outcome: 'skipped', summary: 'Skipped by user' },
3218
3630
  dispatchIds: { primary: err.dispatchId ?? null, repair: null },
3219
3631
  settlementFailureClass: 'agent',
3632
+ ...failedUsage,
3220
3633
  };
3221
3634
  } else {
3222
3635
  if (progress) progress.info(` ↻ Retrying step "${stepId}"`);
@@ -3225,6 +3638,7 @@ export async function runBuild(featureCode, opts = {}) {
3225
3638
  result: { outcome: 'failed', summary: 'Retry requested by user' },
3226
3639
  dispatchIds: { primary: err.dispatchId ?? null, repair: null },
3227
3640
  settlementFailureClass: 'agent',
3641
+ ...failedUsage,
3228
3642
  };
3229
3643
  }
3230
3644
  } else if (err instanceof AgentTimeoutError) {
@@ -3235,18 +3649,115 @@ export async function runBuild(featureCode, opts = {}) {
3235
3649
  result: { outcome: 'failed', summary: `Timed out after ${Math.round(err.durationMs / 1000)}s` },
3236
3650
  dispatchIds: { primary: err.dispatchId ?? null, repair: null },
3237
3651
  settlementFailureClass: 'agent',
3238
- ...(err.usage ? { usage: err.usage } : {}),
3652
+ ...failedUsage,
3239
3653
  };
3240
3654
  } else {
3241
3655
  // Fatal rethrow bypasses the post-stepDone usage fold — bill the
3242
3656
  // attempt's real cost to the accumulator before crashing.
3243
- if (err?.usage && typeof context.recordBuildUsage === 'function') {
3244
- try { context.recordBuildUsage(err.usage); } catch { /* fail-open */ }
3657
+ if (failedUsage.usage && typeof context.recordBuildUsage === 'function') {
3658
+ try {
3659
+ await context.recordBuildUsage(usagePayload(failedUsage.usage, failedUsage.usages), {
3660
+ stepId,
3661
+ source: 'main',
3662
+ dispatchId: err.dispatchId,
3663
+ });
3664
+ } catch { /* fail-open */ }
3245
3665
  }
3246
3666
  streamWriter.write({ type: 'build_error', message: err.message, stepId });
3247
3667
  throw err;
3248
3668
  }
3249
3669
  }
3670
+ // COMP-POLICY-CHECK-3/4: scan the candidate response against the local
3671
+ // adherence catalog before it is accepted, then allow the agent exactly
3672
+ // one revision pass. Never hard-blocks (design: "surfaces violations for
3673
+ // revision; it does not refuse to emit") and never rewrites the draft.
3674
+ let policyViolationStrings = [];
3675
+ let policyUnsuppressedCount = 0;
3676
+ {
3677
+ const skillGated = isGateStep(localSpec, localFlowName, stepId);
3678
+ const traceArgs = { cwd, streamWriter, stepId, featureCode, buildId: build_id };
3679
+
3680
+ let scan = policyScanForStep({ cwd, text: mainResult?.text, skillGated });
3681
+ recordPolicyScan({ ...traceArgs, records: scan.records, userMode: scan.userMode, pass: 'initial' });
3682
+
3683
+ if (scan.violations.length > 0) {
3684
+ if (progress) progress.warn(`Policy check: ${scan.violations.length} unsuppressed violation(s) — requesting one revision`);
3685
+ try {
3686
+ const revised = await runAndNormalize(
3687
+ null,
3688
+ `${prompt}\n\n${buildRevisionNotice(scan.records)}`,
3689
+ stepDispatch,
3690
+ {
3691
+ progress, streamWriter, maxDurationMs, stratum, cwd: agentCwd,
3692
+ reviewMode: isReviewMain,
3693
+ confidenceGate: confGateMain,
3694
+ profile: resolveStepProfile(effectiveProfiles, stepId),
3695
+ telemetry: {
3696
+ site: 'policy-revision',
3697
+ project_cwd: cwd,
3698
+ build_id,
3699
+ feature_code: featureCode,
3700
+ step_id: stepId,
3701
+ ...(typeof readyStep?.attempt === 'number' ? { attempt: readyStep.attempt } : {}),
3702
+ },
3703
+ },
3704
+ );
3705
+ const replaces = revised && !revised.normalizationFailure && revised.result?.outcome !== 'failed';
3706
+ if (!replaces && revised?.usage && typeof context.recordBuildUsage === 'function') {
3707
+ // Rejected revision: its cost still happened, and no merged
3708
+ // usage will carry it, so bill it like the review fixer's.
3709
+ try {
3710
+ await context.recordBuildUsage(usagePayload(revised.usage, revised.usages), {
3711
+ stepId,
3712
+ source: 'policy_revision',
3713
+ dispatchId: revised.dispatchIds?.primary,
3714
+ });
3715
+ } catch { /* fail-open */ }
3716
+ }
3717
+ if (replaces) {
3718
+ await reportUsageReceipts(
3719
+ context,
3720
+ usagePayload(revised.usage, revised.usages),
3721
+ { stepId, source: 'policy_revision', dispatchId: revised.dispatchIds?.primary },
3722
+ );
3723
+ // The replacement carries BOTH dispatch ids (so settlement
3724
+ // settles both) and the summed usage (so step_usage, build
3725
+ // totals, and build-history include the revision's cost — the
3726
+ // step_usage block below is the single accumulator call).
3727
+ mainResult = {
3728
+ ...mainResult,
3729
+ text: revised.text ?? mainResult.text,
3730
+ result: revised.result ?? mainResult.result,
3731
+ usage: mergeUsage(mainResult.usage, revised.usage),
3732
+ dispatchIds: {
3733
+ ...(mainResult.dispatchIds ?? {}),
3734
+ revision: revised.dispatchIds?.primary ?? null,
3735
+ },
3736
+ };
3737
+ scan = policyScanForStep({ cwd, text: mainResult.text, skillGated });
3738
+ recordPolicyScan({ ...traceArgs, records: scan.records, userMode: scan.userMode, pass: 'policy_revision' });
3739
+ }
3740
+ // The second result stands either way — no further passes.
3741
+ } catch (err) {
3742
+ if (err?.usage && typeof context.recordBuildUsage === 'function') {
3743
+ try {
3744
+ await context.recordBuildUsage(usagePayload(err.usage, err.usages), {
3745
+ stepId,
3746
+ source: 'policy_revision',
3747
+ dispatchId: err.dispatchId,
3748
+ });
3749
+ } catch { /* fail-open */ }
3750
+ }
3751
+ if (policyRevisionMustStop(err)) throw err;
3752
+ // eslint-disable-next-line no-console
3753
+ console.warn(`[policy-check] revision pass failed on "${stepId}" — keeping the original draft: ${err.message}`);
3754
+ }
3755
+ }
3756
+
3757
+ policyViolationStrings = scan.violations;
3758
+ policyUnsuppressedCount = scan.violations.length;
3759
+ }
3760
+
3250
3761
  const { result, text: stepText, usage: stepUsage, normalizationFailure } = mainResult;
3251
3762
 
3252
3763
  // Scan agent output for "we should X" / "we could X" patterns that don't map
@@ -3401,17 +3912,36 @@ export async function runBuild(featureCode, opts = {}) {
3401
3912
  // lens rather than a normalization-stamped 'general'.
3402
3913
  lastReviewMergeDirtyLenses = extractDirtyLenses(stepText ?? result);
3403
3914
  }
3915
+ // COMP-POLICY-CHECK-6: expose the unsuppressed count on the step result
3916
+ // so a spec can declare `ensure: ['result.unsuppressed_violations == 0']`.
3917
+ // Engine contracts are strict, so the field is attached only where it is
3918
+ // declared (or where the step has no out contract).
3919
+ const policyResult = attachPolicyCount(result, policyUnsuppressedCount, stepDispatch);
3404
3920
  const stepDoneResult = readyStep
3405
3921
  ? blockingFailure
3406
3922
  ? { failure: blockingFailure }
3407
3923
  : normalizationFailure || result?.outcome === 'failed'
3408
3924
  ? { failure: String(normalizationFailure ?? result?.summary ?? `Step "${stepId}" did not produce structured output`) }
3409
3925
  : stepDispatch.has_out_contract
3410
- ? result != null
3411
- ? { output: result }
3926
+ ? policyResult != null
3927
+ ? { output: policyResult }
3412
3928
  : { failure: `Step "${stepId}" did not produce structured output` }
3413
3929
  : {}
3414
- : result ?? { summary: 'Step complete' };
3930
+ : policyResult ?? { summary: 'Step complete' };
3931
+ // Report each model dispatch before the outcome it funded. The merged
3932
+ // step usage remains the accumulator/build-stream shape used by existing
3933
+ // callers; receipts use mainResult.usages to preserve per-dispatch data.
3934
+ if (toEngineUsage(stepUsage)) {
3935
+ buildCostTotals.input_tokens += stepUsage.input_tokens ?? 0;
3936
+ buildCostTotals.output_tokens += stepUsage.output_tokens ?? 0;
3937
+ buildCostTotals.cost_usd += stepUsage.cost_usd ?? 0;
3938
+ await context.recordBuildUsage(usagePayload(stepUsage, mainResult.usages), {
3939
+ stepId,
3940
+ source: 'main',
3941
+ dispatchId: mainResult.dispatchIds?.primary,
3942
+ });
3943
+ streamWriter.writeUsage(stepId, stepUsage);
3944
+ }
3415
3945
  response = await stratum.stepDone(
3416
3946
  flowId, stepId, stepDoneResult, readyStep?.dispatchToken,
3417
3947
  );
@@ -3528,17 +4058,10 @@ export async function runBuild(featureCode, opts = {}) {
3528
4058
  {
3529
4059
  const buildState = readActiveBuild(dataDir);
3530
4060
  const stepState = buildState?.steps?.find(s => s.id === stepId) ?? {};
3531
- // COMP-OBS-COST: accumulate step usage and emit step_usage event
3532
- if (stepUsage && (stepUsage.input_tokens > 0 || stepUsage.output_tokens > 0 || stepUsage.cost_usd > 0)) {
3533
- buildCostTotals.input_tokens += stepUsage.input_tokens ?? 0;
3534
- buildCostTotals.output_tokens += stepUsage.output_tokens ?? 0;
3535
- buildCostTotals.cost_usd += stepUsage.cost_usd ?? 0;
3536
- context.recordBuildUsage(stepUsage);
3537
- streamWriter.writeUsage(stepId, stepUsage);
3538
- }
3539
-
3540
- // COMP-HEALTH: collect runtime violations for health score signal
3541
- const stepViolations = stepState.violations ?? [];
4061
+ // COMP-HEALTH: collect runtime violations for health score signal.
4062
+ // COMP-POLICY-CHECK-3: unsuppressed policy violations join the same
4063
+ // stream ViolationDetail renders them with zero UI changes.
4064
+ const stepViolations = [...(stepState.violations ?? []), ...policyViolationStrings];
3542
4065
  if (stepViolations.length > 0) {
3543
4066
  allViolations.push(...stepViolations);
3544
4067
  }
@@ -3684,10 +4207,18 @@ export async function runBuild(featureCode, opts = {}) {
3684
4207
  ) ?? null;
3685
4208
  }
3686
4209
  }
4210
+ const repairFor = (error) => {
4211
+ const failure = `${error.code}: ${error.message}`;
4212
+ const decision = decideMergeRepairOutcome(
4213
+ consumerMergeFailures.get(stepId), failure, repairOutcome,
4214
+ );
4215
+ consumerMergeFailures.set(stepId, failure);
4216
+ outcome = decision.outcome;
4217
+ rationale = decision.rationale;
4218
+ };
3687
4219
  if (consumerMergeArtifacts && outcome === 'approve') {
3688
4220
  if (consumerMergePreparationError) {
3689
- outcome = repairOutcome;
3690
- rationale = `${consumerMergePreparationError.code}: ${consumerMergePreparationError.message}`;
4221
+ repairFor(consumerMergePreparationError);
3691
4222
  } else {
3692
4223
  try {
3693
4224
  await consumerMergeArtifacts.applyMerge(consumerMergeTransaction);
@@ -3705,8 +4236,7 @@ export async function runBuild(featureCode, opts = {}) {
3705
4236
  } catch { /* best-effort build context projection */ }
3706
4237
  } catch (error) {
3707
4238
  if (!(error instanceof ConsumerMergeDecisionError)) throw error;
3708
- outcome = repairOutcome;
3709
- rationale = `${error.code}: ${error.message}`;
4239
+ repairFor(error);
3710
4240
  }
3711
4241
  }
3712
4242
  }
@@ -3792,8 +4322,18 @@ export async function runBuild(featureCode, opts = {}) {
3792
4322
  let reviewResult = lastReviewMergeResult;
3793
4323
  let rawDirtyLenses = lastReviewMergeDirtyLenses;
3794
4324
  if (!reviewResult) {
3795
- const auditedOutput = gateAudit?.steps?.review_merge?.output ?? null;
3796
- if (auditedOutput && typeof auditedOutput === 'object') {
4325
+ // COMP-PIPELINE-QUARANTINE follow-up: the re-derive used to read
4326
+ // `steps.review_merge.output` by that literal name, so any pipeline
4327
+ // whose reducer is called something else (team-review's `merge`,
4328
+ // review-fix's `review`) silently got a null stash on resume and had
4329
+ // its clean review treated as dirty. The stash above is already keyed
4330
+ // on the sidecar's _reduceSteps; this now matches it, with the
4331
+ // canonical id kept as the fallback.
4332
+ const reducerIds = [...reduceSteps, 'review_merge'];
4333
+ const auditedOutput = reducerIds
4334
+ .map((id) => gateAudit?.steps?.[id]?.output ?? null)
4335
+ .find((out) => out && typeof out === 'object') ?? null;
4336
+ if (auditedOutput) {
3797
4337
  reviewResult = auditedOutput;
3798
4338
  rawDirtyLenses = extractDirtyLenses(auditedOutput);
3799
4339
  }
@@ -3833,7 +4373,7 @@ export async function runBuild(featureCode, opts = {}) {
3833
4373
  try {
3834
4374
  const gateFixResult = await runAndNormalize(undefined, fixPrompt, { step_id: 'review_fix', agent: fixAgent, flow_id: flowId }, {
3835
4375
  progress, streamWriter, maxDurationMs: STEP_TIMEOUT_MS.review_merge ?? DEFAULT_TIMEOUT_MS,
3836
- stratum, cwd: agentCwd, profile: resolveStepProfile(effectiveProfiles, 'fix'),
4376
+ stratum, cwd: agentCwd, sandboxMode: 'workspace-write', profile: resolveStepProfile(effectiveProfiles, 'fix'),
3837
4377
  telemetry: {
3838
4378
  site: 'review-repair',
3839
4379
  project_cwd: cwd,
@@ -3844,11 +4384,23 @@ export async function runBuild(featureCode, opts = {}) {
3844
4384
  },
3845
4385
  });
3846
4386
  if (gateFixResult?.usage && typeof context.recordBuildUsage === 'function') {
3847
- try { context.recordBuildUsage(gateFixResult.usage); } catch { /* fail-open */ }
4387
+ try {
4388
+ await context.recordBuildUsage(usagePayload(gateFixResult.usage, gateFixResult.usages), {
4389
+ stepId: gateStepId,
4390
+ source: 'gate_fixer',
4391
+ dispatchId: gateFixResult.dispatchIds?.primary,
4392
+ });
4393
+ } catch { /* fail-open */ }
3848
4394
  }
3849
4395
  } catch (err) {
3850
4396
  if (err?.usage && typeof context.recordBuildUsage === 'function') {
3851
- try { context.recordBuildUsage(err.usage); } catch { /* fail-open */ }
4397
+ try {
4398
+ await context.recordBuildUsage(usagePayload(err.usage, err.usages), {
4399
+ stepId: gateStepId,
4400
+ source: 'gate_fixer',
4401
+ dispatchId: err.dispatchId,
4402
+ });
4403
+ } catch { /* fail-open */ }
3852
4404
  }
3853
4405
  if (!(err instanceof AgentTimeoutError)) throw err;
3854
4406
  console.warn('\n⚠ Review fixer timed out');
@@ -4015,6 +4567,9 @@ export async function runBuild(featureCode, opts = {}) {
4015
4567
  }
4016
4568
 
4017
4569
  // Flow complete — write terminal state (file retained per STRAT-COMP-4 contract).
4570
+ // COMP-COMPLETION-GATE slice 2: set when the flow completed and the feature is
4571
+ // eligible to be completed — the actual completion runs after the health gate.
4572
+ let pendingCompletion = null;
4018
4573
  if (response.status === 'completed') buildStatus = 'complete';
4019
4574
  if ((response.status === 'failed' || response.status === 'budget_exhausted') && !killedByGate) {
4020
4575
  buildStatus = 'failed';
@@ -4022,20 +4577,19 @@ export async function runBuild(featureCode, opts = {}) {
4022
4577
  }
4023
4578
  if (response.status === 'completed' && buildStatus === 'complete') {
4024
4579
  console.log('\nBuild complete.');
4025
- await visionWriter.updateItemStatus(itemId, 'complete');
4026
- // COMP-QA: persist filesChanged so `compose qa-scope` can read them post-build.
4027
- // Bug mode skips feature-json bugs don't have feature.json (COMP-FIX-HARD T4).
4028
- if (cfg.tracksFeatureJson) {
4029
- const _bp = await getBuildProvider(cwd);
4030
- // Guard: feature.json may not exist when triage was skipped (test harnesses).
4031
- // Original updateFeature silently no-oped when feature was missing.
4032
- // Single atomic raw write (status + filesChanged together) — restores original
4033
- // updateFeature atomicity. persistFeatureRaw: no policy, no events, no roadmap.
4034
- const _feat = await _bp.getFeature(featureCode);
4035
- if (_feat) {
4036
- await _bp.persistFeatureRaw(featureCode, { ..._feat, status: 'COMPLETE', filesChanged: context.filesChanged ?? [] });
4037
- }
4038
- }
4580
+ // COMP-COMPLETION-GATE slice 2: the COMPLETION does not happen here.
4581
+ //
4582
+ // This block used to flip the vision item and feature.json to COMPLETE
4583
+ // immediately — but the health gate below can still downgrade the build to
4584
+ // `failed`, and the guard ledger is append-only. Completing here meant a
4585
+ // health-rejected build was left marked COMPLETE, and (once gated) the very
4586
+ // first thing the ledger would ever durably attest would be a build the
4587
+ // system itself then judged a failure.
4588
+ //
4589
+ // The health verdict is a PRECONDITION of completion, not its successor, so
4590
+ // the completion is deferred to the gated block below, which runs after the
4591
+ // health gate. See COMP-COMPLETION-GATE design §2.3c.
4592
+ pendingCompletion = { itemId, featureCode };
4039
4593
  const termState = readActiveBuild(dataDir);
4040
4594
  if (termState) {
4041
4595
  writeActiveBuild(dataDir, { ...termState, status: 'complete', completedAt: new Date().toISOString() });
@@ -4095,16 +4649,29 @@ export async function runBuild(featureCode, opts = {}) {
4095
4649
  buildSignals.runtime_errors = [];
4096
4650
  }
4097
4651
 
4098
- // Doc freshness — check staleness of feature artifacts
4652
+ // Doc freshness — derivation-based staleness (COMP-PROV-LINEAGE). An
4653
+ // artifact is stale when an upstream it wasDerivedFrom is newer than it.
4654
+ // This replaced the old phase-marker staleness reader (now removed),
4655
+ // which read a `<!-- phase: -->` marker that no production writer ever
4656
+ // emitted (a dead signal). findStaleArtifacts works off the canonical
4657
+ // chain + mtimes, so it needs no marker to be written first.
4099
4658
  try {
4100
- const { checkStaleness } = await import('./staleness.js');
4101
- const currentPhase = stepHistory.length > 0
4102
- ? stepHistory[stepHistory.length - 1].stepId
4103
- : 'build';
4104
- const stalenessResults = checkStaleness(resolveItemDir(featureCode), currentPhase);
4105
- buildSignals.doc_freshness = stalenessResults;
4659
+ const { findStaleArtifacts } = await import('./lineage.js');
4660
+ buildSignals.doc_freshness = findStaleArtifacts(resolveItemDir(featureCode));
4106
4661
  } catch { /* staleness check is optional — skip on error */ }
4107
4662
 
4663
+ // COMP-PROV-LINEAGE — populate PROV-O lineage markers on the canonical
4664
+ // artifacts that now exist. This runs once per build, in the finalization
4665
+ // pass after the dispatch loop, when the artifact set is complete. Build
4666
+ // is the single writer here, so the read/write/utimes in stampFeatureLineage
4667
+ // is uncontended. Idempotent and mtime-preserving, so it never resets the
4668
+ // derivation clock that staleness reachability depends on. This is the
4669
+ // lifecycle-writer surface that materialises wasGeneratedBy/wasDerivedFrom.
4670
+ try {
4671
+ const { stampFeatureLineage } = await import('./lineage.js');
4672
+ stampFeatureLineage(resolveItemDir(featureCode));
4673
+ } catch { /* lineage stamping is optional — skip on error */ }
4674
+
4108
4675
  const healthSettings = (() => {
4109
4676
  try {
4110
4677
  if (existsSync(settingsPath)) {
@@ -4167,6 +4734,97 @@ export async function runBuild(featureCode, opts = {}) {
4167
4734
  }
4168
4735
  }
4169
4736
 
4737
+ // ---------------------------------------------------------------------
4738
+ // COMP-COMPLETION-GATE slice 2 — THE completion, and the only one.
4739
+ //
4740
+ // Runs here, after the health gate above may have downgraded buildStatus, so
4741
+ // a health-rejected build completes nothing: no completion record, no
4742
+ // COMPLETE status, no vision completion, no guard transition.
4743
+ // ---------------------------------------------------------------------
4744
+ if (pendingCompletion && buildStatus === 'complete') {
4745
+ const ev = context.completionEvidence || {};
4746
+ const acc = readBuildAccumulator(cwd, featureCode);
4747
+ // Persisted, because it survives a resume; the in-memory value wins when
4748
+ // this process ran the ship step itself.
4749
+ const testsAttested = ev.testsAttested ?? acc?.tests_attested ?? 'no-signal';
4750
+ const evidenceRoot = acc?.evidence_root || agentCwd;
4751
+
4752
+ // The SHA is resolved here rather than carried: at terminalization HEAD is
4753
+ // the commit the build produced (or, on the already-committed path, found),
4754
+ // and resolving it at the point of use keeps it verifiable instead of a
4755
+ // stale claim threaded across a resume boundary.
4756
+ let commitSha = ev.commitSha ?? null;
4757
+ if (!commitSha) {
4758
+ try {
4759
+ commitSha = execSync('git rev-parse HEAD', {
4760
+ cwd: evidenceRoot, encoding: 'utf-8', timeout: 5000, stdio: ['ignore', 'pipe', 'pipe'],
4761
+ }).trim() || null;
4762
+ } catch { /* no repo — the no-repo exemption applies below */ }
4763
+ }
4764
+
4765
+ if (cfg.tracksFeatureJson) {
4766
+ const { completionGate, guardEnabled } = await import('./completion-gate.js');
4767
+ // `capabilities.guard: false` is a deliberate opt-OUT, and the gate itself
4768
+ // honors it (completion-gate.js §"Evidence, BEFORE the lock" / AC-5). This
4769
+ // refusal has to sit INSIDE the same regime: enforcing attestation on an
4770
+ // opted-out project would break every one of them — including non-git
4771
+ // workspaces, where the evidence can never pass at all — which is exactly
4772
+ // the reversal already made once during slice 1. An opted-out project keeps
4773
+ // `deriveTestsPass`'s degrade contract: 'no-signal' reads as true there.
4774
+ const guarded = guardEnabled(cwd);
4775
+
4776
+ // `no-signal` is not an attestation. Refusing here is the whole point of
4777
+ // the tri-state: an unreadable test run must not become a passing claim
4778
+ // on a permanent record.
4779
+ if (guarded && testsAttested === 'no-signal') {
4780
+ console.warn(
4781
+ `[completion-gate] ${featureCode}: tests could not be attested (test output was ` +
4782
+ `unreadable). Configure guard.testCommand in .compose/compose.json so the test run ` +
4783
+ `itself attests, or record the completion explicitly. Build is complete; the feature ` +
4784
+ `is NOT marked COMPLETE.`,
4785
+ );
4786
+ } else {
4787
+ const gated = await completionGate({
4788
+ featureCode,
4789
+ commitSha,
4790
+ testsPass: guarded ? testsAttested === 'passed' : testsAttested !== 'failed',
4791
+ filesChanged: ev.filesChanged ?? context.filesChanged ?? [],
4792
+ notes: ev.notes,
4793
+ builtVia: ev.builtVia,
4794
+ workspaceRoot: cwd,
4795
+ evidenceRoot,
4796
+ mode: resolveMode(mode),
4797
+ // Slice 3: the gate owns the vision projection (§2.3a step 4) through
4798
+ // the self-verifying seam; `updateItemStatus(…, 'complete')` refuses
4799
+ // managed build items now (AC-16).
4800
+ visionItemId: pendingCompletion.itemId,
4801
+ visionProjector: ({ featureCode: fc, commitSha: sha, ledgerRef }) =>
4802
+ visionWriter.completeItem(pendingCompletion.itemId, { featureCode: fc, cwd, commitSha: sha, ledgerRef }),
4803
+ });
4804
+ if (gated.ok) {
4805
+ if (gated.partial) {
4806
+ console.warn(
4807
+ `[completion-gate] ${featureCode}: completed, but a projection failed — ` +
4808
+ gated.failures.map(f => `${f.step}: ${f.message} (recover: ${f.recover})`).join('; '),
4809
+ );
4810
+ }
4811
+ } else {
4812
+ // Do NOT fail the build: the work is committed and the flow finished.
4813
+ // But do not claim completion either — say plainly what was refused.
4814
+ console.warn(
4815
+ `[completion-gate] ${featureCode}: completion refused at ${gated.refusedAt} — ` +
4816
+ `${(gated.reasons || []).join('; ')}. The build finished and the commit stands; ` +
4817
+ `the feature is NOT marked COMPLETE.`,
4818
+ );
4819
+ }
4820
+ }
4821
+ } else {
4822
+ // Bug/plan modes have no feature.json to gate on (COMP-FIX-HARD T4), and
4823
+ // slice 1/2 are scoped to build mode — their completion path is unchanged.
4824
+ await visionWriter.updateItemStatus(pendingCompletion.itemId, 'complete');
4825
+ }
4826
+ }
4827
+
4170
4828
  // COMP-COCKPIT-3: archive the run to build-history.jsonl ONCE, here — after
4171
4829
  // the COMP-HEALTH gate above may have downgraded buildStatus to 'failed'.
4172
4830
  // Assembled from the in-memory build context for THIS run (never re-read
@@ -4487,6 +5145,35 @@ function judgmentCanonDriftError({ treeDrift, projectionDrift, recordDrift }) {
4487
5145
  return err;
4488
5146
  }
4489
5147
 
5148
+ /**
5149
+ * Narrow an executeShipStep return value to the fields the TS engine's
5150
+ * PhaseResult contract declares.
5151
+ *
5152
+ * COMP-SHIP-CONTRACT: engine contracts are STRICT Zod objects — any key the
5153
+ * contract does not declare fails the step. executeShipStep's return carries
5154
+ * caller-facing extras (`commit`, `filesChanged`, `testsAttested`,
5155
+ * `test_count`/`pass_rate`, `error_code`) that Compose's own code consumes but
5156
+ * PhaseResult never declared, so the raw object must NEVER be handed to
5157
+ * stepDone as an `output`. Both sites that do so (runBuild, runGsd) go through
5158
+ * here. Note runBuild's non-`ready` branch passes shipResult as the whole
5159
+ * envelope rather than as `{output}` — a legacy shape that never reaches
5160
+ * contract validation, deliberately left untouched.
5161
+ *
5162
+ * @param {object} shipResult Return value from executeShipStep
5163
+ * @returns {object} `{phase, artifact, outcome, summary}` plus
5164
+ * `files_changed`/`commit_hash` when present
5165
+ */
5166
+ export function toPhaseResultOutput(shipResult) {
5167
+ return {
5168
+ phase: shipResult.phase,
5169
+ artifact: shipResult.artifact,
5170
+ outcome: shipResult.outcome,
5171
+ summary: shipResult.summary,
5172
+ ...(Array.isArray(shipResult.filesChanged) ? { files_changed: shipResult.filesChanged } : {}),
5173
+ ...(typeof shipResult.commit === 'string' ? { commit_hash: shipResult.commit } : {}),
5174
+ };
5175
+ }
5176
+
4490
5177
  /**
4491
5178
  * Execute the ship step: run tests, stage feature files, commit.
4492
5179
  * Returns a PhaseResult-shaped object.
@@ -4509,7 +5196,35 @@ export async function executeShipStep(featureCode, agentCwd, cwd, context, descr
4509
5196
  const builtVia = context?.templateName === 'build-quick' ? 'build-quick' : null;
4510
5197
 
4511
5198
  try {
4512
- // 0. Check if we're in a git repository if not, skip git operations
5199
+ // 0. Run tests FIRSTbefore the git-availability branch.
5200
+ //
5201
+ // COMP-COMPLETION-GATE slice 2: this used to live below, after the non-git
5202
+ // branch had already returned. That meant a non-git build never ran tests at
5203
+ // all and hard-coded `tests_pass: true` on its completion record. With the
5204
+ // gate refusing an unattested completion, every non-git build would have been
5205
+ // refused — and "no repo" is a reason to skip the COMMIT, never a reason to
5206
+ // skip the tests. Both paths now attest identically.
5207
+ if (progress) progress.toolUse('ship', 'Running tests...');
5208
+ let testSummary = { test_count: 0, pass_rate: 0, parsed: false };
5209
+ try {
5210
+ // COMP-TEST-BOOTSTRAP item 128: use the detected test command, not a hard-coded `npm test`.
5211
+ const testFramework = detectTestFramework(agentCwd);
5212
+ const testCommand = testFramework?.command ?? 'npm test';
5213
+ const testOutput = execSync(`${testCommand} 2>&1 || true`, { cwd: agentCwd, encoding: 'utf-8', timeout: 120_000 });
5214
+ testSummary = parseTestSummary(testFramework?.framework, testOutput);
5215
+ } catch { /* test runner unavailable or timed out — testSummary stays unparsed */ }
5216
+ const testsPass = deriveTestsPass(testSummary);
5217
+ // The gate's input: 'passed' | 'failed' | 'no-signal'. Unlike testsPass, an
5218
+ // unreadable run does NOT become an attestation here.
5219
+ const testsAttested = deriveTestsAttested(testSummary);
5220
+ if (progress && testSummary.parsed) {
5221
+ progress.toolUse('ship', `Tests: ${testSummary.test_count} run, ${testSummary.pass_rate}% passing`);
5222
+ }
5223
+ // Hand the evidence to terminalization, which is where completion now happens
5224
+ // (after the health gate). The ship step no longer completes anything itself.
5225
+ context.recordCompletionEvidence?.({ testsAttested, testSummary });
5226
+
5227
+ // 1. Check if we're in a git repository — if not, skip git operations
4513
5228
  let isGitRepo = false;
4514
5229
  try {
4515
5230
  execSync('git rev-parse --is-inside-work-tree', { cwd: agentCwd, encoding: 'utf-8', timeout: 5000, stdio: 'pipe' });
@@ -4518,56 +5233,20 @@ export async function executeShipStep(featureCode, agentCwd, cwd, context, descr
4518
5233
 
4519
5234
  if (!isGitRepo) {
4520
5235
  // COMP-PATHS-EXTERNAL D6b: there is no repo to commit into (e.g. a
4521
- // forge-top-shaped workspace), but the lifecycle must still advance
4522
- // record a commit-less completion (null-SHA) so status flips to COMPLETE.
4523
- // Best-effort: a completion failure must not fail ship.
4524
- let completionWarning = null;
4525
- if (featureCode) {
4526
- try {
4527
- const { recordCompletion } = await import('./completion-writer.js');
4528
- await recordCompletion(cwd, {
4529
- feature_code: featureCode,
4530
- // commit_sha omitted — non-git workspace, stamped with the null-SHA
4531
- // COMP-TEST-BOOTSTRAP-4: this path returns before the test run, so
4532
- // there is no parsed signal — degrade to true (no block).
4533
- tests_pass: true,
4534
- files_changed: [],
4535
- notes: description.split('\n')[0].slice(0, 72),
4536
- ...(builtVia ? { built_via: builtVia } : {}),
4537
- });
4538
- } catch (err) {
4539
- completionWarning = `completion record failed (${err.code || 'UNKNOWN'}): ${err.message}`;
4540
- // eslint-disable-next-line no-console
4541
- console.warn(`[build/ship] ${featureCode}: ${completionWarning}`);
4542
- }
4543
- }
5236
+ // forge-top-shaped workspace). The lifecycle still advances, but the
5237
+ // completion is now written at terminalization by the completion gate,
5238
+ // AFTER the health verdict not here. See COMP-COMPLETION-GATE §2.3c.
4544
5239
  return {
4545
5240
  phase: 'ship',
4546
5241
  artifact: 'no-git',
4547
5242
  outcome: 'complete',
4548
- summary: 'No git repository — wrote artifacts, recorded completion (commit skipped)',
5243
+ summary: 'No git repository — wrote artifacts (commit skipped)',
4549
5244
  commit: null,
4550
- ...(completionWarning ? { completionWarning } : {}),
5245
+ noRepo: true,
5246
+ testsAttested,
4551
5247
  };
4552
5248
  }
4553
5249
 
4554
- // 1. Run feature-relevant tests (best-effort — don't block ship on test infra issues)
4555
- if (progress) progress.toolUse('ship', 'Running tests...');
4556
- // COMP-TEST-BOOTSTRAP-4: parse the run output into a structured signal and
4557
- // derive a real tests_pass for the completion attestation. Degrades to
4558
- // `true` (no block) whenever the output can't be parsed — see deriveTestsPass.
4559
- let testSummary = { test_count: 0, pass_rate: 0, parsed: false };
4560
- try {
4561
- // COMP-TEST-BOOTSTRAP item 128: use detected test command instead of hard-coded npm test
4562
- const testFramework = detectTestFramework(agentCwd);
4563
- const testCommand = testFramework?.command ?? 'npm test';
4564
- const testOutput = execSync(`${testCommand} 2>&1 || true`, { cwd: agentCwd, encoding: 'utf-8', timeout: 120_000 });
4565
- testSummary = parseTestSummary(testFramework?.framework, testOutput);
4566
- } catch { /* test runner not available or timed out — proceed (testSummary stays unparsed) */ }
4567
- const testsPass = deriveTestsPass(testSummary);
4568
- if (progress && testSummary.parsed) {
4569
- progress.toolUse('ship', `Tests: ${testSummary.test_count} run, ${testSummary.pass_rate}% passing`);
4570
- }
4571
5250
 
4572
5251
  // COMP-TRIAGE-5 (E3 Expand): if a lane-triaged feature fails its ship-time
4573
5252
  // test gate, escalate the lane so the NEXT build runs wider. Best-effort —
@@ -4806,33 +5485,23 @@ export async function executeShipStep(featureCode, agentCwd, cwd, context, descr
4806
5485
  if (filesChanged.length === 0 && sha) filesChanged = stagedFiles;
4807
5486
  context.recordFilesChanged?.(filesChanged, { authoritativeShip: true });
4808
5487
 
4809
- // COMP-MCP-MIGRATION: write a commit-bound completion record via the
4810
- // typed writer. The writer flips feature.status to COMPLETE atomically
4811
- // and regenerates ROADMAP.md. Best-effort: completion failures must not
4812
- // downgrade the ship outcome since the commit itself succeeded.
4813
- let completionWarning = null;
4814
- if (sha && featureCode) {
4815
- try {
4816
- const { recordCompletion } = await import('./completion-writer.js');
4817
- await recordCompletion(cwd, {
4818
- feature_code: featureCode,
4819
- commit_sha: sha,
4820
- // COMP-TEST-BOOTSTRAP-4: derived from the parsed test run above
4821
- // (true when unparseable — never a false block).
4822
- tests_pass: testsPass,
4823
- files_changed: filesChanged,
4824
- notes: shortDesc,
4825
- ...(builtVia ? { built_via: builtVia } : {}),
4826
- });
4827
- if (progress) progress.toolUse('ship', `Recorded completion for ${featureCode}`);
4828
- } catch (err) {
4829
- completionWarning = err.code === 'STATUS_FLIP_AFTER_COMPLETION_RECORDED'
4830
- ? `completion recorded but status flip failed: ${err.message}`
4831
- : `completion record failed (${err.code || 'UNKNOWN'}): ${err.message}`;
4832
- // eslint-disable-next-line no-console
4833
- console.warn(`[build/ship] ${featureCode}: ${completionWarning}`);
4834
- }
4835
- }
5488
+ // COMP-COMPLETION-GATE slice 2: the ship step no longer completes the feature.
5489
+ //
5490
+ // It used to call recordCompletion here, catch ANY failure, and still return
5491
+ // a successful ship outcome so a completion could fail silently and the
5492
+ // build marched on regardless. Worse, the terminal block then wrote COMPLETE
5493
+ // again independently, and the health gate that can fail the build runs AFTER
5494
+ // both. A health-rejected build was left marked COMPLETE.
5495
+ //
5496
+ // Ship now collects evidence and stops. Exactly one completion happens, at
5497
+ // terminalization, through the gate, after health. See §2.3, §2.3c.
5498
+ context.recordCompletionEvidence?.({
5499
+ commitSha: sha,
5500
+ filesChanged,
5501
+ notes: shortDesc,
5502
+ builtVia,
5503
+ testsAttested,
5504
+ });
4836
5505
 
4837
5506
  // COMP-PATHS-EXTERNAL D6a: if ROADMAP / the feature folder resolved into a
4838
5507
  // DIFFERENT git repo, they were written but not committed here — tell the
@@ -4848,11 +5517,11 @@ export async function executeShipStep(featureCode, agentCwd, cwd, context, descr
4848
5517
  : `Committed: ${commitMsg} (${stagedFiles.length} files)`,
4849
5518
  commit: sha,
4850
5519
  filesChanged,
5520
+ testsAttested,
4851
5521
  // COMP-MODEL-AB: thread structured test counts into the step result so the
4852
5522
  // main loop can persist them to build-history.jsonl for metrics consumers.
4853
5523
  // Only present when testSummary.parsed=true (framework detected + output parsed).
4854
5524
  ...(testSummary.parsed ? { test_count: testSummary.test_count, pass_rate: testSummary.pass_rate } : {}),
4855
- ...(completionWarning ? { completionWarning } : {}),
4856
5525
  };
4857
5526
 
4858
5527
  } catch (err) {
@@ -5025,6 +5694,35 @@ async function pollGateResolution(visionWriter, gateId, intervalMs = 2000) {
5025
5694
  */
5026
5695
  export const MAX_GATE_REENTRIES = 20;
5027
5696
 
5697
+ /**
5698
+ * Decide how a merge gate answers a consumer-merge failure.
5699
+ *
5700
+ * The first failure routes to the gate's repair path (`on_revise`, else kill).
5701
+ * A failure that repeats BYTE-IDENTICALLY for the same gate is not going to be
5702
+ * fixed by re-running the fan-out — the lanes reproduced the same conflict —
5703
+ * so the gate kills instead of paying for another round. Anything different
5704
+ * (a new code, a different file) is genuine progress and revises as before.
5705
+ *
5706
+ * @param {string|undefined} previousFailure - `${code}: ${message}` of the last failure at this gate
5707
+ * @param {string} failure - this round's `${code}: ${message}`
5708
+ * @param {'revise'|'kill'} repairOutcome - the gate's configured repair route
5709
+ * @returns {{ outcome: 'revise'|'kill', rationale: string, repeated: boolean }}
5710
+ */
5711
+ export function decideMergeRepairOutcome(previousFailure, failure, repairOutcome) {
5712
+ const repeated = previousFailure !== undefined && previousFailure === failure;
5713
+ if (repairOutcome === 'revise' && repeated) {
5714
+ return {
5715
+ outcome: 'kill',
5716
+ repeated,
5717
+ rationale: `${failure} — identical to the previous round's failure at this gate; `
5718
+ + 'the fan-out reproduces the same conflict, so revising would only re-dispatch every '
5719
+ + 'lane for the same result. Killed to stop spending. A killed build is not resumable: '
5720
+ + 'fix the conflict (usually lanes editing the same file), then re-run with --fresh.',
5721
+ };
5722
+ }
5723
+ return { outcome: repairOutcome, repeated, rationale: failure };
5724
+ }
5725
+
5028
5726
  export function assertGateReentryWithinCap(count, stepId, cap = MAX_GATE_REENTRIES) {
5029
5727
  if (count > cap) {
5030
5728
  throw new Error(
@@ -5144,3 +5842,9 @@ export async function abortBuild(dataDir, featureCode, cwd, opts = {}) {
5144
5842
  }
5145
5843
  console.log('Build aborted.');
5146
5844
  }
5845
+
5846
+ /** Optional revision failures keep the draft; only user control or uncertain teardown stops the build. */
5847
+ export function policyRevisionMustStop(error) {
5848
+ return error instanceof UserInterruptError
5849
+ || ['CANCELLATION_UNCONFIRMED', 'CANCELLATION_TEARDOWN_TIMEOUT'].includes(error?.code);
5850
+ }