@smartmemory/compose 0.3.7 → 0.3.8
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.compose-deps.json +1 -13
- package/README.md +72 -5
- package/bin/compose.js +470 -351
- package/bin/judgment-migrate.js +387 -0
- package/contracts/comp-obs-contract.schema.json +9 -3
- package/contracts/fluid-record.schema.json +209 -0
- package/contracts/lifecycle-backfill.schema.json +322 -0
- package/dist/assets/App-Z4MU-H_F.js +916 -0
- package/dist/assets/{_baseUniq-Bo837sRJ.js → _baseUniq-ClWoCPFl.js} +1 -1
- package/dist/assets/{arc-BafGpyqE.js → arc-DY26UIVo.js} +1 -1
- package/dist/assets/{architectureDiagram-Q4EWVU46-BOBfUsqL.js → architectureDiagram-Q4EWVU46-6Ggq4DqJ.js} +1 -1
- package/dist/assets/{blockDiagram-DXYQGD6D-Dwodev1a.js → blockDiagram-DXYQGD6D-CH3Ked0l.js} +1 -1
- package/dist/assets/{browser-1ntj1-x_.js → browser-BWkrenen.js} +1 -1
- package/dist/assets/{c4Diagram-AHTNJAMY-CU_bhYag.js → c4Diagram-AHTNJAMY-Bk8dYilu.js} +1 -1
- package/dist/assets/channel-SnZzzh7k.js +1 -0
- package/dist/assets/{chunk-4BX2VUAB-p8WsDwnO.js → chunk-4BX2VUAB-BMR0XaAQ.js} +1 -1
- package/dist/assets/{chunk-4TB4RGXK-B8h7-eR0.js → chunk-4TB4RGXK-JytR14a9.js} +1 -1
- package/dist/assets/{chunk-55IACEB6-DxeEr98s.js → chunk-55IACEB6-B4Q97BCP.js} +1 -1
- package/dist/assets/{chunk-EDXVE4YY-BYt8F151.js → chunk-EDXVE4YY-R_qarkSf.js} +1 -1
- package/dist/assets/{chunk-FMBD7UC4-DGSOVeie.js → chunk-FMBD7UC4-C9s7KR9m.js} +1 -1
- package/dist/assets/{chunk-OYMX7WX6-B-QdgYR2.js → chunk-OYMX7WX6-BySQzVxc.js} +1 -1
- package/dist/assets/{chunk-QZHKN3VN-Du5UAZLs.js → chunk-QZHKN3VN-DdpSYZsW.js} +1 -1
- package/dist/assets/{chunk-YZCP3GAM-C8JbNBSk.js → chunk-YZCP3GAM-iE_tzriw.js} +1 -1
- package/dist/assets/classDiagram-6PBFFD2Q-CBu92dSH.js +1 -0
- package/dist/assets/classDiagram-v2-HSJHXN6E-CBu92dSH.js +1 -0
- package/dist/assets/clone-DgklGjHm.js +1 -0
- package/dist/assets/{cose-bilkent-S5V4N54A-O1ESaqge.js → cose-bilkent-S5V4N54A-BdlU6ZX_.js} +1 -1
- package/dist/assets/{dagre-KV5264BT-CPTmFPHw.js → dagre-KV5264BT-Cp3F5KTn.js} +1 -1
- package/dist/assets/{diagram-5BDNPKRD-B3PNrWs5.js → diagram-5BDNPKRD-DiR6_2q_.js} +1 -1
- package/dist/assets/{diagram-G4DWMVQ6-Cscfr6vc.js → diagram-G4DWMVQ6-w0i-p5HX.js} +1 -1
- package/dist/assets/{diagram-MMDJMWI5-CSfqZ-TM.js → diagram-MMDJMWI5-tIHhwUv3.js} +1 -1
- package/dist/assets/{diagram-TYMM5635-Cg4aYS7W.js → diagram-TYMM5635-BAeY3B19.js} +1 -1
- package/dist/assets/{erDiagram-SMLLAGMA-_ZqwG5pl.js → erDiagram-SMLLAGMA-Ckx_Knko.js} +1 -1
- package/dist/assets/{flowDiagram-DWJPFMVM-C83boxFT.js → flowDiagram-DWJPFMVM-DeoNka6J.js} +1 -1
- package/dist/assets/{ganttDiagram-T4ZO3ILL-CWnIjuEi.js → ganttDiagram-T4ZO3ILL-BmGnFbEg.js} +1 -1
- package/dist/assets/{gitGraphDiagram-UUTBAWPF-DrMdxZfH.js → gitGraphDiagram-UUTBAWPF-Dk48IHsx.js} +1 -1
- package/dist/assets/{graph-RE4I7Ty7.js → graph-BNzKGvoy.js} +1 -1
- package/dist/assets/{graph-Bi99_6Yf.js → graph-CI_1htl0.js} +1 -1
- package/dist/assets/{index-Rm2RE-c0.js → index-BEfrNBp8.js} +3 -3
- package/dist/assets/index-yyrA5OZd.css +1 -0
- package/dist/assets/{infoDiagram-42DDH7IO-BLmP4Epr.js → infoDiagram-42DDH7IO-BRf827i0.js} +1 -1
- package/dist/assets/{ishikawaDiagram-UXIWVN3A-yuWWshKN.js → ishikawaDiagram-UXIWVN3A-0kCZaeCM.js} +1 -1
- package/dist/assets/{journeyDiagram-VCZTEJTY-BOfhaJov.js → journeyDiagram-VCZTEJTY-rvU7ayRt.js} +1 -1
- package/dist/assets/{kanban-definition-6JOO6SKY-Bbolde15.js → kanban-definition-6JOO6SKY-DpQwX1C5.js} +1 -1
- package/dist/assets/{layout-BSf33zm8.js → layout-BI8cXFPI.js} +1 -1
- package/dist/assets/{linear-AvSTWMqx.js → linear-a0glcDiw.js} +1 -1
- package/dist/assets/{min-QBM8H4xN.js → min-vPHfnXcC.js} +1 -1
- package/dist/assets/{mindmap-definition-QFDTVHPH-BuvgtqIc.js → mindmap-definition-QFDTVHPH-D14eF-7C.js} +1 -1
- package/dist/assets/mobile-B7m9EO9D.js +17 -0
- package/dist/assets/{pieDiagram-DEJITSTG-DIzF16vh.js → pieDiagram-DEJITSTG-Cno-gETh.js} +1 -1
- package/dist/assets/{quadrantDiagram-34T5L4WZ-D-mbUIjS.js → quadrantDiagram-34T5L4WZ-BUQM1Hfm.js} +1 -1
- package/dist/assets/{requirementDiagram-MS252O5E-CEs4kCLd.js → requirementDiagram-MS252O5E-pOXlN2-q.js} +1 -1
- package/dist/assets/{sankeyDiagram-XADWPNL6-DFsnCr9n.js → sankeyDiagram-XADWPNL6-Crynd3_b.js} +1 -1
- package/dist/assets/{sequenceDiagram-FGHM5R23-BEJYdTjQ.js → sequenceDiagram-FGHM5R23-D9fZdCM8.js} +1 -1
- package/dist/assets/{stateDiagram-FHFEXIEX-BBXs57uY.js → stateDiagram-FHFEXIEX-CW9qVec8.js} +1 -1
- package/dist/assets/stateDiagram-v2-QKLJ7IA2-DkVLzHbY.js +1 -0
- package/dist/assets/{timeline-definition-GMOUNBTQ-BGvLoVAY.js → timeline-definition-GMOUNBTQ-BcHzhm_8.js} +1 -1
- package/dist/assets/{vennDiagram-DHZGUBPP-9LaBTMe0.js → vennDiagram-DHZGUBPP-BfytJcWk.js} +1 -1
- package/dist/assets/{wardley-RL74JXVD-P4MEqMTP.js → wardley-RL74JXVD-DLj-IjyB.js} +1 -1
- package/dist/assets/{wardleyDiagram-NUSXRM2D-o-tmxnlC.js → wardleyDiagram-NUSXRM2D-Ds0Ue68c.js} +1 -1
- package/dist/assets/{xychartDiagram-5P7HB3ND-Dpn7V6qk.js → xychartDiagram-5P7HB3ND-vjWDXFL6.js} +1 -1
- package/dist/index.html +3 -3
- package/lib/agent-string.js +7 -5
- package/lib/append-integrity.js +81 -0
- package/lib/backfill-evidence.js +109 -0
- package/lib/bug-escalation.js +9 -0
- package/lib/build-stream-schema.js +3 -1
- package/lib/build-stream-writer.js +25 -0
- package/lib/build.js +874 -170
- package/lib/canon-guard.js +28 -6
- package/lib/canon-override.js +196 -0
- package/lib/canon-registry.js +104 -0
- package/lib/cli-commands.js +144 -0
- package/lib/codex-preflight.js +26 -13
- package/lib/colleague/context.js +215 -0
- package/lib/colleague/writeback.js +95 -0
- package/lib/completion-gate.js +1421 -0
- package/lib/completion-writer.js +47 -47
- package/lib/consumer-fanout.js +105 -11
- package/lib/coverage-gate.js +200 -0
- package/lib/dir-lock.js +170 -0
- package/lib/dispatch-ledger.js +3 -3
- package/lib/feature-json.js +1 -1
- package/lib/feature-reconciler.js +8 -0
- package/lib/feature-validator.js +64 -1
- package/lib/feature-writer.js +57 -2
- package/lib/fluid/factory.js +167 -0
- package/lib/fluid/ideabox-dates.js +73 -0
- package/lib/fluid/ideabox-migrate.js +154 -0
- package/lib/fluid/ideabox-ops.js +585 -0
- package/lib/fluid/ideabox-view.js +146 -0
- package/lib/fluid/import-ideabox.js +186 -0
- package/lib/fluid/local-provider.js +606 -0
- package/lib/fluid/provider.js +684 -0
- package/lib/fluid/record-shape.js +214 -0
- package/lib/fluid/record-store.js +328 -0
- package/lib/fluid/render-ideabox.js +261 -0
- package/lib/fluid/schema.js +40 -0
- package/lib/fluid/smartmemory-provider.js +1695 -0
- package/lib/gsd.js +63 -23
- package/lib/guard-cli.js +175 -0
- package/lib/guard-custody.js +141 -0
- package/lib/guard-descriptors.js +530 -0
- package/lib/guard-enrol.js +254 -0
- package/lib/health-score.js +1 -1
- package/lib/ideabox-cli.js +315 -0
- package/lib/ideabox.js +121 -21
- package/lib/judgment/store/index.js +9 -1
- package/lib/judgment/store/records.js +1 -1
- package/lib/judgment/trace.js +380 -0
- package/lib/judgment-decision-write.js +277 -0
- package/lib/judgment-decisions.js +466 -0
- package/lib/judgment-gen.js +5 -1
- package/lib/judgment-writer.js +56 -2
- package/lib/lifecycle-modes.js +4 -4
- package/lib/lineage.js +400 -0
- package/lib/local-claude-connector.js +52 -1
- package/lib/maya-client.js +302 -0
- package/lib/maya-config.js +53 -0
- package/lib/maya-identity.js +283 -0
- package/lib/migrate-anon.js +5 -0
- package/lib/migrate-roadmap.js +15 -0
- package/lib/new.js +13 -1
- package/lib/pipeline-compat.js +104 -0
- package/lib/policy-catalog.js +295 -0
- package/lib/policy-check.js +0 -0
- package/lib/process-termination.js +98 -0
- package/lib/resolve-workspace.js +5 -1
- package/lib/result-normalizer.js +396 -199
- package/lib/roadmap-errors.js +65 -0
- package/lib/roadmap-preservers.js +24 -4
- package/lib/roadmap-residue.js +299 -0
- package/lib/smartmemory-client.js +614 -78
- package/lib/smartmemory-config.js +54 -0
- package/lib/smartmemory-ingest.js +19 -2
- package/lib/step-prompt.js +7 -6
- package/lib/stratum-engine.js +53 -4
- package/lib/stratum-mcp-client.js +271 -36
- package/lib/test-bootstrap.js +31 -0
- package/lib/tool-inventory.js +122 -0
- package/lib/version-check.js +91 -19
- package/lib/vision-writer.js +88 -1
- package/package.json +7 -6
- package/pipelines/bug-fix.stratum.yaml +205 -211
- package/pipelines/build-quick.profiles.json +12 -0
- package/pipelines/build-quick.stratum.yaml +263 -350
- package/pipelines/content.stratum.yaml +81 -77
- package/pipelines/coverage-sweep.stratum.yaml +49 -30
- package/pipelines/plan.stratum.yaml +76 -86
- package/pipelines/refactor.stratum.yaml +125 -125
- package/pipelines/research.stratum.yaml +56 -58
- package/pipelines/review-fix.profiles.json +6 -0
- package/pipelines/review-fix.stratum.yaml +110 -83
- package/presets/team-feature.profiles.json +6 -0
- package/presets/team-feature.stratum.yaml +93 -66
- package/presets/team-research.profiles.json +6 -0
- package/presets/team-research.stratum.yaml +89 -80
- package/presets/team-review.profiles.json +8 -0
- package/presets/team-review.stratum.yaml +98 -80
- package/scripts/cost-census.mjs +70 -0
- package/scripts/guard-sign/compose-guard-sign.sh +62 -0
- package/server/agent-health.js +22 -0
- package/server/agent-hooks.js +14 -1
- package/server/agent-server.js +5 -248
- package/server/agent-spawn.js +3 -4
- package/server/agent-workspace.js +294 -0
- package/server/build-routes.js +6 -5
- package/server/build-stream-bridge.js +53 -0
- package/server/cc-session-watcher.js +4 -1
- package/server/coalescing-buffer.js +7 -1
- package/server/completion-projection.js +228 -0
- package/server/compose-mcp-tools.js +109 -23
- package/server/compose-mcp.js +88 -882
- package/server/decision-event-emit.js +41 -2
- package/server/decision-event-id.js +17 -0
- package/server/decision-events-snapshot.js +3 -0
- package/server/design-routes.js +14 -8
- package/server/feature-scan.js +76 -2
- package/server/file-watcher.js +170 -21
- package/server/ideabox-routes.js +166 -224
- package/server/index.js +70 -100
- package/server/lifecycle-guard.js +240 -10
- package/server/lifecycle-phase-history.js +276 -0
- package/server/maya-routes.js +507 -0
- package/server/mcp-tool-defs.js +940 -0
- package/server/mcp-tool-policy.js +34 -2
- package/server/model-tiers.js +22 -5
- package/server/pipeline-routes.js +21 -11
- package/server/project-root.js +58 -19
- package/server/remote-utils.js +3 -1
- package/server/schema-validator.js +7 -1
- package/server/session-manager.js +5 -6
- package/server/session-routes.js +3 -1
- package/server/stratum-client.js +57 -10
- package/server/stratum-sync.js +6 -3
- package/server/summarizer.js +3 -4
- package/server/supervisor.js +0 -1
- package/server/vision-routes.js +208 -98
- package/server/vision-server.js +86 -23
- package/server/vision-store.js +60 -6
- package/server/vision-utils.js +3 -4
- package/server/workspace-activity.js +18 -0
- package/server/workspace-middleware.js +2 -2
- package/server/workspace-runtime.js +243 -0
- package/server/worktree-gc.js +1 -0
- package/dist/assets/App-PkZzHeMj.js +0 -894
- package/dist/assets/channel-qVK_qn4E.js +0 -1
- package/dist/assets/classDiagram-6PBFFD2Q-B8UcfC1q.js +0 -1
- package/dist/assets/classDiagram-v2-HSJHXN6E-B8UcfC1q.js +0 -1
- package/dist/assets/clone-Pu3RyLUh.js +0 -1
- package/dist/assets/index-LIwREYgH.css +0 -1
- package/dist/assets/mobile-BnXEOE3U.js +0 -17
- package/dist/assets/stateDiagram-v2-QKLJ7IA2-BqKuX4rj.js +0 -1
- package/lib/staleness.js +0 -87
- package/server/ideabox-cache.js +0 -77
package/lib/build.js
CHANGED
|
@@ -16,8 +16,12 @@ import { createHash, randomUUID } from 'node:crypto';
|
|
|
16
16
|
|
|
17
17
|
import { StratumMcpClient, StratumError, resolvePlanSpecValues, resolveStepProfile } from './stratum-mcp-client.js';
|
|
18
18
|
import { resolveStratumMcpConnection } from './stratum-engine.js';
|
|
19
|
-
import { runAndNormalize, AgentTimeoutError, AgentAbortedError, UserInterruptError, AgentError } from './result-normalizer.js';
|
|
19
|
+
import { runAndNormalize, mergeUsage, AgentTimeoutError, AgentAbortedError, UserInterruptError, AgentError } from './result-normalizer.js';
|
|
20
20
|
import { checkCapabilityViolation } from './capability-checker.js';
|
|
21
|
+
import { getCatalog as getPolicyCatalog, getPolicyCheckConfig } from './policy-catalog.js';
|
|
22
|
+
import {
|
|
23
|
+
resolveBuildUserMode, scanResponse, toViolationStrings, buildRevisionNotice, attachPolicyCount,
|
|
24
|
+
} from './policy-check.js';
|
|
21
25
|
import { preflightCodexWorktreeProbe, codexProbeAbortMessage } from './codex-preflight.js';
|
|
22
26
|
import { buildStepPrompt, buildGateContext, clearAmbientContextCache } from './step-prompt.js';
|
|
23
27
|
import { promptGate } from './gate-prompt.js';
|
|
@@ -33,11 +37,12 @@ import { resolveAgentConfig, parseAgentString } from './agent-string.js';
|
|
|
33
37
|
import { emitSections as emitPlanSections, appendTrailers as appendSectionTrailers, analyzeRollup, writeRollup } from './sections.js';
|
|
34
38
|
import { SECTIONS_DIR } from './constants.js';
|
|
35
39
|
import { rtkPrefix } from './rtk.js';
|
|
40
|
+
import { tsCompatibilityOf, quarantineMessage, INIT_PROVISIONED_SPECS } from './pipeline-compat.js';
|
|
36
41
|
|
|
37
42
|
import YAML from 'yaml';
|
|
38
43
|
// feature-json direct imports removed — mutations now go through TrackerProvider (T9)
|
|
39
44
|
import { loadFeaturesDir, resolveContextPath, resolveRoadmapPath, resolveFeaturesPath } from './project-paths.js';
|
|
40
|
-
import { getMode } from './lifecycle-modes.js';
|
|
45
|
+
import { getMode, resolveMode } from './lifecycle-modes.js';
|
|
41
46
|
import { vocabularyEnabled, tagVocabularyViolations, VOCABULARY_FILE } from './vocabulary-inject.js';
|
|
42
47
|
import { vocabularyCompliance } from './vocabulary-compliance.js';
|
|
43
48
|
|
|
@@ -54,7 +59,7 @@ import { applyFrontTriage, maybeEscalateLane } from './lane-gate.js';
|
|
|
54
59
|
import { LENS_DEFINITIONS } from './review-lenses.js';
|
|
55
60
|
import { injectCertInstructions } from './cert-inject.js';
|
|
56
61
|
import { buildReviewPrompt } from './review-prompt.js';
|
|
57
|
-
import { detectTestFramework, scaffoldTestFramework, parseTestSummary, deriveTestsPass, isTestFile } from './test-bootstrap.js';
|
|
62
|
+
import { detectTestFramework, scaffoldTestFramework, parseTestSummary, deriveTestsPass, deriveTestsAttested, isTestFile } from './test-bootstrap.js';
|
|
58
63
|
import { classifyStepAsTier, evaluateTiers } from './gate-tiers.js';
|
|
59
64
|
import { mapFilesToRoutes, classifyRoutes, isDocsOnlyDiff } from './qa-scoping.js';
|
|
60
65
|
import { computeCompositeScore } from './health-score.js';
|
|
@@ -76,6 +81,7 @@ import {
|
|
|
76
81
|
verifyConsumerRunRevision,
|
|
77
82
|
} from './consumer-fanout.js';
|
|
78
83
|
import { appendEvent as appendDispatchEvent, readEvents as readDispatchEvents } from './dispatch-ledger.js';
|
|
84
|
+
import { appendEvent as appendFeatureEvent } from './feature-events.js';
|
|
79
85
|
|
|
80
86
|
// ---------------------------------------------------------------------------
|
|
81
87
|
// COMP-ROADMAP-PLAN S8: gate the `ship` interception by mode.
|
|
@@ -97,6 +103,85 @@ export function shouldInterceptShip(stepId, mode) {
|
|
|
97
103
|
return stepId === 'ship' && mode !== 'plan';
|
|
98
104
|
}
|
|
99
105
|
|
|
106
|
+
// ---------------------------------------------------------------------------
|
|
107
|
+
// COMP-POLICY-CHECK: pre-response policy check (adherence enforcement).
|
|
108
|
+
// ---------------------------------------------------------------------------
|
|
109
|
+
|
|
110
|
+
/**
|
|
111
|
+
* Does this step declare a gate? A gate step is SKILL_GATED by construction —
|
|
112
|
+
* asking the user for a decision is the point of the step, so policy matches on
|
|
113
|
+
* its response are suppressed rather than flagged.
|
|
114
|
+
*
|
|
115
|
+
* @param {object} spec the local pipeline spec
|
|
116
|
+
* @param {string} flowName active flow
|
|
117
|
+
* @param {string} stepId ready-step id (scoped ids resolve to their bare tail)
|
|
118
|
+
* @returns {boolean}
|
|
119
|
+
*/
|
|
120
|
+
export function isGateStep(spec, flowName, stepId) {
|
|
121
|
+
const bare = String(stepId ?? '').split('/').pop();
|
|
122
|
+
const steps = spec?.flows?.[flowName]?.steps;
|
|
123
|
+
if (!Array.isArray(steps)) return false;
|
|
124
|
+
return steps.some(st => st?.id === bare && !!st.gate);
|
|
125
|
+
}
|
|
126
|
+
|
|
127
|
+
/**
|
|
128
|
+
* COMP-POLICY-CHECK-2/3: scan one step response against the local catalog.
|
|
129
|
+
* Total — a broken catalog or scan degrades to "no findings" with a WARNING and
|
|
130
|
+
* never fails the step.
|
|
131
|
+
*
|
|
132
|
+
* The user mode is EXPLICIT here, never inferred: a build has no user turns to
|
|
133
|
+
* classify (see `resolveBuildUserMode`). Config override → gate step → default.
|
|
134
|
+
*
|
|
135
|
+
* @param {{cwd: string, text: string, skillGated: boolean}} args
|
|
136
|
+
* @returns {{records: object[], violations: string[], userMode: string}}
|
|
137
|
+
*/
|
|
138
|
+
export function policyScanForStep({ cwd, text, skillGated }) {
|
|
139
|
+
const empty = { records: [], violations: [], userMode: 'AUTONOMOUS' };
|
|
140
|
+
try {
|
|
141
|
+
const config = getPolicyCheckConfig(cwd);
|
|
142
|
+
const catalog = getPolicyCatalog(cwd);
|
|
143
|
+
if (catalog.length === 0) return empty;
|
|
144
|
+
const userMode = resolveBuildUserMode(config.userMode, { skillGated });
|
|
145
|
+
const records = scanResponse(text ?? '', catalog, userMode);
|
|
146
|
+
return { records, violations: toViolationStrings(records), userMode };
|
|
147
|
+
} catch (err) {
|
|
148
|
+
// eslint-disable-next-line no-console
|
|
149
|
+
console.warn(`[policy-check] scan skipped: ${err.message}`);
|
|
150
|
+
return empty;
|
|
151
|
+
}
|
|
152
|
+
}
|
|
153
|
+
|
|
154
|
+
/**
|
|
155
|
+
* COMP-POLICY-CHECK-5: trace every match (flagged AND suppressed) to the
|
|
156
|
+
* append-only feature-events bus — which syncs into SmartMemory, closing the
|
|
157
|
+
* measurement loop — plus the build stream for live cockpit visibility.
|
|
158
|
+
*
|
|
159
|
+
* @param {object} args
|
|
160
|
+
* @param {string} args.pass 'initial' | 'policy_revision'
|
|
161
|
+
*/
|
|
162
|
+
export function recordPolicyScan({ cwd, streamWriter, stepId, records, userMode, featureCode, buildId, pass = 'initial' }) {
|
|
163
|
+
for (const record of records ?? []) {
|
|
164
|
+
try {
|
|
165
|
+
appendFeatureEvent(cwd, {
|
|
166
|
+
tool: 'policy_check',
|
|
167
|
+
build_id: buildId ?? null,
|
|
168
|
+
step_id: stepId,
|
|
169
|
+
rule: record.rule,
|
|
170
|
+
matched: record.matched,
|
|
171
|
+
suppressed: record.suppressed,
|
|
172
|
+
user_mode: userMode,
|
|
173
|
+
pass,
|
|
174
|
+
});
|
|
175
|
+
} catch (err) {
|
|
176
|
+
// eslint-disable-next-line no-console
|
|
177
|
+
console.warn(`[policy-check] trace append failed: ${err.message}`);
|
|
178
|
+
}
|
|
179
|
+
try {
|
|
180
|
+
streamWriter?.writePolicyViolation(stepId, record, userMode, featureCode ?? null, buildId ?? null);
|
|
181
|
+
} catch { /* stream emit is best-effort */ }
|
|
182
|
+
}
|
|
183
|
+
}
|
|
184
|
+
|
|
100
185
|
// ---------------------------------------------------------------------------
|
|
101
186
|
// COMP-ROADMAP-PLAN S5: ratify a plan-authored design instead of clobbering it.
|
|
102
187
|
// ---------------------------------------------------------------------------
|
|
@@ -445,6 +530,39 @@ export function deriveOrdinaryReviewScaffold({ contractName = null, stepId = '',
|
|
|
445
530
|
return { isReviewMain, isReduceMain, isReviewScaffoldMain: isReviewMain && !isReduceMain };
|
|
446
531
|
}
|
|
447
532
|
|
|
533
|
+
// COMP-AGENT-LANES: one lane per parallel worker slot. Identity is
|
|
534
|
+
// flowId:stepId:itemIndex (stepId/itemIndex RECUR across builds, so flowId is
|
|
535
|
+
// load-bearing); version is the ordered tuple (generation, attempt) — the UI
|
|
536
|
+
// resets a lane on a higher version and rejects lower (stale) events. The
|
|
537
|
+
// label is the human mandate: the review lens id when the item is a review,
|
|
538
|
+
// else the step intent truncated.
|
|
539
|
+
const LANE_LABEL_MAX = 80;
|
|
540
|
+
|
|
541
|
+
export function buildLaneEnvelope(descriptor, flowId, { lens = null } = {}) {
|
|
542
|
+
const rawLabel = (typeof lens === 'string' && lens)
|
|
543
|
+
|| (typeof descriptor?.do === 'string' && descriptor.do)
|
|
544
|
+
|| String(descriptor?.id ?? '');
|
|
545
|
+
const label = rawLabel.length > LANE_LABEL_MAX
|
|
546
|
+
? `${rawLabel.slice(0, LANE_LABEL_MAX - 1)}…`
|
|
547
|
+
: rawLabel;
|
|
548
|
+
return {
|
|
549
|
+
flowId,
|
|
550
|
+
stepId: descriptor.id,
|
|
551
|
+
itemIndex: descriptor.itemIndex,
|
|
552
|
+
generation: descriptor.generation ?? 0,
|
|
553
|
+
attempt: descriptor.attempt ?? 1,
|
|
554
|
+
label,
|
|
555
|
+
agent: descriptor.agent ?? 'claude',
|
|
556
|
+
};
|
|
557
|
+
}
|
|
558
|
+
|
|
559
|
+
function deriveConsumerLane(descriptor, flowId) {
|
|
560
|
+
const reviewOpts = deriveConsumerReviewOptions(descriptor);
|
|
561
|
+
return buildLaneEnvelope(descriptor, flowId, {
|
|
562
|
+
lens: reviewOpts.reviewMode ? reviewOpts.lens : null,
|
|
563
|
+
});
|
|
564
|
+
}
|
|
565
|
+
|
|
448
566
|
export function deriveConsumerReviewOptions(descriptor) {
|
|
449
567
|
const reviewMode = descriptor?.contract?.root === 'ReviewResult';
|
|
450
568
|
const item = (descriptor?.item && typeof descriptor.item === 'object') ? descriptor.item : {};
|
|
@@ -620,6 +738,7 @@ async function reportConsumerStepDone({
|
|
|
620
738
|
itemIndex: descriptor.itemIndex,
|
|
621
739
|
stage: descriptor.stage,
|
|
622
740
|
generation: descriptor.generation,
|
|
741
|
+
lane: deriveConsumerLane(descriptor, flowId),
|
|
623
742
|
});
|
|
624
743
|
return { response, skipped: true };
|
|
625
744
|
}
|
|
@@ -786,6 +905,12 @@ export async function runConsumerIssuance({
|
|
|
786
905
|
// contract the python parallel-dispatch path emitted — otherwise the fanout runs
|
|
787
906
|
// invisibly and the parallel progress bar never appears.
|
|
788
907
|
const parallelStepNum = `∥${descriptor.itemIndex}`;
|
|
908
|
+
// COMP-AGENT-LANES: the same lane envelope rides every lifecycle write for
|
|
909
|
+
// this item AND (via runAndNormalize opts) every relayed output write, so the
|
|
910
|
+
// cockpit can attribute each event to its worker slot.
|
|
911
|
+
const lane = buildLaneEnvelope(descriptor, flowId, {
|
|
912
|
+
lens: reviewOpts.reviewMode ? reviewOpts.lens : null,
|
|
913
|
+
});
|
|
789
914
|
progress.stepStart(parallelStepNum, '?', descriptor.id);
|
|
790
915
|
streamWriter.write({
|
|
791
916
|
type: 'build_step_start',
|
|
@@ -800,6 +925,7 @@ export async function runConsumerIssuance({
|
|
|
800
925
|
itemIndex: descriptor.itemIndex,
|
|
801
926
|
stage: descriptor.stage,
|
|
802
927
|
generation: descriptor.generation,
|
|
928
|
+
lane,
|
|
803
929
|
});
|
|
804
930
|
|
|
805
931
|
let mainResult;
|
|
@@ -809,7 +935,9 @@ export async function runConsumerIssuance({
|
|
|
809
935
|
streamWriter,
|
|
810
936
|
maxDurationMs,
|
|
811
937
|
stratum,
|
|
938
|
+
lane,
|
|
812
939
|
cwd: recovery.worktree,
|
|
940
|
+
sandboxMode: descriptor.policy?.isolation === 'worktree' ? 'workspace-write' : 'read-only',
|
|
813
941
|
onAgentEvent,
|
|
814
942
|
profile,
|
|
815
943
|
reviewMode: reviewOpts.reviewMode,
|
|
@@ -823,21 +951,24 @@ export async function runConsumerIssuance({
|
|
|
823
951
|
step_id: descriptor.id,
|
|
824
952
|
...(typeof descriptor.attempt === 'number' ? { attempt: descriptor.attempt } : {}),
|
|
825
953
|
},
|
|
826
|
-
//
|
|
827
|
-
//
|
|
828
|
-
// its read-only tool restrictions actually BIND (the engine's sync
|
|
829
|
-
// agent_run can't carry claude allowlists and its sandboxMode binds only
|
|
830
|
-
// codex) and its per-item timeout / stuck abort truly INTERRUPTS it. Write
|
|
831
|
-
// (worktree) items keep the engine seam: their per-item timeout still fails
|
|
832
|
-
// the item, but interrupting an in-flight workspace-write run needs a
|
|
833
|
-
// stratum follow-up (background mode is codex+read-only-only). Codex review
|
|
834
|
-
// items also fall back to the sync seam (compose has no codex SDK).
|
|
954
|
+
// Local Claude owns its SDK process group for review fanout and drains
|
|
955
|
+
// graceful teardown before timeout/interrupt returns, as MCP does.
|
|
835
956
|
localExecution: descriptor.policy?.isolation === 'none',
|
|
836
957
|
});
|
|
837
958
|
} catch (error) {
|
|
838
|
-
|
|
839
|
-
// failures
|
|
840
|
-
|
|
959
|
+
const failedUsage = failureUsageFields(error);
|
|
960
|
+
// Control failures do not settle the item, so record known dispatch usage
|
|
961
|
+
// before aborting the pump. Never retry work with uncertain termination.
|
|
962
|
+
if (error instanceof UserInterruptError || ['INJECTED_CONSUMER_CRASH', 'CANCELLATION_UNCONFIRMED', 'CANCELLATION_TEARDOWN_TIMEOUT'].includes(error?.code)) {
|
|
963
|
+
if (failedUsage.usage && typeof context?.onUsage === 'function') {
|
|
964
|
+
try {
|
|
965
|
+
await context.onUsage(usagePayload(failedUsage.usage, failedUsage.usages), {
|
|
966
|
+
dispatchId: error.dispatchId, stepId: descriptor.step ?? descriptor.id, source: 'consumer',
|
|
967
|
+
});
|
|
968
|
+
} catch (usageError) {
|
|
969
|
+
console.warn(`[consumer] Could not record cancelled usage: ${usageError?.message ?? usageError}`);
|
|
970
|
+
}
|
|
971
|
+
}
|
|
841
972
|
throw error;
|
|
842
973
|
}
|
|
843
974
|
// D3: a stuck verdict halts the whole GSD run (not a per-item retry) — the
|
|
@@ -846,8 +977,12 @@ export async function runConsumerIssuance({
|
|
|
846
977
|
// G3: no step_done envelope is sent on the stuck/abort path (the run halts),
|
|
847
978
|
// so the billable usage the aborted run consumed would be lost. Record it
|
|
848
979
|
// into compose's cumulative ledger before converting to the stuck signal.
|
|
849
|
-
if (
|
|
850
|
-
context.onUsage(
|
|
980
|
+
if (failedUsage.usage && typeof context?.onUsage === 'function') {
|
|
981
|
+
await context.onUsage(usagePayload(failedUsage.usage, failedUsage.usages), {
|
|
982
|
+
dispatchId: error.dispatchId,
|
|
983
|
+
stepId: descriptor.step ?? descriptor.id,
|
|
984
|
+
source: 'consumer',
|
|
985
|
+
});
|
|
851
986
|
}
|
|
852
987
|
throw new ConsumerStuckError(stuckTaskId, error.reason);
|
|
853
988
|
}
|
|
@@ -859,7 +994,7 @@ export async function runConsumerIssuance({
|
|
|
859
994
|
settlementFailureClass: 'agent',
|
|
860
995
|
// G3: a timed-out run still consumed billable usage — forward it so the
|
|
861
996
|
// failure envelope debits the engine ledger (same mechanism as F3).
|
|
862
|
-
...
|
|
997
|
+
...failedUsage,
|
|
863
998
|
};
|
|
864
999
|
} else {
|
|
865
1000
|
// A non-timeout agent/connector error must fail ONLY this item, not abort
|
|
@@ -877,7 +1012,7 @@ export async function runConsumerIssuance({
|
|
|
877
1012
|
// it to the error. Forward it so the failure envelope (and compose's
|
|
878
1013
|
// cumulative ledger) debit the attempt instead of letting failures evade
|
|
879
1014
|
// budget exhaustion.
|
|
880
|
-
...
|
|
1015
|
+
...failedUsage,
|
|
881
1016
|
};
|
|
882
1017
|
}
|
|
883
1018
|
}
|
|
@@ -886,7 +1021,11 @@ export async function runConsumerIssuance({
|
|
|
886
1021
|
// D2(b): forward the item's agent usage so GSD can debit the cumulative
|
|
887
1022
|
// budget ledger. Build mode passes no onUsage sink → byte-identical no-op.
|
|
888
1023
|
if (typeof context?.onUsage === 'function' && mainResult?.usage) {
|
|
889
|
-
context.onUsage(mainResult.usage,
|
|
1024
|
+
await context.onUsage(usagePayload(mainResult.usage, mainResult.usages), {
|
|
1025
|
+
dispatchId: mainResult.dispatchIds?.primary,
|
|
1026
|
+
stepId: descriptor.step ?? descriptor.id,
|
|
1027
|
+
source: 'fanout',
|
|
1028
|
+
});
|
|
890
1029
|
}
|
|
891
1030
|
const finalStage = isFinalConsumerStage(localSpec, descriptor);
|
|
892
1031
|
let localFailure = normalizationFailure
|
|
@@ -917,7 +1056,7 @@ export async function runConsumerIssuance({
|
|
|
917
1056
|
// succeeded, so usage rides both the success and failure envelope. Compose's
|
|
918
1057
|
// cumulative ledger (context.onUsage) is separate, compose-side accounting.
|
|
919
1058
|
const engineUsage = toEngineUsage(mainResult?.usage);
|
|
920
|
-
if (engineUsage) envelope.usage = engineUsage;
|
|
1059
|
+
if (engineUsage && !context?.receiptsMode) envelope.usage = engineUsage;
|
|
921
1060
|
|
|
922
1061
|
if (typeof artifacts.hooks.afterAgentMutationBeforePrepared === 'function') {
|
|
923
1062
|
await artifacts.hooks.afterAgentMutationBeforePrepared({
|
|
@@ -981,6 +1120,14 @@ export async function runConsumerIssuance({
|
|
|
981
1120
|
// H6: matches the item's start stepId so the UI decrements the same task
|
|
982
1121
|
// (AgentStream keys the per-task done on parallel:true + a known stepId).
|
|
983
1122
|
parallel: true,
|
|
1123
|
+
// COMP-AGENT-LANES (C4): terminal status is explicit at source — the UI
|
|
1124
|
+
// must not infer "complete" from the done event's existence.
|
|
1125
|
+
status: localFailure ? 'failed' : 'succeeded',
|
|
1126
|
+
outcome: result?.outcome ?? (localFailure ? 'failed' : 'succeeded'),
|
|
1127
|
+
itemIndex: descriptor.itemIndex,
|
|
1128
|
+
stage: descriptor.stage,
|
|
1129
|
+
generation: descriptor.generation,
|
|
1130
|
+
lane,
|
|
984
1131
|
});
|
|
985
1132
|
return response;
|
|
986
1133
|
}
|
|
@@ -1066,6 +1213,91 @@ export function toEngineUsage(usage) {
|
|
|
1066
1213
|
return Object.keys(out).length > 0 ? out : null;
|
|
1067
1214
|
}
|
|
1068
1215
|
|
|
1216
|
+
// A repair failure can carry two dispatch records. Keep those records for
|
|
1217
|
+
// receipts and aggregate both for the legacy step_done budget envelope.
|
|
1218
|
+
function failureUsageFields(error) {
|
|
1219
|
+
const usages = Array.isArray(error?.usages) ? error.usages : null;
|
|
1220
|
+
if (!usages?.length) return error?.usage ? { usage: error.usage } : {};
|
|
1221
|
+
if (usages.length === 1 && error?.usage) return { usage: error.usage, usages };
|
|
1222
|
+
const usage = {
|
|
1223
|
+
input_tokens: 0, output_tokens: 0, cache_creation_input_tokens: 0,
|
|
1224
|
+
cache_read_input_tokens: 0, cost_usd: 0, duration_ms: 0, model: null,
|
|
1225
|
+
};
|
|
1226
|
+
for (const entry of usages) {
|
|
1227
|
+
if (!entry || typeof entry !== 'object') continue;
|
|
1228
|
+
usage.input_tokens += entry.input_tokens ?? 0;
|
|
1229
|
+
usage.output_tokens += entry.output_tokens ?? 0;
|
|
1230
|
+
usage.cache_creation_input_tokens += entry.cache_creation ?? entry.cache_creation_input_tokens ?? 0;
|
|
1231
|
+
usage.cache_read_input_tokens += entry.cache_read ?? entry.cache_read_input_tokens ?? 0;
|
|
1232
|
+
usage.cost_usd += entry.cost_usd ?? 0;
|
|
1233
|
+
usage.duration_ms += entry.duration_ms ?? 0;
|
|
1234
|
+
usage.model = entry.model ?? usage.model;
|
|
1235
|
+
}
|
|
1236
|
+
return { usage, usages };
|
|
1237
|
+
}
|
|
1238
|
+
|
|
1239
|
+
function usagePayload(usage, usages) {
|
|
1240
|
+
if (!usage || typeof usage !== 'object') return usage;
|
|
1241
|
+
return Array.isArray(usages) ? { ...usage, usages } : usage;
|
|
1242
|
+
}
|
|
1243
|
+
|
|
1244
|
+
/** Send one surface-15 receipt per underlying model dispatch. */
|
|
1245
|
+
export async function reportUsageReceipts(context, usage, meta = {}) {
|
|
1246
|
+
if (!context?.receiptsMode || !context.flowId || typeof context.stratum?.usageReport !== 'function') {
|
|
1247
|
+
return [];
|
|
1248
|
+
}
|
|
1249
|
+
const entries = Array.isArray(usage)
|
|
1250
|
+
? usage
|
|
1251
|
+
: (Array.isArray(usage?.usages) ? usage.usages : (usage ? [usage] : []));
|
|
1252
|
+
const responses = [];
|
|
1253
|
+
for (const entry of entries) {
|
|
1254
|
+
if (!entry || typeof entry !== 'object') continue;
|
|
1255
|
+
const engineUsage = toEngineUsage(entry);
|
|
1256
|
+
if (!engineUsage) continue;
|
|
1257
|
+
// Surface 15 requires explicit USD provenance. Normalized UsageRecords carry
|
|
1258
|
+
// `usd_source`; raw engine usage ({tokens, usd, ms}) does not. Preserve raw
|
|
1259
|
+
// token/time usage, but fail closed on an unlabelled dollar value instead of
|
|
1260
|
+
// manufacturing "reported" provenance.
|
|
1261
|
+
const usdSource = ['reported', 'estimated'].includes(entry.usd_source)
|
|
1262
|
+
? entry.usd_source
|
|
1263
|
+
: null;
|
|
1264
|
+
if (Object.hasOwn(engineUsage, 'usd') && !usdSource) delete engineUsage.usd;
|
|
1265
|
+
if (Object.keys(engineUsage).length === 0) continue;
|
|
1266
|
+
const input = entry.input_tokens;
|
|
1267
|
+
const output = entry.output_tokens;
|
|
1268
|
+
const receipt = {
|
|
1269
|
+
dispatchId: entry.dispatch_id ?? meta.dispatchId ?? randomUUID(),
|
|
1270
|
+
...(meta.stepId ? { stepId: meta.stepId } : {}),
|
|
1271
|
+
source: meta.source ?? 'main',
|
|
1272
|
+
usage: engineUsage,
|
|
1273
|
+
telemetry: {
|
|
1274
|
+
model: typeof entry.model === 'string' && entry.model.length > 0 ? entry.model : 'unknown',
|
|
1275
|
+
...(typeof entry.effort === 'string' && entry.effort.length > 0 ? { effort: entry.effort } : {}),
|
|
1276
|
+
durationMs: entry.duration_ms ?? entry.ms ?? 0,
|
|
1277
|
+
},
|
|
1278
|
+
...(typeof input === 'number' || typeof output === 'number'
|
|
1279
|
+
? { split: {
|
|
1280
|
+
input: input ?? 0,
|
|
1281
|
+
output: output ?? 0,
|
|
1282
|
+
...(typeof (entry.cache_read ?? entry.cache_read_input_tokens) === 'number'
|
|
1283
|
+
? { cacheRead: entry.cache_read ?? entry.cache_read_input_tokens }
|
|
1284
|
+
: {}),
|
|
1285
|
+
...(typeof (entry.cache_creation ?? entry.cache_creation_input_tokens) === 'number'
|
|
1286
|
+
? { cacheCreation: entry.cache_creation ?? entry.cache_creation_input_tokens }
|
|
1287
|
+
: {}),
|
|
1288
|
+
} }
|
|
1289
|
+
: {}),
|
|
1290
|
+
...(Object.hasOwn(engineUsage, 'usd') ? { usdSource } : {}),
|
|
1291
|
+
};
|
|
1292
|
+
try {
|
|
1293
|
+
responses.push(await context.stratum.usageReport(context.flowId, receipt));
|
|
1294
|
+
} catch (error) {
|
|
1295
|
+
console.warn(`[usage-receipt] failed for ${receipt.dispatchId}: ${error?.message ?? error}`);
|
|
1296
|
+
}
|
|
1297
|
+
}
|
|
1298
|
+
return responses;
|
|
1299
|
+
}
|
|
1300
|
+
|
|
1069
1301
|
/**
|
|
1070
1302
|
* F5: deterministic v1 vocabulary enforcement, evaluated compose-side at the
|
|
1071
1303
|
* review_merge step (the step the now-dropped judged ensure was attached to). The
|
|
@@ -1244,7 +1476,20 @@ function writeActiveBuild(dataDir, state) {
|
|
|
1244
1476
|
renameSync(tmp, target);
|
|
1245
1477
|
}
|
|
1246
1478
|
|
|
1247
|
-
|
|
1479
|
+
// COMP-COMPLETION-GATE slice 2: v2 adds the completion evidence the gate needs
|
|
1480
|
+
// at terminalization — `tests_attested` (tri-state) and `evidence_root`.
|
|
1481
|
+
//
|
|
1482
|
+
// `test_count`/`pass_rate` were already here but are metrics, not attestation:
|
|
1483
|
+
// they are only populated when the output PARSED, so their absence is ambiguous
|
|
1484
|
+
// between "no tests" and "could not read the output". The gate cannot act on an
|
|
1485
|
+
// ambiguous signal, hence an explicit tri-state.
|
|
1486
|
+
//
|
|
1487
|
+
// `evidence_root` is persisted because a cross-repo build runs git and tests in
|
|
1488
|
+
// the agent's tree while feature metadata lives in the project tree — and
|
|
1489
|
+
// `runBuild` reconstructs that root from the CURRENT invocation, so a resumed
|
|
1490
|
+
// cross-repo build would otherwise fall back to the project root and verify the
|
|
1491
|
+
// wrong repository's HEAD.
|
|
1492
|
+
const BUILD_ACCUMULATOR_VERSION = 2;
|
|
1248
1493
|
const BUILD_ACCUMULATOR_FIELDS = new Set([
|
|
1249
1494
|
'v',
|
|
1250
1495
|
'build_id',
|
|
@@ -1256,9 +1501,13 @@ const BUILD_ACCUMULATOR_FIELDS = new Set([
|
|
|
1256
1501
|
'ship_files_changed',
|
|
1257
1502
|
'test_count',
|
|
1258
1503
|
'pass_rate',
|
|
1504
|
+
'tests_attested',
|
|
1505
|
+
'evidence_root',
|
|
1259
1506
|
'tokens_total',
|
|
1260
1507
|
'usd',
|
|
1261
1508
|
]);
|
|
1509
|
+
/** The only values `tests_attested` may hold. See deriveTestsAttested. */
|
|
1510
|
+
const TESTS_ATTESTED_VALUES = new Set(['passed', 'failed', 'no-signal']);
|
|
1262
1511
|
const UUID_RE = /^[0-9a-f]{8}-[0-9a-f]{4}-[1-8][0-9a-f]{3}-[89ab][0-9a-f]{3}-[0-9a-f]{12}$/i;
|
|
1263
1512
|
|
|
1264
1513
|
function assertFeatureCodeForAccumulator(featureCode) {
|
|
@@ -1326,9 +1575,35 @@ function validateBuildAccumulator(value, expectedFeatureCode = null) {
|
|
|
1326
1575
|
throw new Error(`Build accumulator is corrupt: ${key} must be a non-negative finite number`);
|
|
1327
1576
|
}
|
|
1328
1577
|
}
|
|
1578
|
+
if (!TESTS_ATTESTED_VALUES.has(value.tests_attested)) {
|
|
1579
|
+
throw new Error(
|
|
1580
|
+
`Build accumulator is corrupt: tests_attested must be one of ${[...TESTS_ATTESTED_VALUES].join('|')}`,
|
|
1581
|
+
);
|
|
1582
|
+
}
|
|
1583
|
+
if (value.evidence_root !== null && typeof value.evidence_root !== 'string') {
|
|
1584
|
+
throw new Error('Build accumulator is corrupt: evidence_root must be null or a string');
|
|
1585
|
+
}
|
|
1329
1586
|
return value;
|
|
1330
1587
|
}
|
|
1331
1588
|
|
|
1589
|
+
/**
|
|
1590
|
+
* Bring a v1 accumulator forward. A v1 record predates completion evidence, so
|
|
1591
|
+
* it cannot say anything about whether tests were attested — and the honest value
|
|
1592
|
+
* for "we do not know" is `no-signal`, which the completion gate REFUSES. A build
|
|
1593
|
+
* resumed across this upgrade therefore has to re-attest rather than inheriting a
|
|
1594
|
+
* pass it never recorded. That is the intended direction: absence of signal is
|
|
1595
|
+
* never attestation.
|
|
1596
|
+
*/
|
|
1597
|
+
function migrateBuildAccumulator(value) {
|
|
1598
|
+
if (!value || typeof value !== 'object' || value.v !== 1) return value;
|
|
1599
|
+
return {
|
|
1600
|
+
...value,
|
|
1601
|
+
v: BUILD_ACCUMULATOR_VERSION,
|
|
1602
|
+
tests_attested: 'no-signal',
|
|
1603
|
+
evidence_root: null,
|
|
1604
|
+
};
|
|
1605
|
+
}
|
|
1606
|
+
|
|
1332
1607
|
export function readBuildAccumulator(projectCwd, featureCode) {
|
|
1333
1608
|
const path = buildAccumulatorPath(projectCwd, featureCode);
|
|
1334
1609
|
if (!existsSync(path)) return null;
|
|
@@ -1338,7 +1613,7 @@ export function readBuildAccumulator(projectCwd, featureCode) {
|
|
|
1338
1613
|
} catch (error) {
|
|
1339
1614
|
throw new Error(`Build accumulator is corrupt at ${path}: ${error.message}`);
|
|
1340
1615
|
}
|
|
1341
|
-
return validateBuildAccumulator(parsed, featureCode);
|
|
1616
|
+
return validateBuildAccumulator(migrateBuildAccumulator(parsed), featureCode);
|
|
1342
1617
|
}
|
|
1343
1618
|
|
|
1344
1619
|
export function writeBuildAccumulator(projectCwd, accumulator) {
|
|
@@ -1368,6 +1643,8 @@ export function newBuildAccumulatorRecord(featureCode) {
|
|
|
1368
1643
|
ship_files_changed: null,
|
|
1369
1644
|
test_count: null,
|
|
1370
1645
|
pass_rate: null,
|
|
1646
|
+
tests_attested: 'no-signal',
|
|
1647
|
+
evidence_root: null,
|
|
1371
1648
|
tokens_total: 0,
|
|
1372
1649
|
usd: 0,
|
|
1373
1650
|
};
|
|
@@ -1473,7 +1750,11 @@ export function settleDispatches(projectCwd, buildId, stepId, {
|
|
|
1473
1750
|
if (gsd) return [];
|
|
1474
1751
|
const primary = dispatchIds?.primary;
|
|
1475
1752
|
const repair = dispatchIds?.repair;
|
|
1476
|
-
|
|
1753
|
+
// COMP-POLICY-CHECK-4: a policy revision is a second dispatch whose output
|
|
1754
|
+
// replaced the primary's. It settles on the same verdict as the run it
|
|
1755
|
+
// replaced — same path, one more id, no parallel settlement loop.
|
|
1756
|
+
const revision = dispatchIds?.revision;
|
|
1757
|
+
if (!primary && !repair && !revision) return [];
|
|
1477
1758
|
const rows = [];
|
|
1478
1759
|
const appendSettlement = (dispatchId, isAccepted, rejectedClass) => {
|
|
1479
1760
|
if (typeof dispatchId !== 'string' || dispatchId.length === 0) return;
|
|
@@ -1501,6 +1782,14 @@ export function settleDispatches(projectCwd, buildId, stepId, {
|
|
|
1501
1782
|
isEnsureRetry ? 'ensure-retry' : failureClass,
|
|
1502
1783
|
);
|
|
1503
1784
|
}
|
|
1785
|
+
|
|
1786
|
+
if (revision) {
|
|
1787
|
+
appendSettlement(
|
|
1788
|
+
revision,
|
|
1789
|
+
isEnsureRetry ? false : accepted === true,
|
|
1790
|
+
isEnsureRetry ? 'ensure-retry' : failureClass,
|
|
1791
|
+
);
|
|
1792
|
+
}
|
|
1504
1793
|
return rows;
|
|
1505
1794
|
}
|
|
1506
1795
|
|
|
@@ -1668,10 +1957,12 @@ function isProcessAlive(pid) {
|
|
|
1668
1957
|
* @param {object} gateDispatch - Stratum gate dispatch (step_id, on_approve, on_revise, on_kill)
|
|
1669
1958
|
* @param {object} [gateExtras] - Optional enrichment (fromPhase, toPhase, summary)
|
|
1670
1959
|
*/
|
|
1671
|
-
function makeAskAgent(stratum, context, gateDispatch, gateExtras) {
|
|
1960
|
+
export function makeAskAgent(stratum, context, gateDispatch, gateExtras) {
|
|
1672
1961
|
const preamble = buildGateContext(gateDispatch, context, gateExtras);
|
|
1962
|
+
let budgetExhausted = false;
|
|
1673
1963
|
|
|
1674
1964
|
return async function askAgent(question, artifactPath) {
|
|
1965
|
+
if (budgetExhausted) return '(budget exhausted)';
|
|
1675
1966
|
const fileRef = artifactPath && !artifactPath.endsWith('/')
|
|
1676
1967
|
? `Read the file "${artifactPath}" and answer`
|
|
1677
1968
|
: `Look at the project files in the working directory and answer`;
|
|
@@ -1690,6 +1981,15 @@ function makeAskAgent(stratum, context, gateDispatch, gateExtras) {
|
|
|
1690
1981
|
step_id: gateDispatch.step_id ?? gateDispatch.id,
|
|
1691
1982
|
...(typeof gateDispatch.attempt === 'number' ? { attempt: gateDispatch.attempt } : {}),
|
|
1692
1983
|
},
|
|
1984
|
+
onUsage: async (usages) => {
|
|
1985
|
+
const results = await context.recordBuildUsage?.(usages, {
|
|
1986
|
+
stepId: gateDispatch.step_id ?? gateDispatch.id,
|
|
1987
|
+
source: 'gate_qa',
|
|
1988
|
+
});
|
|
1989
|
+
if (results?.some((result) => ['flow_exhausted', 'flow_exhausted_after_terminal'].includes(result?.budget))) {
|
|
1990
|
+
budgetExhausted = true;
|
|
1991
|
+
}
|
|
1992
|
+
},
|
|
1693
1993
|
});
|
|
1694
1994
|
return text || '(no answer)';
|
|
1695
1995
|
};
|
|
@@ -1740,6 +2040,22 @@ export function resolveTemplatePath(name, cwd) {
|
|
|
1740
2040
|
const presetsPath = join(packageDir, '..', 'presets', `${templateName}.stratum.yaml`);
|
|
1741
2041
|
if (existsSync(presetsPath)) return presetsPath;
|
|
1742
2042
|
|
|
2043
|
+
// COMP-PIPELINE-QUARANTINE follow-up: fall back to the BUNDLED pipelines too,
|
|
2044
|
+
// not just presets. `compose init` seeds a curated few specs, so every other
|
|
2045
|
+
// shipped pipeline (content, coverage-sweep, refactor, research, review-fix)
|
|
2046
|
+
// was unreachable from a workspace no matter how it was invoked — the resolver
|
|
2047
|
+
// simply had no path to them. Project-local still wins, so a workspace that
|
|
2048
|
+
// customizes a spec keeps its own copy.
|
|
2049
|
+
//
|
|
2050
|
+
// NOT for the init-provisioned specs: if `build` is missing, the workspace was
|
|
2051
|
+
// never initialized, and answering with our bundled copy would silently run
|
|
2052
|
+
// Compose's own pipeline against an uninitialized project instead of raising
|
|
2053
|
+
// "Lifecycle spec not found".
|
|
2054
|
+
if (!INIT_PROVISIONED_SPECS.includes(templateName)) {
|
|
2055
|
+
const bundledPath = join(packageDir, '..', 'pipelines', `${templateName}.stratum.yaml`);
|
|
2056
|
+
if (existsSync(bundledPath)) return bundledPath;
|
|
2057
|
+
}
|
|
2058
|
+
|
|
1743
2059
|
return projectPath;
|
|
1744
2060
|
}
|
|
1745
2061
|
|
|
@@ -2160,6 +2476,15 @@ export async function runBuild(featureCode, opts = {}) {
|
|
|
2160
2476
|
const stepProfiles = loadPipelineProfiles(specPath);
|
|
2161
2477
|
let specYaml = readFileSync(specPath, 'utf-8');
|
|
2162
2478
|
|
|
2479
|
+
// COMP-PIPELINE-QUARANTINE: refuse a retired-dialect spec HERE, at the one
|
|
2480
|
+
// seam every template passes through (build, fix, plan, --quick, --template,
|
|
2481
|
+
// bundled presets), rather than letting the engine answer with a bare
|
|
2482
|
+
// `-32602: spec validation failed` that names neither the file nor the cause.
|
|
2483
|
+
const specCompat = tsCompatibilityOf(specYaml);
|
|
2484
|
+
if (!specCompat.compatible) {
|
|
2485
|
+
throw new Error(quarantineMessage(specPath, specCompat));
|
|
2486
|
+
}
|
|
2487
|
+
|
|
2163
2488
|
// STRAT-IMMUTABLE: hash the on-disk spec BEFORE triage mutation for tamper detection.
|
|
2164
2489
|
// verifyPipelineIntegrity() re-reads from disk, so we must compare against the original file content.
|
|
2165
2490
|
const specFileHash = _sha256(specYaml);
|
|
@@ -2266,6 +2591,9 @@ export async function runBuild(featureCode, opts = {}) {
|
|
|
2266
2591
|
// Stratum MCP client (test override permitted via opts.stratum)
|
|
2267
2592
|
stratum = opts.stratum ?? new StratumMcpClient();
|
|
2268
2593
|
if (!opts.stratum) await stratum.connect(resolveStratumMcpConnection(cwd));
|
|
2594
|
+
const receiptsMode = typeof stratum.hasTool === 'function'
|
|
2595
|
+
? await stratum.hasTool('stratum_usage_report')
|
|
2596
|
+
: false;
|
|
2269
2597
|
|
|
2270
2598
|
// Update feature.json status to IN_PROGRESS (only modes that track
|
|
2271
2599
|
// feature.json lifecycle status; bug AND plan do not).
|
|
@@ -2396,6 +2724,26 @@ export async function runBuild(featureCode, opts = {}) {
|
|
|
2396
2724
|
}
|
|
2397
2725
|
reviewerAgent = opts.reviewer;
|
|
2398
2726
|
}
|
|
2727
|
+
// COMP-PIPELINE-QUARANTINE round 3: keep the two roles on DIFFERENT providers
|
|
2728
|
+
// unless the caller asked for both explicitly. `--implementer codex` alone
|
|
2729
|
+
// leaves the reviewer at its codex default, so cross-model review silently
|
|
2730
|
+
// becomes Codex reviewing its own work — which is exactly the defect the
|
|
2731
|
+
// review-fix pipeline was corrected for, reintroduced one layer down. When
|
|
2732
|
+
// only one role is overridden, the other flips to the opposite provider.
|
|
2733
|
+
// Setting both to the same provider stays possible, but only deliberately.
|
|
2734
|
+
if (parseAgentString(implementerAgent).provider === parseAgentString(reviewerAgent).provider) {
|
|
2735
|
+
const bothExplicit = opts.implementer != null && opts.reviewer != null;
|
|
2736
|
+
if (!bothExplicit) {
|
|
2737
|
+
const flipped = parseAgentString(implementerAgent).provider === 'codex' ? 'claude' : 'codex';
|
|
2738
|
+
if (opts.reviewer == null) reviewerAgent = flipped;
|
|
2739
|
+
else implementerAgent = flipped;
|
|
2740
|
+
} else {
|
|
2741
|
+
console.warn(
|
|
2742
|
+
`⚠ implementer and reviewer are both ${parseAgentString(implementerAgent).provider}; ` +
|
|
2743
|
+
'cross-model review is disabled for this run.'
|
|
2744
|
+
);
|
|
2745
|
+
}
|
|
2746
|
+
}
|
|
2399
2747
|
const roles = { implementerAgent, reviewerAgent };
|
|
2400
2748
|
// Restore persisted roles when (and only when) a resume actually happens.
|
|
2401
2749
|
const restoreRolesFromActive = (src) => {
|
|
@@ -2562,7 +2910,10 @@ export async function runBuild(featureCode, opts = {}) {
|
|
|
2562
2910
|
// The plan/resume above only created the flow object; no agent/worktree work has
|
|
2563
2911
|
// happened yet, so aborting here still means we never reach `execute` (Codex
|
|
2564
2912
|
// review finding #2). Cached + skippable via COMPOSE_SKIP_CODEX_PROBE.
|
|
2565
|
-
|
|
2913
|
+
// Compare the PROVIDER, not the raw string: `codex:orchestrator` and
|
|
2914
|
+
// `codex::critical` are Codex implementers too, and an exact-string check let
|
|
2915
|
+
// them skip the mandatory worktree probe entirely (round 4 review).
|
|
2916
|
+
if (parseAgentString(implementerAgent).provider === 'codex') {
|
|
2566
2917
|
const probe = await preflightCodexWorktreeProbe({
|
|
2567
2918
|
cwd: agentCwd,
|
|
2568
2919
|
projectCwd: cwd,
|
|
@@ -2667,6 +3018,9 @@ export async function runBuild(featureCode, opts = {}) {
|
|
|
2667
3018
|
})();
|
|
2668
3019
|
|
|
2669
3020
|
const context = {
|
|
3021
|
+
stratum,
|
|
3022
|
+
flowId: response.runId,
|
|
3023
|
+
receiptsMode,
|
|
2670
3024
|
cwd: agentCwd,
|
|
2671
3025
|
projectCwd: cwd,
|
|
2672
3026
|
featureCode,
|
|
@@ -2691,24 +3045,33 @@ export async function runBuild(featureCode, opts = {}) {
|
|
|
2691
3045
|
filesChanged: [...activeAccumulator.files_changed],
|
|
2692
3046
|
...(isBugMode ? { bug_code: featureCode } : {}),
|
|
2693
3047
|
};
|
|
2694
|
-
context.recordBuildUsage = (usage) => {
|
|
3048
|
+
context.recordBuildUsage = async (usage, meta) => {
|
|
2695
3049
|
if (!usage || typeof usage !== 'object') return;
|
|
2696
|
-
const
|
|
2697
|
-
|
|
3050
|
+
const accumulatorUsage = Array.isArray(usage)
|
|
3051
|
+
? usage.reduce((sum, entry) => ({
|
|
3052
|
+
input_tokens: sum.input_tokens + (entry?.input_tokens ?? 0),
|
|
3053
|
+
output_tokens: sum.output_tokens + (entry?.output_tokens ?? entry?.tokens ?? 0),
|
|
3054
|
+
cost_usd: sum.cost_usd + (entry?.cost_usd ?? entry?.usd ?? 0),
|
|
3055
|
+
}), { input_tokens: 0, output_tokens: 0, cost_usd: 0 })
|
|
3056
|
+
: usage;
|
|
3057
|
+
const componentTokens = (typeof accumulatorUsage.input_tokens === 'number' ? accumulatorUsage.input_tokens : 0)
|
|
3058
|
+
+ (typeof accumulatorUsage.output_tokens === 'number' ? accumulatorUsage.output_tokens : 0);
|
|
2698
3059
|
const tokens = componentTokens > 0
|
|
2699
3060
|
? componentTokens
|
|
2700
|
-
: (typeof
|
|
2701
|
-
?
|
|
2702
|
-
: (typeof
|
|
2703
|
-
const usd = typeof
|
|
2704
|
-
?
|
|
2705
|
-
: (typeof
|
|
2706
|
-
if (tokens
|
|
2707
|
-
|
|
2708
|
-
|
|
2709
|
-
|
|
2710
|
-
|
|
2711
|
-
|
|
3061
|
+
: (typeof accumulatorUsage.tokens_total === 'number'
|
|
3062
|
+
? accumulatorUsage.tokens_total
|
|
3063
|
+
: (typeof accumulatorUsage.tokens === 'number' ? accumulatorUsage.tokens : 0));
|
|
3064
|
+
const usd = typeof accumulatorUsage.cost_usd === 'number'
|
|
3065
|
+
? accumulatorUsage.cost_usd
|
|
3066
|
+
: (typeof accumulatorUsage.usd === 'number' ? accumulatorUsage.usd : 0);
|
|
3067
|
+
if (tokens !== 0 || usd !== 0) {
|
|
3068
|
+
updateBuildAccumulator(cwd, featureCode, (accumulator) => ({
|
|
3069
|
+
...accumulator,
|
|
3070
|
+
tokens_total: accumulator.tokens_total + tokens,
|
|
3071
|
+
usd: accumulator.usd + usd,
|
|
3072
|
+
}));
|
|
3073
|
+
}
|
|
3074
|
+
return reportUsageReceipts(context, usage, meta);
|
|
2712
3075
|
};
|
|
2713
3076
|
context.recordFilesChanged = (paths, { authoritativeShip = false } = {}) => {
|
|
2714
3077
|
const normalized = Array.isArray(paths)
|
|
@@ -2721,6 +3084,27 @@ export async function runBuild(featureCode, opts = {}) {
|
|
|
2721
3084
|
: { files_changed: [...new Set([...accumulator.files_changed, ...normalized])] }),
|
|
2722
3085
|
}));
|
|
2723
3086
|
};
|
|
3087
|
+
// COMP-COMPLETION-GATE slice 2: the ship step's completion evidence, carried
|
|
3088
|
+
// to terminalization (where completion now happens, after the health gate).
|
|
3089
|
+
//
|
|
3090
|
+
// `tests_attested` and `evidence_root` are PERSISTED because they cannot be
|
|
3091
|
+
// re-derived later: re-running the suite at terminalization would be a second
|
|
3092
|
+
// run with a different result, and `runBuild` rebuilds the agent cwd from the
|
|
3093
|
+
// current invocation — so a resumed cross-repo build would otherwise verify
|
|
3094
|
+
// the wrong repository. The commit SHA is deliberately NOT persisted; it is
|
|
3095
|
+
// resolved from HEAD at terminalization, where it is verifiable.
|
|
3096
|
+
context.completionEvidence = null;
|
|
3097
|
+
context.recordCompletionEvidence = (evidence = {}) => {
|
|
3098
|
+
context.completionEvidence = { ...(context.completionEvidence || {}), ...evidence };
|
|
3099
|
+
const attested = evidence.testsAttested;
|
|
3100
|
+
if (attested !== undefined) {
|
|
3101
|
+
updateBuildAccumulator(cwd, featureCode, (accumulator) => ({
|
|
3102
|
+
...accumulator,
|
|
3103
|
+
tests_attested: attested,
|
|
3104
|
+
evidence_root: agentCwd,
|
|
3105
|
+
}));
|
|
3106
|
+
}
|
|
3107
|
+
};
|
|
2724
3108
|
context.recordShipTestMetrics = (metrics) => {
|
|
2725
3109
|
if (!metrics || typeof metrics.test_count !== 'number') return;
|
|
2726
3110
|
updateBuildAccumulator(cwd, featureCode, (accumulator) => ({
|
|
@@ -2779,6 +3163,11 @@ export async function runBuild(featureCode, opts = {}) {
|
|
|
2779
3163
|
// is the backstop — if the round can't be threaded for any reason, it trips
|
|
2780
3164
|
// instead of letting the gate spin unbounded (the 52-round loop).
|
|
2781
3165
|
const gateReentries = new Map();
|
|
3166
|
+
// Last consumer-merge preparation/apply failure per merge gate. A failure
|
|
3167
|
+
// that repeats byte-identically means the fan-out re-produced the same
|
|
3168
|
+
// conflict; revising again only re-dispatches every lane for the same result
|
|
3169
|
+
// (observed 2026-08-30: 4 paid rounds on one MERGE_WITNESS_PRECOMPUTE_FAILED).
|
|
3170
|
+
const consumerMergeFailures = new Map();
|
|
2782
3171
|
|
|
2783
3172
|
// The run's effective-spec digest, carried on plan/resume responses only
|
|
2784
3173
|
// (step_done responses omit it). Captured so the merge-gate path can pin the
|
|
@@ -3080,14 +3469,12 @@ export async function runBuild(featureCode, opts = {}) {
|
|
|
3080
3469
|
};
|
|
3081
3470
|
}
|
|
3082
3471
|
verifyPipelineIntegrity(specPath, specFileHash);
|
|
3472
|
+
// `plan_items` is declared by build.stratum.yaml's PhaseResult but NOT
|
|
3473
|
+
// by gsd's, so it stays at this call site rather than in the shared
|
|
3474
|
+
// narrowing helper.
|
|
3083
3475
|
const tsShipOutput = {
|
|
3084
|
-
|
|
3085
|
-
artifact: shipResult.artifact,
|
|
3086
|
-
outcome: shipResult.outcome,
|
|
3087
|
-
summary: shipResult.summary,
|
|
3476
|
+
...toPhaseResultOutput(shipResult),
|
|
3088
3477
|
...(Array.isArray(shipResult.plan_items) ? { plan_items: shipResult.plan_items } : {}),
|
|
3089
|
-
...(Array.isArray(shipResult.filesChanged) ? { files_changed: shipResult.filesChanged } : {}),
|
|
3090
|
-
...(typeof shipResult.commit === 'string' ? { commit_hash: shipResult.commit } : {}),
|
|
3091
3478
|
};
|
|
3092
3479
|
const shipStepResult = response.status === 'ready'
|
|
3093
3480
|
? { output: tsShipOutput }
|
|
@@ -3124,7 +3511,14 @@ export async function runBuild(featureCode, opts = {}) {
|
|
|
3124
3511
|
maxDurationMs: STEP_TIMEOUT_MS[stepId] ?? DEFAULT_TIMEOUT_MS,
|
|
3125
3512
|
stratum,
|
|
3126
3513
|
cwd: agentCwd,
|
|
3127
|
-
|
|
3514
|
+
// C9: the sidecar's `fix` profile carries the fixer's tool
|
|
3515
|
+
// restrictions and model tier. Passing the bare agent literal
|
|
3516
|
+
// here handed the fixer an unrestricted profile; the sibling
|
|
3517
|
+
// review-repair site keys off `fix` the same way. Identity still
|
|
3518
|
+
// comes from the dispatch (`agent: fixAgent`).
|
|
3519
|
+
profile: resolveStepProfile(context.stepProfiles, 'fix')
|
|
3520
|
+
?? resolveStepProfile(context.stepProfiles, stepId),
|
|
3521
|
+
sandboxMode: 'workspace-write',
|
|
3128
3522
|
telemetry: {
|
|
3129
3523
|
site: 'review-repair',
|
|
3130
3524
|
project_cwd: cwd,
|
|
@@ -3135,11 +3529,23 @@ export async function runBuild(featureCode, opts = {}) {
|
|
|
3135
3529
|
},
|
|
3136
3530
|
});
|
|
3137
3531
|
if (fixResult?.usage && typeof context.recordBuildUsage === 'function') {
|
|
3138
|
-
try {
|
|
3532
|
+
try {
|
|
3533
|
+
await context.recordBuildUsage(usagePayload(fixResult.usage, fixResult.usages), {
|
|
3534
|
+
stepId,
|
|
3535
|
+
source: 'fixer',
|
|
3536
|
+
dispatchId: fixResult.dispatchIds?.primary,
|
|
3537
|
+
});
|
|
3538
|
+
} catch { /* fail-open */ }
|
|
3139
3539
|
}
|
|
3140
3540
|
} catch (err) {
|
|
3141
3541
|
if (err?.usage && typeof context.recordBuildUsage === 'function') {
|
|
3142
|
-
try {
|
|
3542
|
+
try {
|
|
3543
|
+
await context.recordBuildUsage(usagePayload(err.usage, err.usages), {
|
|
3544
|
+
stepId,
|
|
3545
|
+
source: 'fixer',
|
|
3546
|
+
dispatchId: err.dispatchId,
|
|
3547
|
+
});
|
|
3548
|
+
} catch { /* fail-open */ }
|
|
3143
3549
|
}
|
|
3144
3550
|
if (err instanceof AgentTimeoutError) {
|
|
3145
3551
|
console.warn(`\n⚠ Fix agent timed out on "${stepId}"`);
|
|
@@ -3154,7 +3560,9 @@ export async function runBuild(featureCode, opts = {}) {
|
|
|
3154
3560
|
const stepStartMs = Date.now();
|
|
3155
3561
|
const agentType = readyStep?.agent ?? response.agent ?? 'claude';
|
|
3156
3562
|
const basePrompt = buildStepPrompt(stepDispatch, context);
|
|
3157
|
-
const maxDurationMs =
|
|
3563
|
+
const maxDurationMs = process.env.NODE_ENV === 'test' && Number.isFinite(opts.stepTimeoutMs)
|
|
3564
|
+
? opts.stepTimeoutMs
|
|
3565
|
+
: (STEP_TIMEOUT_MS[stepId] ?? DEFAULT_TIMEOUT_MS);
|
|
3158
3566
|
|
|
3159
3567
|
// MF-1/SF-4: Prepend shared review scaffold when this is a review step.
|
|
3160
3568
|
// Also covers a ReviewResult merge step so its output is normalized via
|
|
@@ -3209,14 +3617,19 @@ export async function runBuild(featureCode, opts = {}) {
|
|
|
3209
3617
|
profile: resolveStepProfile(effectiveProfiles, stepId),
|
|
3210
3618
|
});
|
|
3211
3619
|
} catch (err) {
|
|
3620
|
+
const failedUsage = failureUsageFields(err);
|
|
3212
3621
|
if (err instanceof UserInterruptError) {
|
|
3213
3622
|
if (err.action === 'skip') {
|
|
3214
3623
|
if (progress) progress.info(` ⏭ Skipped step "${stepId}"`);
|
|
3215
3624
|
mainResult = {
|
|
3216
3625
|
text: '',
|
|
3217
|
-
|
|
3626
|
+
// `phase` is carried because every pipeline's result contract
|
|
3627
|
+
// declares it and engine contracts are strict — without it a
|
|
3628
|
+
// user-initiated skip fails the step it was meant to bypass.
|
|
3629
|
+
result: { phase: stepId, outcome: 'skipped', summary: 'Skipped by user' },
|
|
3218
3630
|
dispatchIds: { primary: err.dispatchId ?? null, repair: null },
|
|
3219
3631
|
settlementFailureClass: 'agent',
|
|
3632
|
+
...failedUsage,
|
|
3220
3633
|
};
|
|
3221
3634
|
} else {
|
|
3222
3635
|
if (progress) progress.info(` ↻ Retrying step "${stepId}"`);
|
|
@@ -3225,6 +3638,7 @@ export async function runBuild(featureCode, opts = {}) {
|
|
|
3225
3638
|
result: { outcome: 'failed', summary: 'Retry requested by user' },
|
|
3226
3639
|
dispatchIds: { primary: err.dispatchId ?? null, repair: null },
|
|
3227
3640
|
settlementFailureClass: 'agent',
|
|
3641
|
+
...failedUsage,
|
|
3228
3642
|
};
|
|
3229
3643
|
}
|
|
3230
3644
|
} else if (err instanceof AgentTimeoutError) {
|
|
@@ -3235,18 +3649,115 @@ export async function runBuild(featureCode, opts = {}) {
|
|
|
3235
3649
|
result: { outcome: 'failed', summary: `Timed out after ${Math.round(err.durationMs / 1000)}s` },
|
|
3236
3650
|
dispatchIds: { primary: err.dispatchId ?? null, repair: null },
|
|
3237
3651
|
settlementFailureClass: 'agent',
|
|
3238
|
-
...
|
|
3652
|
+
...failedUsage,
|
|
3239
3653
|
};
|
|
3240
3654
|
} else {
|
|
3241
3655
|
// Fatal rethrow bypasses the post-stepDone usage fold — bill the
|
|
3242
3656
|
// attempt's real cost to the accumulator before crashing.
|
|
3243
|
-
if (
|
|
3244
|
-
try {
|
|
3657
|
+
if (failedUsage.usage && typeof context.recordBuildUsage === 'function') {
|
|
3658
|
+
try {
|
|
3659
|
+
await context.recordBuildUsage(usagePayload(failedUsage.usage, failedUsage.usages), {
|
|
3660
|
+
stepId,
|
|
3661
|
+
source: 'main',
|
|
3662
|
+
dispatchId: err.dispatchId,
|
|
3663
|
+
});
|
|
3664
|
+
} catch { /* fail-open */ }
|
|
3245
3665
|
}
|
|
3246
3666
|
streamWriter.write({ type: 'build_error', message: err.message, stepId });
|
|
3247
3667
|
throw err;
|
|
3248
3668
|
}
|
|
3249
3669
|
}
|
|
3670
|
+
// COMP-POLICY-CHECK-3/4: scan the candidate response against the local
|
|
3671
|
+
// adherence catalog before it is accepted, then allow the agent exactly
|
|
3672
|
+
// one revision pass. Never hard-blocks (design: "surfaces violations for
|
|
3673
|
+
// revision; it does not refuse to emit") and never rewrites the draft.
|
|
3674
|
+
let policyViolationStrings = [];
|
|
3675
|
+
let policyUnsuppressedCount = 0;
|
|
3676
|
+
{
|
|
3677
|
+
const skillGated = isGateStep(localSpec, localFlowName, stepId);
|
|
3678
|
+
const traceArgs = { cwd, streamWriter, stepId, featureCode, buildId: build_id };
|
|
3679
|
+
|
|
3680
|
+
let scan = policyScanForStep({ cwd, text: mainResult?.text, skillGated });
|
|
3681
|
+
recordPolicyScan({ ...traceArgs, records: scan.records, userMode: scan.userMode, pass: 'initial' });
|
|
3682
|
+
|
|
3683
|
+
if (scan.violations.length > 0) {
|
|
3684
|
+
if (progress) progress.warn(`Policy check: ${scan.violations.length} unsuppressed violation(s) — requesting one revision`);
|
|
3685
|
+
try {
|
|
3686
|
+
const revised = await runAndNormalize(
|
|
3687
|
+
null,
|
|
3688
|
+
`${prompt}\n\n${buildRevisionNotice(scan.records)}`,
|
|
3689
|
+
stepDispatch,
|
|
3690
|
+
{
|
|
3691
|
+
progress, streamWriter, maxDurationMs, stratum, cwd: agentCwd,
|
|
3692
|
+
reviewMode: isReviewMain,
|
|
3693
|
+
confidenceGate: confGateMain,
|
|
3694
|
+
profile: resolveStepProfile(effectiveProfiles, stepId),
|
|
3695
|
+
telemetry: {
|
|
3696
|
+
site: 'policy-revision',
|
|
3697
|
+
project_cwd: cwd,
|
|
3698
|
+
build_id,
|
|
3699
|
+
feature_code: featureCode,
|
|
3700
|
+
step_id: stepId,
|
|
3701
|
+
...(typeof readyStep?.attempt === 'number' ? { attempt: readyStep.attempt } : {}),
|
|
3702
|
+
},
|
|
3703
|
+
},
|
|
3704
|
+
);
|
|
3705
|
+
const replaces = revised && !revised.normalizationFailure && revised.result?.outcome !== 'failed';
|
|
3706
|
+
if (!replaces && revised?.usage && typeof context.recordBuildUsage === 'function') {
|
|
3707
|
+
// Rejected revision: its cost still happened, and no merged
|
|
3708
|
+
// usage will carry it, so bill it like the review fixer's.
|
|
3709
|
+
try {
|
|
3710
|
+
await context.recordBuildUsage(usagePayload(revised.usage, revised.usages), {
|
|
3711
|
+
stepId,
|
|
3712
|
+
source: 'policy_revision',
|
|
3713
|
+
dispatchId: revised.dispatchIds?.primary,
|
|
3714
|
+
});
|
|
3715
|
+
} catch { /* fail-open */ }
|
|
3716
|
+
}
|
|
3717
|
+
if (replaces) {
|
|
3718
|
+
await reportUsageReceipts(
|
|
3719
|
+
context,
|
|
3720
|
+
usagePayload(revised.usage, revised.usages),
|
|
3721
|
+
{ stepId, source: 'policy_revision', dispatchId: revised.dispatchIds?.primary },
|
|
3722
|
+
);
|
|
3723
|
+
// The replacement carries BOTH dispatch ids (so settlement
|
|
3724
|
+
// settles both) and the summed usage (so step_usage, build
|
|
3725
|
+
// totals, and build-history include the revision's cost — the
|
|
3726
|
+
// step_usage block below is the single accumulator call).
|
|
3727
|
+
mainResult = {
|
|
3728
|
+
...mainResult,
|
|
3729
|
+
text: revised.text ?? mainResult.text,
|
|
3730
|
+
result: revised.result ?? mainResult.result,
|
|
3731
|
+
usage: mergeUsage(mainResult.usage, revised.usage),
|
|
3732
|
+
dispatchIds: {
|
|
3733
|
+
...(mainResult.dispatchIds ?? {}),
|
|
3734
|
+
revision: revised.dispatchIds?.primary ?? null,
|
|
3735
|
+
},
|
|
3736
|
+
};
|
|
3737
|
+
scan = policyScanForStep({ cwd, text: mainResult.text, skillGated });
|
|
3738
|
+
recordPolicyScan({ ...traceArgs, records: scan.records, userMode: scan.userMode, pass: 'policy_revision' });
|
|
3739
|
+
}
|
|
3740
|
+
// The second result stands either way — no further passes.
|
|
3741
|
+
} catch (err) {
|
|
3742
|
+
if (err?.usage && typeof context.recordBuildUsage === 'function') {
|
|
3743
|
+
try {
|
|
3744
|
+
await context.recordBuildUsage(usagePayload(err.usage, err.usages), {
|
|
3745
|
+
stepId,
|
|
3746
|
+
source: 'policy_revision',
|
|
3747
|
+
dispatchId: err.dispatchId,
|
|
3748
|
+
});
|
|
3749
|
+
} catch { /* fail-open */ }
|
|
3750
|
+
}
|
|
3751
|
+
if (policyRevisionMustStop(err)) throw err;
|
|
3752
|
+
// eslint-disable-next-line no-console
|
|
3753
|
+
console.warn(`[policy-check] revision pass failed on "${stepId}" — keeping the original draft: ${err.message}`);
|
|
3754
|
+
}
|
|
3755
|
+
}
|
|
3756
|
+
|
|
3757
|
+
policyViolationStrings = scan.violations;
|
|
3758
|
+
policyUnsuppressedCount = scan.violations.length;
|
|
3759
|
+
}
|
|
3760
|
+
|
|
3250
3761
|
const { result, text: stepText, usage: stepUsage, normalizationFailure } = mainResult;
|
|
3251
3762
|
|
|
3252
3763
|
// Scan agent output for "we should X" / "we could X" patterns that don't map
|
|
@@ -3401,17 +3912,36 @@ export async function runBuild(featureCode, opts = {}) {
|
|
|
3401
3912
|
// lens rather than a normalization-stamped 'general'.
|
|
3402
3913
|
lastReviewMergeDirtyLenses = extractDirtyLenses(stepText ?? result);
|
|
3403
3914
|
}
|
|
3915
|
+
// COMP-POLICY-CHECK-6: expose the unsuppressed count on the step result
|
|
3916
|
+
// so a spec can declare `ensure: ['result.unsuppressed_violations == 0']`.
|
|
3917
|
+
// Engine contracts are strict, so the field is attached only where it is
|
|
3918
|
+
// declared (or where the step has no out contract).
|
|
3919
|
+
const policyResult = attachPolicyCount(result, policyUnsuppressedCount, stepDispatch);
|
|
3404
3920
|
const stepDoneResult = readyStep
|
|
3405
3921
|
? blockingFailure
|
|
3406
3922
|
? { failure: blockingFailure }
|
|
3407
3923
|
: normalizationFailure || result?.outcome === 'failed'
|
|
3408
3924
|
? { failure: String(normalizationFailure ?? result?.summary ?? `Step "${stepId}" did not produce structured output`) }
|
|
3409
3925
|
: stepDispatch.has_out_contract
|
|
3410
|
-
?
|
|
3411
|
-
? { output:
|
|
3926
|
+
? policyResult != null
|
|
3927
|
+
? { output: policyResult }
|
|
3412
3928
|
: { failure: `Step "${stepId}" did not produce structured output` }
|
|
3413
3929
|
: {}
|
|
3414
|
-
:
|
|
3930
|
+
: policyResult ?? { summary: 'Step complete' };
|
|
3931
|
+
// Report each model dispatch before the outcome it funded. The merged
|
|
3932
|
+
// step usage remains the accumulator/build-stream shape used by existing
|
|
3933
|
+
// callers; receipts use mainResult.usages to preserve per-dispatch data.
|
|
3934
|
+
if (toEngineUsage(stepUsage)) {
|
|
3935
|
+
buildCostTotals.input_tokens += stepUsage.input_tokens ?? 0;
|
|
3936
|
+
buildCostTotals.output_tokens += stepUsage.output_tokens ?? 0;
|
|
3937
|
+
buildCostTotals.cost_usd += stepUsage.cost_usd ?? 0;
|
|
3938
|
+
await context.recordBuildUsage(usagePayload(stepUsage, mainResult.usages), {
|
|
3939
|
+
stepId,
|
|
3940
|
+
source: 'main',
|
|
3941
|
+
dispatchId: mainResult.dispatchIds?.primary,
|
|
3942
|
+
});
|
|
3943
|
+
streamWriter.writeUsage(stepId, stepUsage);
|
|
3944
|
+
}
|
|
3415
3945
|
response = await stratum.stepDone(
|
|
3416
3946
|
flowId, stepId, stepDoneResult, readyStep?.dispatchToken,
|
|
3417
3947
|
);
|
|
@@ -3528,17 +4058,10 @@ export async function runBuild(featureCode, opts = {}) {
|
|
|
3528
4058
|
{
|
|
3529
4059
|
const buildState = readActiveBuild(dataDir);
|
|
3530
4060
|
const stepState = buildState?.steps?.find(s => s.id === stepId) ?? {};
|
|
3531
|
-
// COMP-
|
|
3532
|
-
|
|
3533
|
-
|
|
3534
|
-
|
|
3535
|
-
buildCostTotals.cost_usd += stepUsage.cost_usd ?? 0;
|
|
3536
|
-
context.recordBuildUsage(stepUsage);
|
|
3537
|
-
streamWriter.writeUsage(stepId, stepUsage);
|
|
3538
|
-
}
|
|
3539
|
-
|
|
3540
|
-
// COMP-HEALTH: collect runtime violations for health score signal
|
|
3541
|
-
const stepViolations = stepState.violations ?? [];
|
|
4061
|
+
// COMP-HEALTH: collect runtime violations for health score signal.
|
|
4062
|
+
// COMP-POLICY-CHECK-3: unsuppressed policy violations join the same
|
|
4063
|
+
// stream — ViolationDetail renders them with zero UI changes.
|
|
4064
|
+
const stepViolations = [...(stepState.violations ?? []), ...policyViolationStrings];
|
|
3542
4065
|
if (stepViolations.length > 0) {
|
|
3543
4066
|
allViolations.push(...stepViolations);
|
|
3544
4067
|
}
|
|
@@ -3684,10 +4207,18 @@ export async function runBuild(featureCode, opts = {}) {
|
|
|
3684
4207
|
) ?? null;
|
|
3685
4208
|
}
|
|
3686
4209
|
}
|
|
4210
|
+
const repairFor = (error) => {
|
|
4211
|
+
const failure = `${error.code}: ${error.message}`;
|
|
4212
|
+
const decision = decideMergeRepairOutcome(
|
|
4213
|
+
consumerMergeFailures.get(stepId), failure, repairOutcome,
|
|
4214
|
+
);
|
|
4215
|
+
consumerMergeFailures.set(stepId, failure);
|
|
4216
|
+
outcome = decision.outcome;
|
|
4217
|
+
rationale = decision.rationale;
|
|
4218
|
+
};
|
|
3687
4219
|
if (consumerMergeArtifacts && outcome === 'approve') {
|
|
3688
4220
|
if (consumerMergePreparationError) {
|
|
3689
|
-
|
|
3690
|
-
rationale = `${consumerMergePreparationError.code}: ${consumerMergePreparationError.message}`;
|
|
4221
|
+
repairFor(consumerMergePreparationError);
|
|
3691
4222
|
} else {
|
|
3692
4223
|
try {
|
|
3693
4224
|
await consumerMergeArtifacts.applyMerge(consumerMergeTransaction);
|
|
@@ -3705,8 +4236,7 @@ export async function runBuild(featureCode, opts = {}) {
|
|
|
3705
4236
|
} catch { /* best-effort build context projection */ }
|
|
3706
4237
|
} catch (error) {
|
|
3707
4238
|
if (!(error instanceof ConsumerMergeDecisionError)) throw error;
|
|
3708
|
-
|
|
3709
|
-
rationale = `${error.code}: ${error.message}`;
|
|
4239
|
+
repairFor(error);
|
|
3710
4240
|
}
|
|
3711
4241
|
}
|
|
3712
4242
|
}
|
|
@@ -3792,8 +4322,18 @@ export async function runBuild(featureCode, opts = {}) {
|
|
|
3792
4322
|
let reviewResult = lastReviewMergeResult;
|
|
3793
4323
|
let rawDirtyLenses = lastReviewMergeDirtyLenses;
|
|
3794
4324
|
if (!reviewResult) {
|
|
3795
|
-
|
|
3796
|
-
|
|
4325
|
+
// COMP-PIPELINE-QUARANTINE follow-up: the re-derive used to read
|
|
4326
|
+
// `steps.review_merge.output` by that literal name, so any pipeline
|
|
4327
|
+
// whose reducer is called something else (team-review's `merge`,
|
|
4328
|
+
// review-fix's `review`) silently got a null stash on resume and had
|
|
4329
|
+
// its clean review treated as dirty. The stash above is already keyed
|
|
4330
|
+
// on the sidecar's _reduceSteps; this now matches it, with the
|
|
4331
|
+
// canonical id kept as the fallback.
|
|
4332
|
+
const reducerIds = [...reduceSteps, 'review_merge'];
|
|
4333
|
+
const auditedOutput = reducerIds
|
|
4334
|
+
.map((id) => gateAudit?.steps?.[id]?.output ?? null)
|
|
4335
|
+
.find((out) => out && typeof out === 'object') ?? null;
|
|
4336
|
+
if (auditedOutput) {
|
|
3797
4337
|
reviewResult = auditedOutput;
|
|
3798
4338
|
rawDirtyLenses = extractDirtyLenses(auditedOutput);
|
|
3799
4339
|
}
|
|
@@ -3833,7 +4373,7 @@ export async function runBuild(featureCode, opts = {}) {
|
|
|
3833
4373
|
try {
|
|
3834
4374
|
const gateFixResult = await runAndNormalize(undefined, fixPrompt, { step_id: 'review_fix', agent: fixAgent, flow_id: flowId }, {
|
|
3835
4375
|
progress, streamWriter, maxDurationMs: STEP_TIMEOUT_MS.review_merge ?? DEFAULT_TIMEOUT_MS,
|
|
3836
|
-
stratum, cwd: agentCwd, profile: resolveStepProfile(effectiveProfiles, 'fix'),
|
|
4376
|
+
stratum, cwd: agentCwd, sandboxMode: 'workspace-write', profile: resolveStepProfile(effectiveProfiles, 'fix'),
|
|
3837
4377
|
telemetry: {
|
|
3838
4378
|
site: 'review-repair',
|
|
3839
4379
|
project_cwd: cwd,
|
|
@@ -3844,11 +4384,23 @@ export async function runBuild(featureCode, opts = {}) {
|
|
|
3844
4384
|
},
|
|
3845
4385
|
});
|
|
3846
4386
|
if (gateFixResult?.usage && typeof context.recordBuildUsage === 'function') {
|
|
3847
|
-
try {
|
|
4387
|
+
try {
|
|
4388
|
+
await context.recordBuildUsage(usagePayload(gateFixResult.usage, gateFixResult.usages), {
|
|
4389
|
+
stepId: gateStepId,
|
|
4390
|
+
source: 'gate_fixer',
|
|
4391
|
+
dispatchId: gateFixResult.dispatchIds?.primary,
|
|
4392
|
+
});
|
|
4393
|
+
} catch { /* fail-open */ }
|
|
3848
4394
|
}
|
|
3849
4395
|
} catch (err) {
|
|
3850
4396
|
if (err?.usage && typeof context.recordBuildUsage === 'function') {
|
|
3851
|
-
try {
|
|
4397
|
+
try {
|
|
4398
|
+
await context.recordBuildUsage(usagePayload(err.usage, err.usages), {
|
|
4399
|
+
stepId: gateStepId,
|
|
4400
|
+
source: 'gate_fixer',
|
|
4401
|
+
dispatchId: err.dispatchId,
|
|
4402
|
+
});
|
|
4403
|
+
} catch { /* fail-open */ }
|
|
3852
4404
|
}
|
|
3853
4405
|
if (!(err instanceof AgentTimeoutError)) throw err;
|
|
3854
4406
|
console.warn('\n⚠ Review fixer timed out');
|
|
@@ -4015,6 +4567,9 @@ export async function runBuild(featureCode, opts = {}) {
|
|
|
4015
4567
|
}
|
|
4016
4568
|
|
|
4017
4569
|
// Flow complete — write terminal state (file retained per STRAT-COMP-4 contract).
|
|
4570
|
+
// COMP-COMPLETION-GATE slice 2: set when the flow completed and the feature is
|
|
4571
|
+
// eligible to be completed — the actual completion runs after the health gate.
|
|
4572
|
+
let pendingCompletion = null;
|
|
4018
4573
|
if (response.status === 'completed') buildStatus = 'complete';
|
|
4019
4574
|
if ((response.status === 'failed' || response.status === 'budget_exhausted') && !killedByGate) {
|
|
4020
4575
|
buildStatus = 'failed';
|
|
@@ -4022,20 +4577,19 @@ export async function runBuild(featureCode, opts = {}) {
|
|
|
4022
4577
|
}
|
|
4023
4578
|
if (response.status === 'completed' && buildStatus === 'complete') {
|
|
4024
4579
|
console.log('\nBuild complete.');
|
|
4025
|
-
|
|
4026
|
-
//
|
|
4027
|
-
//
|
|
4028
|
-
|
|
4029
|
-
|
|
4030
|
-
|
|
4031
|
-
|
|
4032
|
-
|
|
4033
|
-
|
|
4034
|
-
|
|
4035
|
-
|
|
4036
|
-
|
|
4037
|
-
|
|
4038
|
-
}
|
|
4580
|
+
// COMP-COMPLETION-GATE slice 2: the COMPLETION does not happen here.
|
|
4581
|
+
//
|
|
4582
|
+
// This block used to flip the vision item and feature.json to COMPLETE
|
|
4583
|
+
// immediately — but the health gate below can still downgrade the build to
|
|
4584
|
+
// `failed`, and the guard ledger is append-only. Completing here meant a
|
|
4585
|
+
// health-rejected build was left marked COMPLETE, and (once gated) the very
|
|
4586
|
+
// first thing the ledger would ever durably attest would be a build the
|
|
4587
|
+
// system itself then judged a failure.
|
|
4588
|
+
//
|
|
4589
|
+
// The health verdict is a PRECONDITION of completion, not its successor, so
|
|
4590
|
+
// the completion is deferred to the gated block below, which runs after the
|
|
4591
|
+
// health gate. See COMP-COMPLETION-GATE design §2.3c.
|
|
4592
|
+
pendingCompletion = { itemId, featureCode };
|
|
4039
4593
|
const termState = readActiveBuild(dataDir);
|
|
4040
4594
|
if (termState) {
|
|
4041
4595
|
writeActiveBuild(dataDir, { ...termState, status: 'complete', completedAt: new Date().toISOString() });
|
|
@@ -4095,16 +4649,29 @@ export async function runBuild(featureCode, opts = {}) {
|
|
|
4095
4649
|
buildSignals.runtime_errors = [];
|
|
4096
4650
|
}
|
|
4097
4651
|
|
|
4098
|
-
// Doc freshness —
|
|
4652
|
+
// Doc freshness — derivation-based staleness (COMP-PROV-LINEAGE). An
|
|
4653
|
+
// artifact is stale when an upstream it wasDerivedFrom is newer than it.
|
|
4654
|
+
// This replaced the old phase-marker staleness reader (now removed),
|
|
4655
|
+
// which read a `<!-- phase: -->` marker that no production writer ever
|
|
4656
|
+
// emitted (a dead signal). findStaleArtifacts works off the canonical
|
|
4657
|
+
// chain + mtimes, so it needs no marker to be written first.
|
|
4099
4658
|
try {
|
|
4100
|
-
const {
|
|
4101
|
-
|
|
4102
|
-
? stepHistory[stepHistory.length - 1].stepId
|
|
4103
|
-
: 'build';
|
|
4104
|
-
const stalenessResults = checkStaleness(resolveItemDir(featureCode), currentPhase);
|
|
4105
|
-
buildSignals.doc_freshness = stalenessResults;
|
|
4659
|
+
const { findStaleArtifacts } = await import('./lineage.js');
|
|
4660
|
+
buildSignals.doc_freshness = findStaleArtifacts(resolveItemDir(featureCode));
|
|
4106
4661
|
} catch { /* staleness check is optional — skip on error */ }
|
|
4107
4662
|
|
|
4663
|
+
// COMP-PROV-LINEAGE — populate PROV-O lineage markers on the canonical
|
|
4664
|
+
// artifacts that now exist. This runs once per build, in the finalization
|
|
4665
|
+
// pass after the dispatch loop, when the artifact set is complete. Build
|
|
4666
|
+
// is the single writer here, so the read/write/utimes in stampFeatureLineage
|
|
4667
|
+
// is uncontended. Idempotent and mtime-preserving, so it never resets the
|
|
4668
|
+
// derivation clock that staleness reachability depends on. This is the
|
|
4669
|
+
// lifecycle-writer surface that materialises wasGeneratedBy/wasDerivedFrom.
|
|
4670
|
+
try {
|
|
4671
|
+
const { stampFeatureLineage } = await import('./lineage.js');
|
|
4672
|
+
stampFeatureLineage(resolveItemDir(featureCode));
|
|
4673
|
+
} catch { /* lineage stamping is optional — skip on error */ }
|
|
4674
|
+
|
|
4108
4675
|
const healthSettings = (() => {
|
|
4109
4676
|
try {
|
|
4110
4677
|
if (existsSync(settingsPath)) {
|
|
@@ -4167,6 +4734,97 @@ export async function runBuild(featureCode, opts = {}) {
|
|
|
4167
4734
|
}
|
|
4168
4735
|
}
|
|
4169
4736
|
|
|
4737
|
+
// ---------------------------------------------------------------------
|
|
4738
|
+
// COMP-COMPLETION-GATE slice 2 — THE completion, and the only one.
|
|
4739
|
+
//
|
|
4740
|
+
// Runs here, after the health gate above may have downgraded buildStatus, so
|
|
4741
|
+
// a health-rejected build completes nothing: no completion record, no
|
|
4742
|
+
// COMPLETE status, no vision completion, no guard transition.
|
|
4743
|
+
// ---------------------------------------------------------------------
|
|
4744
|
+
if (pendingCompletion && buildStatus === 'complete') {
|
|
4745
|
+
const ev = context.completionEvidence || {};
|
|
4746
|
+
const acc = readBuildAccumulator(cwd, featureCode);
|
|
4747
|
+
// Persisted, because it survives a resume; the in-memory value wins when
|
|
4748
|
+
// this process ran the ship step itself.
|
|
4749
|
+
const testsAttested = ev.testsAttested ?? acc?.tests_attested ?? 'no-signal';
|
|
4750
|
+
const evidenceRoot = acc?.evidence_root || agentCwd;
|
|
4751
|
+
|
|
4752
|
+
// The SHA is resolved here rather than carried: at terminalization HEAD is
|
|
4753
|
+
// the commit the build produced (or, on the already-committed path, found),
|
|
4754
|
+
// and resolving it at the point of use keeps it verifiable instead of a
|
|
4755
|
+
// stale claim threaded across a resume boundary.
|
|
4756
|
+
let commitSha = ev.commitSha ?? null;
|
|
4757
|
+
if (!commitSha) {
|
|
4758
|
+
try {
|
|
4759
|
+
commitSha = execSync('git rev-parse HEAD', {
|
|
4760
|
+
cwd: evidenceRoot, encoding: 'utf-8', timeout: 5000, stdio: ['ignore', 'pipe', 'pipe'],
|
|
4761
|
+
}).trim() || null;
|
|
4762
|
+
} catch { /* no repo — the no-repo exemption applies below */ }
|
|
4763
|
+
}
|
|
4764
|
+
|
|
4765
|
+
if (cfg.tracksFeatureJson) {
|
|
4766
|
+
const { completionGate, guardEnabled } = await import('./completion-gate.js');
|
|
4767
|
+
// `capabilities.guard: false` is a deliberate opt-OUT, and the gate itself
|
|
4768
|
+
// honors it (completion-gate.js §"Evidence, BEFORE the lock" / AC-5). This
|
|
4769
|
+
// refusal has to sit INSIDE the same regime: enforcing attestation on an
|
|
4770
|
+
// opted-out project would break every one of them — including non-git
|
|
4771
|
+
// workspaces, where the evidence can never pass at all — which is exactly
|
|
4772
|
+
// the reversal already made once during slice 1. An opted-out project keeps
|
|
4773
|
+
// `deriveTestsPass`'s degrade contract: 'no-signal' reads as true there.
|
|
4774
|
+
const guarded = guardEnabled(cwd);
|
|
4775
|
+
|
|
4776
|
+
// `no-signal` is not an attestation. Refusing here is the whole point of
|
|
4777
|
+
// the tri-state: an unreadable test run must not become a passing claim
|
|
4778
|
+
// on a permanent record.
|
|
4779
|
+
if (guarded && testsAttested === 'no-signal') {
|
|
4780
|
+
console.warn(
|
|
4781
|
+
`[completion-gate] ${featureCode}: tests could not be attested (test output was ` +
|
|
4782
|
+
`unreadable). Configure guard.testCommand in .compose/compose.json so the test run ` +
|
|
4783
|
+
`itself attests, or record the completion explicitly. Build is complete; the feature ` +
|
|
4784
|
+
`is NOT marked COMPLETE.`,
|
|
4785
|
+
);
|
|
4786
|
+
} else {
|
|
4787
|
+
const gated = await completionGate({
|
|
4788
|
+
featureCode,
|
|
4789
|
+
commitSha,
|
|
4790
|
+
testsPass: guarded ? testsAttested === 'passed' : testsAttested !== 'failed',
|
|
4791
|
+
filesChanged: ev.filesChanged ?? context.filesChanged ?? [],
|
|
4792
|
+
notes: ev.notes,
|
|
4793
|
+
builtVia: ev.builtVia,
|
|
4794
|
+
workspaceRoot: cwd,
|
|
4795
|
+
evidenceRoot,
|
|
4796
|
+
mode: resolveMode(mode),
|
|
4797
|
+
// Slice 3: the gate owns the vision projection (§2.3a step 4) through
|
|
4798
|
+
// the self-verifying seam; `updateItemStatus(…, 'complete')` refuses
|
|
4799
|
+
// managed build items now (AC-16).
|
|
4800
|
+
visionItemId: pendingCompletion.itemId,
|
|
4801
|
+
visionProjector: ({ featureCode: fc, commitSha: sha, ledgerRef }) =>
|
|
4802
|
+
visionWriter.completeItem(pendingCompletion.itemId, { featureCode: fc, cwd, commitSha: sha, ledgerRef }),
|
|
4803
|
+
});
|
|
4804
|
+
if (gated.ok) {
|
|
4805
|
+
if (gated.partial) {
|
|
4806
|
+
console.warn(
|
|
4807
|
+
`[completion-gate] ${featureCode}: completed, but a projection failed — ` +
|
|
4808
|
+
gated.failures.map(f => `${f.step}: ${f.message} (recover: ${f.recover})`).join('; '),
|
|
4809
|
+
);
|
|
4810
|
+
}
|
|
4811
|
+
} else {
|
|
4812
|
+
// Do NOT fail the build: the work is committed and the flow finished.
|
|
4813
|
+
// But do not claim completion either — say plainly what was refused.
|
|
4814
|
+
console.warn(
|
|
4815
|
+
`[completion-gate] ${featureCode}: completion refused at ${gated.refusedAt} — ` +
|
|
4816
|
+
`${(gated.reasons || []).join('; ')}. The build finished and the commit stands; ` +
|
|
4817
|
+
`the feature is NOT marked COMPLETE.`,
|
|
4818
|
+
);
|
|
4819
|
+
}
|
|
4820
|
+
}
|
|
4821
|
+
} else {
|
|
4822
|
+
// Bug/plan modes have no feature.json to gate on (COMP-FIX-HARD T4), and
|
|
4823
|
+
// slice 1/2 are scoped to build mode — their completion path is unchanged.
|
|
4824
|
+
await visionWriter.updateItemStatus(pendingCompletion.itemId, 'complete');
|
|
4825
|
+
}
|
|
4826
|
+
}
|
|
4827
|
+
|
|
4170
4828
|
// COMP-COCKPIT-3: archive the run to build-history.jsonl ONCE, here — after
|
|
4171
4829
|
// the COMP-HEALTH gate above may have downgraded buildStatus to 'failed'.
|
|
4172
4830
|
// Assembled from the in-memory build context for THIS run (never re-read
|
|
@@ -4487,6 +5145,35 @@ function judgmentCanonDriftError({ treeDrift, projectionDrift, recordDrift }) {
|
|
|
4487
5145
|
return err;
|
|
4488
5146
|
}
|
|
4489
5147
|
|
|
5148
|
+
/**
|
|
5149
|
+
* Narrow an executeShipStep return value to the fields the TS engine's
|
|
5150
|
+
* PhaseResult contract declares.
|
|
5151
|
+
*
|
|
5152
|
+
* COMP-SHIP-CONTRACT: engine contracts are STRICT Zod objects — any key the
|
|
5153
|
+
* contract does not declare fails the step. executeShipStep's return carries
|
|
5154
|
+
* caller-facing extras (`commit`, `filesChanged`, `testsAttested`,
|
|
5155
|
+
* `test_count`/`pass_rate`, `error_code`) that Compose's own code consumes but
|
|
5156
|
+
* PhaseResult never declared, so the raw object must NEVER be handed to
|
|
5157
|
+
* stepDone as an `output`. Both sites that do so (runBuild, runGsd) go through
|
|
5158
|
+
* here. Note runBuild's non-`ready` branch passes shipResult as the whole
|
|
5159
|
+
* envelope rather than as `{output}` — a legacy shape that never reaches
|
|
5160
|
+
* contract validation, deliberately left untouched.
|
|
5161
|
+
*
|
|
5162
|
+
* @param {object} shipResult Return value from executeShipStep
|
|
5163
|
+
* @returns {object} `{phase, artifact, outcome, summary}` plus
|
|
5164
|
+
* `files_changed`/`commit_hash` when present
|
|
5165
|
+
*/
|
|
5166
|
+
export function toPhaseResultOutput(shipResult) {
|
|
5167
|
+
return {
|
|
5168
|
+
phase: shipResult.phase,
|
|
5169
|
+
artifact: shipResult.artifact,
|
|
5170
|
+
outcome: shipResult.outcome,
|
|
5171
|
+
summary: shipResult.summary,
|
|
5172
|
+
...(Array.isArray(shipResult.filesChanged) ? { files_changed: shipResult.filesChanged } : {}),
|
|
5173
|
+
...(typeof shipResult.commit === 'string' ? { commit_hash: shipResult.commit } : {}),
|
|
5174
|
+
};
|
|
5175
|
+
}
|
|
5176
|
+
|
|
4490
5177
|
/**
|
|
4491
5178
|
* Execute the ship step: run tests, stage feature files, commit.
|
|
4492
5179
|
* Returns a PhaseResult-shaped object.
|
|
@@ -4509,7 +5196,35 @@ export async function executeShipStep(featureCode, agentCwd, cwd, context, descr
|
|
|
4509
5196
|
const builtVia = context?.templateName === 'build-quick' ? 'build-quick' : null;
|
|
4510
5197
|
|
|
4511
5198
|
try {
|
|
4512
|
-
// 0.
|
|
5199
|
+
// 0. Run tests FIRST — before the git-availability branch.
|
|
5200
|
+
//
|
|
5201
|
+
// COMP-COMPLETION-GATE slice 2: this used to live below, after the non-git
|
|
5202
|
+
// branch had already returned. That meant a non-git build never ran tests at
|
|
5203
|
+
// all and hard-coded `tests_pass: true` on its completion record. With the
|
|
5204
|
+
// gate refusing an unattested completion, every non-git build would have been
|
|
5205
|
+
// refused — and "no repo" is a reason to skip the COMMIT, never a reason to
|
|
5206
|
+
// skip the tests. Both paths now attest identically.
|
|
5207
|
+
if (progress) progress.toolUse('ship', 'Running tests...');
|
|
5208
|
+
let testSummary = { test_count: 0, pass_rate: 0, parsed: false };
|
|
5209
|
+
try {
|
|
5210
|
+
// COMP-TEST-BOOTSTRAP item 128: use the detected test command, not a hard-coded `npm test`.
|
|
5211
|
+
const testFramework = detectTestFramework(agentCwd);
|
|
5212
|
+
const testCommand = testFramework?.command ?? 'npm test';
|
|
5213
|
+
const testOutput = execSync(`${testCommand} 2>&1 || true`, { cwd: agentCwd, encoding: 'utf-8', timeout: 120_000 });
|
|
5214
|
+
testSummary = parseTestSummary(testFramework?.framework, testOutput);
|
|
5215
|
+
} catch { /* test runner unavailable or timed out — testSummary stays unparsed */ }
|
|
5216
|
+
const testsPass = deriveTestsPass(testSummary);
|
|
5217
|
+
// The gate's input: 'passed' | 'failed' | 'no-signal'. Unlike testsPass, an
|
|
5218
|
+
// unreadable run does NOT become an attestation here.
|
|
5219
|
+
const testsAttested = deriveTestsAttested(testSummary);
|
|
5220
|
+
if (progress && testSummary.parsed) {
|
|
5221
|
+
progress.toolUse('ship', `Tests: ${testSummary.test_count} run, ${testSummary.pass_rate}% passing`);
|
|
5222
|
+
}
|
|
5223
|
+
// Hand the evidence to terminalization, which is where completion now happens
|
|
5224
|
+
// (after the health gate). The ship step no longer completes anything itself.
|
|
5225
|
+
context.recordCompletionEvidence?.({ testsAttested, testSummary });
|
|
5226
|
+
|
|
5227
|
+
// 1. Check if we're in a git repository — if not, skip git operations
|
|
4513
5228
|
let isGitRepo = false;
|
|
4514
5229
|
try {
|
|
4515
5230
|
execSync('git rev-parse --is-inside-work-tree', { cwd: agentCwd, encoding: 'utf-8', timeout: 5000, stdio: 'pipe' });
|
|
@@ -4518,56 +5233,20 @@ export async function executeShipStep(featureCode, agentCwd, cwd, context, descr
|
|
|
4518
5233
|
|
|
4519
5234
|
if (!isGitRepo) {
|
|
4520
5235
|
// COMP-PATHS-EXTERNAL D6b: there is no repo to commit into (e.g. a
|
|
4521
|
-
// forge-top-shaped workspace)
|
|
4522
|
-
//
|
|
4523
|
-
//
|
|
4524
|
-
let completionWarning = null;
|
|
4525
|
-
if (featureCode) {
|
|
4526
|
-
try {
|
|
4527
|
-
const { recordCompletion } = await import('./completion-writer.js');
|
|
4528
|
-
await recordCompletion(cwd, {
|
|
4529
|
-
feature_code: featureCode,
|
|
4530
|
-
// commit_sha omitted — non-git workspace, stamped with the null-SHA
|
|
4531
|
-
// COMP-TEST-BOOTSTRAP-4: this path returns before the test run, so
|
|
4532
|
-
// there is no parsed signal — degrade to true (no block).
|
|
4533
|
-
tests_pass: true,
|
|
4534
|
-
files_changed: [],
|
|
4535
|
-
notes: description.split('\n')[0].slice(0, 72),
|
|
4536
|
-
...(builtVia ? { built_via: builtVia } : {}),
|
|
4537
|
-
});
|
|
4538
|
-
} catch (err) {
|
|
4539
|
-
completionWarning = `completion record failed (${err.code || 'UNKNOWN'}): ${err.message}`;
|
|
4540
|
-
// eslint-disable-next-line no-console
|
|
4541
|
-
console.warn(`[build/ship] ${featureCode}: ${completionWarning}`);
|
|
4542
|
-
}
|
|
4543
|
-
}
|
|
5236
|
+
// forge-top-shaped workspace). The lifecycle still advances, but the
|
|
5237
|
+
// completion is now written at terminalization by the completion gate,
|
|
5238
|
+
// AFTER the health verdict — not here. See COMP-COMPLETION-GATE §2.3c.
|
|
4544
5239
|
return {
|
|
4545
5240
|
phase: 'ship',
|
|
4546
5241
|
artifact: 'no-git',
|
|
4547
5242
|
outcome: 'complete',
|
|
4548
|
-
summary: 'No git repository — wrote artifacts
|
|
5243
|
+
summary: 'No git repository — wrote artifacts (commit skipped)',
|
|
4549
5244
|
commit: null,
|
|
4550
|
-
|
|
5245
|
+
noRepo: true,
|
|
5246
|
+
testsAttested,
|
|
4551
5247
|
};
|
|
4552
5248
|
}
|
|
4553
5249
|
|
|
4554
|
-
// 1. Run feature-relevant tests (best-effort — don't block ship on test infra issues)
|
|
4555
|
-
if (progress) progress.toolUse('ship', 'Running tests...');
|
|
4556
|
-
// COMP-TEST-BOOTSTRAP-4: parse the run output into a structured signal and
|
|
4557
|
-
// derive a real tests_pass for the completion attestation. Degrades to
|
|
4558
|
-
// `true` (no block) whenever the output can't be parsed — see deriveTestsPass.
|
|
4559
|
-
let testSummary = { test_count: 0, pass_rate: 0, parsed: false };
|
|
4560
|
-
try {
|
|
4561
|
-
// COMP-TEST-BOOTSTRAP item 128: use detected test command instead of hard-coded npm test
|
|
4562
|
-
const testFramework = detectTestFramework(agentCwd);
|
|
4563
|
-
const testCommand = testFramework?.command ?? 'npm test';
|
|
4564
|
-
const testOutput = execSync(`${testCommand} 2>&1 || true`, { cwd: agentCwd, encoding: 'utf-8', timeout: 120_000 });
|
|
4565
|
-
testSummary = parseTestSummary(testFramework?.framework, testOutput);
|
|
4566
|
-
} catch { /* test runner not available or timed out — proceed (testSummary stays unparsed) */ }
|
|
4567
|
-
const testsPass = deriveTestsPass(testSummary);
|
|
4568
|
-
if (progress && testSummary.parsed) {
|
|
4569
|
-
progress.toolUse('ship', `Tests: ${testSummary.test_count} run, ${testSummary.pass_rate}% passing`);
|
|
4570
|
-
}
|
|
4571
5250
|
|
|
4572
5251
|
// COMP-TRIAGE-5 (E3 Expand): if a lane-triaged feature fails its ship-time
|
|
4573
5252
|
// test gate, escalate the lane so the NEXT build runs wider. Best-effort —
|
|
@@ -4806,33 +5485,23 @@ export async function executeShipStep(featureCode, agentCwd, cwd, context, descr
|
|
|
4806
5485
|
if (filesChanged.length === 0 && sha) filesChanged = stagedFiles;
|
|
4807
5486
|
context.recordFilesChanged?.(filesChanged, { authoritativeShip: true });
|
|
4808
5487
|
|
|
4809
|
-
// COMP-
|
|
4810
|
-
//
|
|
4811
|
-
//
|
|
4812
|
-
//
|
|
4813
|
-
|
|
4814
|
-
|
|
4815
|
-
|
|
4816
|
-
|
|
4817
|
-
|
|
4818
|
-
|
|
4819
|
-
|
|
4820
|
-
|
|
4821
|
-
|
|
4822
|
-
|
|
4823
|
-
|
|
4824
|
-
|
|
4825
|
-
|
|
4826
|
-
});
|
|
4827
|
-
if (progress) progress.toolUse('ship', `Recorded completion for ${featureCode}`);
|
|
4828
|
-
} catch (err) {
|
|
4829
|
-
completionWarning = err.code === 'STATUS_FLIP_AFTER_COMPLETION_RECORDED'
|
|
4830
|
-
? `completion recorded but status flip failed: ${err.message}`
|
|
4831
|
-
: `completion record failed (${err.code || 'UNKNOWN'}): ${err.message}`;
|
|
4832
|
-
// eslint-disable-next-line no-console
|
|
4833
|
-
console.warn(`[build/ship] ${featureCode}: ${completionWarning}`);
|
|
4834
|
-
}
|
|
4835
|
-
}
|
|
5488
|
+
// COMP-COMPLETION-GATE slice 2: the ship step no longer completes the feature.
|
|
5489
|
+
//
|
|
5490
|
+
// It used to call recordCompletion here, catch ANY failure, and still return
|
|
5491
|
+
// a successful ship outcome — so a completion could fail silently and the
|
|
5492
|
+
// build marched on regardless. Worse, the terminal block then wrote COMPLETE
|
|
5493
|
+
// again independently, and the health gate that can fail the build runs AFTER
|
|
5494
|
+
// both. A health-rejected build was left marked COMPLETE.
|
|
5495
|
+
//
|
|
5496
|
+
// Ship now collects evidence and stops. Exactly one completion happens, at
|
|
5497
|
+
// terminalization, through the gate, after health. See §2.3, §2.3c.
|
|
5498
|
+
context.recordCompletionEvidence?.({
|
|
5499
|
+
commitSha: sha,
|
|
5500
|
+
filesChanged,
|
|
5501
|
+
notes: shortDesc,
|
|
5502
|
+
builtVia,
|
|
5503
|
+
testsAttested,
|
|
5504
|
+
});
|
|
4836
5505
|
|
|
4837
5506
|
// COMP-PATHS-EXTERNAL D6a: if ROADMAP / the feature folder resolved into a
|
|
4838
5507
|
// DIFFERENT git repo, they were written but not committed here — tell the
|
|
@@ -4848,11 +5517,11 @@ export async function executeShipStep(featureCode, agentCwd, cwd, context, descr
|
|
|
4848
5517
|
: `Committed: ${commitMsg} (${stagedFiles.length} files)`,
|
|
4849
5518
|
commit: sha,
|
|
4850
5519
|
filesChanged,
|
|
5520
|
+
testsAttested,
|
|
4851
5521
|
// COMP-MODEL-AB: thread structured test counts into the step result so the
|
|
4852
5522
|
// main loop can persist them to build-history.jsonl for metrics consumers.
|
|
4853
5523
|
// Only present when testSummary.parsed=true (framework detected + output parsed).
|
|
4854
5524
|
...(testSummary.parsed ? { test_count: testSummary.test_count, pass_rate: testSummary.pass_rate } : {}),
|
|
4855
|
-
...(completionWarning ? { completionWarning } : {}),
|
|
4856
5525
|
};
|
|
4857
5526
|
|
|
4858
5527
|
} catch (err) {
|
|
@@ -5025,6 +5694,35 @@ async function pollGateResolution(visionWriter, gateId, intervalMs = 2000) {
|
|
|
5025
5694
|
*/
|
|
5026
5695
|
export const MAX_GATE_REENTRIES = 20;
|
|
5027
5696
|
|
|
5697
|
+
/**
|
|
5698
|
+
* Decide how a merge gate answers a consumer-merge failure.
|
|
5699
|
+
*
|
|
5700
|
+
* The first failure routes to the gate's repair path (`on_revise`, else kill).
|
|
5701
|
+
* A failure that repeats BYTE-IDENTICALLY for the same gate is not going to be
|
|
5702
|
+
* fixed by re-running the fan-out — the lanes reproduced the same conflict —
|
|
5703
|
+
* so the gate kills instead of paying for another round. Anything different
|
|
5704
|
+
* (a new code, a different file) is genuine progress and revises as before.
|
|
5705
|
+
*
|
|
5706
|
+
* @param {string|undefined} previousFailure - `${code}: ${message}` of the last failure at this gate
|
|
5707
|
+
* @param {string} failure - this round's `${code}: ${message}`
|
|
5708
|
+
* @param {'revise'|'kill'} repairOutcome - the gate's configured repair route
|
|
5709
|
+
* @returns {{ outcome: 'revise'|'kill', rationale: string, repeated: boolean }}
|
|
5710
|
+
*/
|
|
5711
|
+
export function decideMergeRepairOutcome(previousFailure, failure, repairOutcome) {
|
|
5712
|
+
const repeated = previousFailure !== undefined && previousFailure === failure;
|
|
5713
|
+
if (repairOutcome === 'revise' && repeated) {
|
|
5714
|
+
return {
|
|
5715
|
+
outcome: 'kill',
|
|
5716
|
+
repeated,
|
|
5717
|
+
rationale: `${failure} — identical to the previous round's failure at this gate; `
|
|
5718
|
+
+ 'the fan-out reproduces the same conflict, so revising would only re-dispatch every '
|
|
5719
|
+
+ 'lane for the same result. Killed to stop spending. A killed build is not resumable: '
|
|
5720
|
+
+ 'fix the conflict (usually lanes editing the same file), then re-run with --fresh.',
|
|
5721
|
+
};
|
|
5722
|
+
}
|
|
5723
|
+
return { outcome: repairOutcome, repeated, rationale: failure };
|
|
5724
|
+
}
|
|
5725
|
+
|
|
5028
5726
|
export function assertGateReentryWithinCap(count, stepId, cap = MAX_GATE_REENTRIES) {
|
|
5029
5727
|
if (count > cap) {
|
|
5030
5728
|
throw new Error(
|
|
@@ -5144,3 +5842,9 @@ export async function abortBuild(dataDir, featureCode, cwd, opts = {}) {
|
|
|
5144
5842
|
}
|
|
5145
5843
|
console.log('Build aborted.');
|
|
5146
5844
|
}
|
|
5845
|
+
|
|
5846
|
+
/** Optional revision failures keep the draft; only user control or uncertain teardown stops the build. */
|
|
5847
|
+
export function policyRevisionMustStop(error) {
|
|
5848
|
+
return error instanceof UserInterruptError
|
|
5849
|
+
|| ['CANCELLATION_UNCONFIRMED', 'CANCELLATION_TEARDOWN_TIMEOUT'].includes(error?.code);
|
|
5850
|
+
}
|