@smartmemory/compose 0.2.52-beta → 0.2.53-beta

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (138) hide show
  1. package/.claude/skills/compose/SKILL.md +0 -8
  2. package/README.md +11 -0
  3. package/bin/compose.js +335 -3
  4. package/dist/assets/App-Bu9KMtTa.js +891 -0
  5. package/dist/assets/abnfDiagram-VRR7QNED-6zp3w9rx.js +1 -0
  6. package/dist/assets/arc-CnLxxuah.js +1 -0
  7. package/dist/assets/architectureDiagram-ZJ3FMSHR-COM46S9k.js +36 -0
  8. package/dist/assets/blockDiagram-677ZJIJ3-wKzgOwF8.js +132 -0
  9. package/dist/assets/{browser-CnKiSnlr.js → browser-Cr0recrN.js} +6 -6
  10. package/dist/assets/{c4Diagram-AHTNJAMY-DtDSNlOX.js → c4Diagram-LMCZKHZV-DIXC_mR6.js} +1 -1
  11. package/dist/assets/channel-DQJPWkb_.js +1 -0
  12. package/dist/assets/{chunk-QZHKN3VN-BMWFLZ8t.js → chunk-2Q5K7J3B-BqLqZX6m.js} +1 -1
  13. package/dist/assets/{chunk-YZCP3GAM-CqjNXooj.js → chunk-32BRIVSS-CarSmVlX.js} +1 -1
  14. package/dist/assets/{chunk-FMBD7UC4-BAa7Edp6.js → chunk-5VM5RSS4-D-f_Jn3G.js} +1 -1
  15. package/dist/assets/chunk-EX3LRPZG-DT42xo8E.js +231 -0
  16. package/dist/assets/{chunk-4BX2VUAB-oXU2npoL.js → chunk-JWPE2WC7-CwY1aZc4.js} +1 -1
  17. package/dist/assets/chunk-MOJQB5TN-DBbZU_KX.js +88 -0
  18. package/dist/assets/chunk-RYQCIY6F-DqxtcLJ3.js +1 -0
  19. package/dist/assets/chunk-V7JOEXUC-Cy4Ixww6.js +206 -0
  20. package/dist/assets/{chunk-EDXVE4YY-Bse2kdRh.js → chunk-VR4S4FIN-CFexfU_a.js} +1 -1
  21. package/dist/assets/{chunk-55IACEB6-BEEk2FdH.js → chunk-XXDRQBXY-uE-zN_vc.js} +1 -1
  22. package/dist/assets/classDiagram-OUVF2IWQ-CTLbAiUK.js +1 -0
  23. package/dist/assets/classDiagram-v2-EOCWNBFH-CTLbAiUK.js +1 -0
  24. package/dist/assets/{cose-bilkent-S5V4N54A-NYnH3mVW.js → cose-bilkent-JH36ORCC-DIjpekos.js} +1 -1
  25. package/dist/assets/cynefin-VYW2F7L2-Ba2gbQow.js +178 -0
  26. package/dist/assets/cynefinDiagram-TSTJHNR4-njGzb7Tg.js +62 -0
  27. package/dist/assets/dagre-VKFMJZFB-DuyZREZ3.js +4 -0
  28. package/dist/assets/diagram-FQU43EPY-Npq-5o3a.js +3 -0
  29. package/dist/assets/diagram-G47NLZAW-fm4k3axC.js +24 -0
  30. package/dist/assets/diagram-NH7WQ7WH-B8EFGrTG.js +24 -0
  31. package/dist/assets/diagram-OA4YK3LP-BC7UHx9Q.js +30 -0
  32. package/dist/assets/diagram-WEI45ONY-BqjHLcCX.js +41 -0
  33. package/dist/assets/ebnfDiagram-CCIWWBDH-D5zXd4Qg.js +1 -0
  34. package/dist/assets/erDiagram-Q63AITRT-DC2FMcra.js +85 -0
  35. package/dist/assets/flowDiagram-23GEKE2U-D7Dx7JU4.js +156 -0
  36. package/dist/assets/ganttDiagram-NO4QXBWP-8sQH2y5K.js +292 -0
  37. package/dist/assets/gitGraphDiagram-IHSO6WYX-Ci_rhxur.js +106 -0
  38. package/dist/assets/graph-C9eacEi8.js +1 -0
  39. package/dist/assets/graph-xkel59g2.js +331 -0
  40. package/dist/assets/index-B7-HQenC.js +119 -0
  41. package/dist/assets/infoDiagram-FWYZ7A6U-B9FGQ0Cb.js +2 -0
  42. package/dist/assets/{ishikawaDiagram-UXIWVN3A-CK2IFFAP.js → ishikawaDiagram-FXEZZL3T-BEKLyH6A.js} +5 -5
  43. package/dist/assets/{journeyDiagram-VCZTEJTY-DAH5Stkf.js → journeyDiagram-5HDEW3XC-BM-IDVRo.js} +1 -1
  44. package/dist/assets/{kanban-definition-6JOO6SKY-Bi0aCqkv.js → kanban-definition-HUTT4EX6-Beb2k3pt.js} +7 -7
  45. package/dist/assets/katex-C5jXJg4s.js +257 -0
  46. package/dist/assets/layout-DEXfKzaS.js +1 -0
  47. package/dist/assets/{linear-ClGEGlS2.js → linear-Dwg7dTpz.js} +1 -1
  48. package/dist/assets/map-Czzmt4hB.js +1 -0
  49. package/dist/assets/{mindmap-definition-QFDTVHPH-Wjr-PgC6.js → mindmap-definition-LN4V7U3C-CeaynfcE.js} +7 -7
  50. package/dist/assets/{mobile-Chw8RWyH.js → mobile-CNLMhdFP.js} +2 -2
  51. package/dist/assets/pegDiagram-2B236MQR-7SVdAVqj.js +1 -0
  52. package/dist/assets/pieDiagram-ENE6RG2P-CzqQ4TaE.js +39 -0
  53. package/dist/assets/quadrantDiagram-ABIIQ3AL-B9B6W80N.js +7 -0
  54. package/dist/assets/railroadDiagram-RFXS5EU6-BRNLawsr.js +1 -0
  55. package/dist/assets/{requirementDiagram-MS252O5E-CU9Rlyq5.js → requirementDiagram-TGXJPOKE-CmASubeG.js} +3 -3
  56. package/dist/assets/sankeyDiagram-HTMAVEWB-DiReR0OB.js +40 -0
  57. package/dist/assets/sequenceDiagram-DBY2YBRQ-DVn5iZKR.js +162 -0
  58. package/dist/assets/sizeCapture-X5ZJPWSS-C8CuDOdp.js +1 -0
  59. package/dist/assets/stateDiagram-2N3HPSRC-BIdkHglY.js +1 -0
  60. package/dist/assets/stateDiagram-v2-6OUMAXLB-B9yje3Er.js +1 -0
  61. package/dist/assets/swimlanes-5IMT3BWC-BlVsYyyN.js +2 -0
  62. package/dist/assets/swimlanesDiagram-G3AALYLV-BTLpU45n.js +8 -0
  63. package/dist/assets/{timeline-definition-GMOUNBTQ-C8SZmaLp.js → timeline-definition-FHXFAJF6-CnTqJmQ2.js} +3 -3
  64. package/dist/assets/vennDiagram-L72KCM5P-D_qZxRhe.js +34 -0
  65. package/dist/assets/wardleyDiagram-EHGQE667-CT-WHFEO.js +78 -0
  66. package/dist/assets/xychartDiagram-FW5EYKEG-CVRFHwUQ.js +7 -0
  67. package/dist/index.html +2 -2
  68. package/lib/agent-string.js +34 -0
  69. package/lib/build.js +542 -148
  70. package/lib/experiment-judge.js +145 -0
  71. package/lib/experiment-metrics.js +265 -0
  72. package/lib/experiment-pricing.js +60 -0
  73. package/lib/experiment-report.js +306 -0
  74. package/lib/experiment-sandbox.js +147 -0
  75. package/lib/experiment.js +539 -0
  76. package/lib/feature-json.js +1 -1
  77. package/lib/feature-writer.js +30 -0
  78. package/lib/flow-state.js +36 -0
  79. package/lib/gate-prompt.js +66 -6
  80. package/lib/lifecycle-modes.js +213 -0
  81. package/lib/new.js +6 -2
  82. package/lib/roadmap-graph/index.js +26 -28
  83. package/lib/roadmap-graph/vision-adapter.js +126 -0
  84. package/lib/stratum-mcp-client.js +9 -0
  85. package/lib/triage.js +7 -1
  86. package/lib/vision-writer.js +30 -6
  87. package/package.json +1 -1
  88. package/pipelines/plan.stratum.yaml +161 -0
  89. package/server/artifact-manager.js +30 -3
  90. package/server/compose-mcp-tools.js +11 -9
  91. package/server/compose-mcp.js +4 -0
  92. package/server/feature-scan.js +87 -173
  93. package/server/graph-export.js +0 -0
  94. package/server/index.js +3 -5
  95. package/server/lifecycle-guard.js +62 -25
  96. package/server/roadmap-graph-vision.js +138 -0
  97. package/server/status-snapshot.js +8 -7
  98. package/server/vision-routes.js +64 -33
  99. package/server/vision-store.js +13 -4
  100. package/.claude/skills/compose/references/hermes-tools.md +0 -80
  101. package/dist/assets/App-Ddhfx18X.js +0 -889
  102. package/dist/assets/_baseUniq-CoEVVYyC.js +0 -1
  103. package/dist/assets/arc-DFnGxuCB.js +0 -1
  104. package/dist/assets/architectureDiagram-Q4EWVU46-66rx01Tn.js +0 -36
  105. package/dist/assets/blockDiagram-DXYQGD6D-AE5PKSak.js +0 -132
  106. package/dist/assets/channel-CScyWprq.js +0 -1
  107. package/dist/assets/chunk-4TB4RGXK-5w2lZAjW.js +0 -206
  108. package/dist/assets/chunk-OYMX7WX6-CGr_YiAc.js +0 -231
  109. package/dist/assets/classDiagram-6PBFFD2Q-DibVYYdr.js +0 -1
  110. package/dist/assets/classDiagram-v2-HSJHXN6E-DibVYYdr.js +0 -1
  111. package/dist/assets/clone-ZMALIFmw.js +0 -1
  112. package/dist/assets/dagre-KV5264BT--mrfaTHR.js +0 -4
  113. package/dist/assets/diagram-5BDNPKRD-DjXvyFzh.js +0 -10
  114. package/dist/assets/diagram-G4DWMVQ6-BrrexEvm.js +0 -24
  115. package/dist/assets/diagram-MMDJMWI5-D3Q0OT2M.js +0 -43
  116. package/dist/assets/diagram-TYMM5635-DKTntmxE.js +0 -24
  117. package/dist/assets/erDiagram-SMLLAGMA-BWbR0dPj.js +0 -85
  118. package/dist/assets/flowDiagram-DWJPFMVM-Dj5M5XaY.js +0 -162
  119. package/dist/assets/ganttDiagram-T4ZO3ILL-DKppFh_5.js +0 -292
  120. package/dist/assets/gitGraphDiagram-UUTBAWPF-9xQ9O25V.js +0 -106
  121. package/dist/assets/graph-Cgmhvu1T.js +0 -1
  122. package/dist/assets/graph-DZe55uk8.js +0 -331
  123. package/dist/assets/index-D8uDfn8y.js +0 -123
  124. package/dist/assets/infoDiagram-42DDH7IO-DQYWVzVN.js +0 -2
  125. package/dist/assets/katex-DkKDou_j.js +0 -257
  126. package/dist/assets/layout-DN13RkA1.js +0 -1
  127. package/dist/assets/min-CXr-fUFS.js +0 -1
  128. package/dist/assets/pieDiagram-DEJITSTG-CVipbIlr.js +0 -30
  129. package/dist/assets/quadrantDiagram-34T5L4WZ-CHJFyhm-.js +0 -7
  130. package/dist/assets/sankeyDiagram-XADWPNL6-BSULeqST.js +0 -10
  131. package/dist/assets/sequenceDiagram-FGHM5R23-CCOKXvjj.js +0 -157
  132. package/dist/assets/stateDiagram-FHFEXIEX-B91gKA7Q.js +0 -1
  133. package/dist/assets/stateDiagram-v2-QKLJ7IA2-Dc8ur9ta.js +0 -1
  134. package/dist/assets/vennDiagram-DHZGUBPP-YvEPDlsM.js +0 -34
  135. package/dist/assets/wardley-RL74JXVD-TMBUq_w1.js +0 -162
  136. package/dist/assets/wardleyDiagram-NUSXRM2D-DzEkPrri.js +0 -20
  137. package/dist/assets/xychartDiagram-5P7HB3ND-BIiCeiYk.js +0 -7
  138. package/lib/roadmap-graph/collect.js +0 -178
@@ -0,0 +1,145 @@
1
+ /**
2
+ * experiment-judge.js — LLM-judge rubric for COMP-MODEL-AB.
3
+ *
4
+ * Rates a build's produced diff against the goal on three axes (1–10 each):
5
+ * correctness — does the code solve the stated goal?
6
+ * clarity — is the code clear and idiomatic?
7
+ * idiomaticity — does it follow language/project conventions?
8
+ *
9
+ * Dispatched via the existing stratum agent runner pinned to judgeModel.
10
+ * Any failure (LLM error, JSON parse, schema mismatch) degrades to null —
11
+ * a judge failure never aborts the experiment.
12
+ *
13
+ * COMP-MODEL-AB design: the judge model is held constant across all configs
14
+ * under test. Caller is responsible for not using a config's implementer as
15
+ * the judge model (bias guard — warn in the orchestrator, not enforced here).
16
+ */
17
+
18
+ import { resolveAgentConfig } from './agent-string.js';
19
+ import { injectSchema } from './inject-schema.js';
20
+
21
+ // ---------------------------------------------------------------------------
22
+ // Rubric schema injected into the judge prompt
23
+ // ---------------------------------------------------------------------------
24
+
25
+ const JUDGE_SCHEMA = {
26
+ type: 'object',
27
+ required: ['correctness', 'clarity', 'idiomaticity', 'rationale'],
28
+ properties: {
29
+ correctness: { type: 'integer', minimum: 1, maximum: 10 },
30
+ clarity: { type: 'integer', minimum: 1, maximum: 10 },
31
+ idiomaticity: { type: 'integer', minimum: 1, maximum: 10 },
32
+ rationale: { type: 'string' },
33
+ },
34
+ };
35
+
36
+ /**
37
+ * Build the judge prompt.
38
+ *
39
+ * @param {string} diff Full git diff of the build's produced changes.
40
+ * @param {string} goal The natural-language goal the build was given.
41
+ * @returns {string}
42
+ */
43
+ function buildJudgePrompt(diff, goal) {
44
+ const base = [
45
+ 'You are an expert code reviewer evaluating an AI-generated implementation.',
46
+ '',
47
+ `## Goal`,
48
+ goal,
49
+ '',
50
+ `## Implementation Diff`,
51
+ '```diff',
52
+ diff,
53
+ '```',
54
+ '',
55
+ 'Rate the implementation on three axes, each on a scale of 1–10:',
56
+ '',
57
+ '- **correctness** (1–10): Does the code correctly solve the stated goal?',
58
+ ' Consider: does it handle the described requirements, pass tests if present, and avoid obvious bugs?',
59
+ '- **clarity** (1–10): Is the code readable and well-structured?',
60
+ ' Consider: naming, comments, function decomposition, absence of unnecessary complexity.',
61
+ '- **idiomaticity** (1–10): Does the code follow language and project conventions?',
62
+ ' Consider: style, idiomatic patterns, appropriate use of language features.',
63
+ '',
64
+ 'Then provide a one-line rationale summarising your overall assessment.',
65
+ ].join('\n');
66
+
67
+ return injectSchema(base, JUDGE_SCHEMA);
68
+ }
69
+
70
+ /**
71
+ * Try to extract the judge result from agent text.
72
+ *
73
+ * @param {string} text
74
+ * @returns {{ correctness: number, clarity: number, idiomaticity: number, rationale: string }|null}
75
+ */
76
+ function extractJudgeResult(text) {
77
+ // Find last ```json ... ``` block
78
+ const matches = [...text.matchAll(/```json\s*([\s\S]*?)```/g)];
79
+ if (!matches.length) return null;
80
+ const lastJson = matches[matches.length - 1][1].trim();
81
+ let parsed;
82
+ try { parsed = JSON.parse(lastJson); } catch { return null; }
83
+
84
+ // Validate required fields
85
+ const { correctness, clarity, idiomaticity, rationale } = parsed;
86
+ if (
87
+ typeof correctness !== 'number' || correctness < 1 || correctness > 10 ||
88
+ typeof clarity !== 'number' || clarity < 1 || clarity > 10 ||
89
+ typeof idiomaticity !== 'number' || idiomaticity < 1 || idiomaticity > 10 ||
90
+ typeof rationale !== 'string'
91
+ ) {
92
+ return null;
93
+ }
94
+
95
+ return {
96
+ correctness: Math.round(correctness),
97
+ clarity: Math.round(clarity),
98
+ idiomaticity: Math.round(idiomaticity),
99
+ rationale: rationale.trim(),
100
+ };
101
+ }
102
+
103
+ // ---------------------------------------------------------------------------
104
+ // Public API
105
+ // ---------------------------------------------------------------------------
106
+
107
+ /**
108
+ * Run the LLM judge over a build's diff + goal.
109
+ *
110
+ * @param {object} args
111
+ * @param {string} args.diff Full git diff produced by the build.
112
+ * @param {string} args.goal Natural-language goal string.
113
+ * @param {string} args.judgeModel Agent string for the judge (e.g. "claude::critical").
114
+ * @param {object} args.stratum Connected StratumMcpClient.
115
+ * @param {string} [args.cwd] Working directory for the agent call.
116
+ *
117
+ * @returns {Promise<{ correctness: number, clarity: number, idiomaticity: number, rationale: string }|null>}
118
+ * Structured scores, or null on any failure (degrade, never throw).
119
+ */
120
+ export async function judge({ diff, goal, judgeModel, stratum, cwd }) {
121
+ try {
122
+ const prompt = buildJudgePrompt(diff, goal);
123
+ const { provider, modelID, thinking, effort } = resolveAgentConfig(judgeModel);
124
+
125
+ let text;
126
+ if (typeof stratum.agentRun === 'function') {
127
+ // Use agentRun so we can pin the concrete model ID resolved from the tier.
128
+ const result = await stratum.agentRun(provider, prompt, {
129
+ modelID: modelID ?? undefined,
130
+ thinking: thinking ?? undefined,
131
+ effort: effort ?? undefined,
132
+ cwd: cwd ?? undefined,
133
+ });
134
+ text = result?.text ?? '';
135
+ } else {
136
+ // Fallback: runAgentText (no model pinning — test harnesses may use this).
137
+ text = await stratum.runAgentText(provider, prompt, { cwd: cwd ?? undefined });
138
+ }
139
+
140
+ return extractJudgeResult(text);
141
+ } catch {
142
+ // Any failure — network error, LLM refusal, JSON parse — degrades to null.
143
+ return null;
144
+ }
145
+ }
@@ -0,0 +1,265 @@
1
+ /**
2
+ * experiment-metrics.js — Collect metrics from a completed sandbox run.
3
+ *
4
+ * COMP-MODEL-AB: four metric axes per run:
5
+ * cost — tokens in/out, call count, wall-clock, USD (derived via pricing table)
6
+ * outcome — completed, health score, test pass rate, files/lines changed
7
+ * process — review iterations, gate failures, retries, escalations
8
+ *
9
+ * Reads ONLY sandbox artifacts on disk; makes no LLM calls and runs no builds.
10
+ * A crashed build (exitCode ≠ 0, no history record) still yields a record with
11
+ * outcome.completed=false and whatever partial data exists.
12
+ */
13
+
14
+ import { readFileSync, existsSync } from 'node:fs';
15
+ import { join } from 'node:path';
16
+ import { execSync } from 'node:child_process';
17
+ import { deriveUsd } from './experiment-pricing.js';
18
+
19
+ // ---------------------------------------------------------------------------
20
+ // Helpers
21
+ // ---------------------------------------------------------------------------
22
+
23
+ /**
24
+ * Read the last record from a JSONL file (most recent build history entry).
25
+ * @param {string} filePath
26
+ * @returns {object|null}
27
+ */
28
+ function readLastJsonlRecord(filePath) {
29
+ if (!existsSync(filePath)) return null;
30
+ let raw;
31
+ try { raw = readFileSync(filePath, 'utf-8'); } catch { return null; }
32
+ const lines = raw.split('\n').filter(l => l.trim());
33
+ if (!lines.length) return null;
34
+ try { return JSON.parse(lines[lines.length - 1]); } catch { return null; }
35
+ }
36
+
37
+ /**
38
+ * Read all records from a JSONL file.
39
+ * @param {string} filePath
40
+ * @returns {object[]}
41
+ */
42
+ function readAllJsonlRecords(filePath) {
43
+ if (!existsSync(filePath)) return [];
44
+ let raw;
45
+ try { raw = readFileSync(filePath, 'utf-8'); } catch { return []; }
46
+ const records = [];
47
+ for (const line of raw.split('\n')) {
48
+ const t = line.trim();
49
+ if (!t) continue;
50
+ try { records.push(JSON.parse(t)); } catch { /* skip malformed */ }
51
+ }
52
+ return records;
53
+ }
54
+
55
+ /**
56
+ * Run `git diff --stat` in the workspace and parse files/lines changed.
57
+ * When baselineSha is provided, diffs against that commit so changes committed
58
+ * in-process by the real build (ship step) are captured rather than returning
59
+ * 0/0 from a clean post-commit working tree.
60
+ * Returns { filesChanged: 0, linesChanged: 0 } on any error.
61
+ *
62
+ * @param {string} workspace
63
+ * @param {string|null} [baselineSha] SHA of the pre-build baseline commit (fix #2)
64
+ * @returns {{ filesChanged: number, linesChanged: number }}
65
+ */
66
+ function gitDiffStat(workspace, baselineSha = null) {
67
+ try {
68
+ // Stage untracked files so new files created by the build appear in the stat.
69
+ // Idempotent — safe to call after executeRun already ran git add -A.
70
+ execSync('git add -A 2>/dev/null', { cwd: workspace, encoding: 'utf-8', timeout: 10_000 });
71
+ const diffCmd = baselineSha
72
+ ? `git diff --stat ${baselineSha} 2>/dev/null`
73
+ : 'git diff --stat HEAD 2>/dev/null';
74
+ const out = execSync(diffCmd, {
75
+ cwd: workspace,
76
+ encoding: 'utf-8',
77
+ timeout: 10_000,
78
+ });
79
+ // Last line: "N files changed, M insertions(+), K deletions(-)"
80
+ const summary = out.split('\n').filter(Boolean).pop() ?? '';
81
+ const filesMatch = summary.match(/(\d+)\s+file/);
82
+ const insertMatch = summary.match(/(\d+)\s+insertion/);
83
+ const deleteMatch = summary.match(/(\d+)\s+deletion/);
84
+ const filesChanged = filesMatch ? parseInt(filesMatch[1], 10) : 0;
85
+ const linesChanged = (insertMatch ? parseInt(insertMatch[1], 10) : 0)
86
+ + (deleteMatch ? parseInt(deleteMatch[1], 10) : 0);
87
+ return { filesChanged, linesChanged };
88
+ } catch {
89
+ return { filesChanged: 0, linesChanged: 0 };
90
+ }
91
+ }
92
+
93
+ /**
94
+ * Extract process friction metrics from build-stream.jsonl events.
95
+ *
96
+ * Signal mapping (verified against lib/build.js):
97
+ * retries — sum of `ev.retries` on `build_step_done` events (~line 1811).
98
+ * The build does NOT emit step_retry / build_retry events.
99
+ * gateFailures — count of `build_gate_resolved` events with outcome 'revise' or
100
+ * 'kill' (~lines 1877–1988). Auto-approvals (skip/flag policy modes)
101
+ * always emit outcome='approve' and are not counted. The old
102
+ * ensure_failed event does NOT exist in the real build stream.
103
+ * escalations — count of `build_error` events whose message matches /escalat/i
104
+ * (~line 1766). The 'escalation' event type is written only to the
105
+ * debug ledger, not the build stream.
106
+ * reviewIters — count of `build_step_done` events with stepId 'review' or
107
+ * 'codex_review' (unchanged — these stepIds do occur).
108
+ *
109
+ * @param {object[]} events All parsed build stream event objects
110
+ * @returns {{ reviewIters: number, gateFailures: number, retries: number, escalations: number }}
111
+ */
112
+ function parseProcessFromStream(events) {
113
+ let reviewIters = 0;
114
+ let gateFailures = 0;
115
+ let retries = 0;
116
+ let escalations = 0;
117
+
118
+ for (const ev of events) {
119
+ const type = ev?.type ?? ev?.kind;
120
+
121
+ // review step completions count as review iterations; retries are a per-step
122
+ // field on the same event, not a separate event type.
123
+ // v1 limitation: retries is only non-zero for top-level steps; child-flow
124
+ // steps and parallel dispatch completions always emit retries:0 in their
125
+ // build_step_done payloads, so the sum undercounts multi-flow runs.
126
+ if (type === 'build_step_done') {
127
+ const stepId = ev?.stepId ?? ev?.step_id ?? '';
128
+ if (stepId === 'review' || stepId === 'codex_review') reviewIters++;
129
+ retries += typeof ev?.retries === 'number' ? ev.retries : 0;
130
+ }
131
+
132
+ // Gate failures: human gates resolved as 'revise' (rejected, needs rework)
133
+ // or 'kill' (terminated). Policy-auto-approved gates emit outcome='approve'.
134
+ if (type === 'build_gate_resolved') {
135
+ const outcome = ev?.outcome ?? '';
136
+ if (outcome === 'revise' || outcome === 'kill') gateFailures++;
137
+ }
138
+
139
+ // Escalations are signalled via build_error (not a separate 'escalation' event).
140
+ if (type === 'build_error') {
141
+ if (/escalat/i.test(ev?.message ?? '')) escalations++;
142
+ }
143
+ }
144
+ return { reviewIters, gateFailures, retries, escalations };
145
+ }
146
+
147
+ /**
148
+ * Extract health score from build-stream events.
149
+ *
150
+ * @param {object[]} events
151
+ * @returns {number|null}
152
+ */
153
+ function parseHealthFromStream(events) {
154
+ for (let i = events.length - 1; i >= 0; i--) {
155
+ const ev = events[i];
156
+ if ((ev?.type === 'health_score' || ev?.kind === 'health_score') && typeof ev?.score === 'number') {
157
+ return ev.score;
158
+ }
159
+ // health_score embedded in kind/metadata envelope format
160
+ if (ev?.kind === 'health_score' && typeof ev?.metadata?.score === 'number') {
161
+ return ev.metadata.score;
162
+ }
163
+ }
164
+ return null;
165
+ }
166
+
167
+ // ---------------------------------------------------------------------------
168
+ // Public API
169
+ // ---------------------------------------------------------------------------
170
+
171
+ /**
172
+ * Collect all four metric axes from a sandbox's build artifacts.
173
+ *
174
+ * @param {object} args
175
+ * @param {{ workspace: string, runDir: string }} args.sandbox
176
+ * workspace — the git workspace dir (for git diff --stat)
177
+ * runDir — the run's output dir (contains manifest.json, build artifacts)
178
+ * @param {{ exitCode?: number, stdout?: string, wallMs?: number }} [args.buildResult]
179
+ * Optional output from the build process. `wallMs` is used as a cost fallback
180
+ * when the history record has no durationMs (e.g. crash before history write).
181
+ * `stdout` is retained for backward-compat but is no longer parsed for tests.
182
+ * @param {string|null} [args.baselineSha]
183
+ * SHA of the pre-build baseline commit. When provided, gitDiffStat diffs against
184
+ * this commit so in-process ship commits are counted (fix #2). Pass null / omit
185
+ * to fall back to `git diff --stat HEAD` (backward-compatible, greenfield fakes).
186
+ *
187
+ * @returns {{ cost: object, outcome: object, process: object }}
188
+ */
189
+ export function collect({ sandbox, buildResult = {}, baselineSha = null }) {
190
+ const dataDir = join(sandbox.workspace, '.compose', 'data');
191
+ const composeDir = join(sandbox.workspace, '.compose');
192
+
193
+ // ---------------------------------------------------------------------------
194
+ // 1. Build-history record (terminal record — most recent)
195
+ // ---------------------------------------------------------------------------
196
+ const historyRecord = readLastJsonlRecord(join(dataDir, 'build-history.jsonl'));
197
+
198
+ // ---------------------------------------------------------------------------
199
+ // 2. Build stream events
200
+ // ---------------------------------------------------------------------------
201
+ const streamEvents = readAllJsonlRecords(join(composeDir, 'build-stream.jsonl'));
202
+
203
+ // ---------------------------------------------------------------------------
204
+ // 3. Cost axis
205
+ // ---------------------------------------------------------------------------
206
+ const tokensIn = historyRecord?.input_tokens ?? 0;
207
+ const tokensOut = historyRecord?.output_tokens ?? 0;
208
+ // calls = stepCount (step records in history, including gate records). This is
209
+ // NOT raw model invocations — one step can make multiple LLM calls internally.
210
+ const calls = historyRecord?.stepCount ?? 0;
211
+ const wallMs = historyRecord?.durationMs ?? (buildResult?.wallMs ?? 0);
212
+
213
+ // Try to derive USD from the history record's embedded cost first.
214
+ // Fall back to deriveUsd with a model from the stream if needed.
215
+ let usd = null;
216
+ if (typeof historyRecord?.cost_usd === 'number') {
217
+ usd = historyRecord.cost_usd;
218
+ } else {
219
+ // Find any step_model event to get the model ID for pricing
220
+ const modelEv = streamEvents.find(
221
+ e => (e?.type === 'step_model' || e?.kind === 'step_model') && (e?.modelID ?? e?.metadata?.modelID)
222
+ );
223
+ const modelID = modelEv?.modelID ?? modelEv?.metadata?.modelID ?? null;
224
+ if (modelID) usd = deriveUsd(modelID, tokensIn, tokensOut);
225
+ }
226
+
227
+ const cost = { tokensIn, tokensOut, calls, wallMs, usd };
228
+
229
+ // ---------------------------------------------------------------------------
230
+ // 4. Outcome axis
231
+ // ---------------------------------------------------------------------------
232
+ const completed = historyRecord?.status === 'complete';
233
+ const health = parseHealthFromStream(streamEvents);
234
+
235
+ // Test pass rate: the build's ship step runs tests via execSync and persists
236
+ // structured counts to build-history.jsonl as test_count and pass_rate fields
237
+ // (COMP-MODEL-AB fix B). Read directly from the history record — the ship step's
238
+ // execSync output never reaches realHeadlessBuild's stdout buffer (only the
239
+ // outer `node compose build` process's own stdout is captured there).
240
+ //
241
+ // v1 limitation: testsTotal/testsPass are null on failed, aborted, or thrown
242
+ // builds even if tests ran before the failure. _extractShipTestMetrics only
243
+ // fires on the success path (ship step completes + testSummary.parsed=true);
244
+ // terminalizeThrownBuild and the early-abort path never carry test_count/pass_rate.
245
+ // outcome.completed=false already signals the failure; null test metrics on those
246
+ // paths are expected and should not be treated as a data gap.
247
+ const testsTotal = historyRecord?.test_count ?? null;
248
+ let testsPass = null;
249
+ if (testsTotal !== null && historyRecord?.pass_rate != null) {
250
+ testsPass = historyRecord.pass_rate === 100
251
+ ? testsTotal
252
+ : Math.round((historyRecord.pass_rate / 100) * testsTotal);
253
+ }
254
+
255
+ const { filesChanged, linesChanged } = gitDiffStat(sandbox.workspace, baselineSha);
256
+
257
+ const outcome = { completed, health, testsPass, testsTotal, filesChanged, linesChanged };
258
+
259
+ // ---------------------------------------------------------------------------
260
+ // 5. Process axis
261
+ // ---------------------------------------------------------------------------
262
+ const process = parseProcessFromStream(streamEvents);
263
+
264
+ return { cost, outcome, process };
265
+ }
@@ -0,0 +1,60 @@
1
+ /**
2
+ * experiment-pricing.js — Static model→$/MTok table for COMP-MODEL-AB.
3
+ *
4
+ * Used by experiment-metrics.js to derive a USD cost estimate from raw token
5
+ * counts when build artifacts don't already carry a cost field. Unknown
6
+ * model IDs degrade to usd:null rather than crashing — a crashed / future
7
+ * model still yields a record with partial metrics.
8
+ *
9
+ * Price source: Anthropic public pricing page + OpenAI pricing (as of 2026-06).
10
+ * Keys are prefix-matched so dated variants (e.g. claude-sonnet-4-6-20250514)
11
+ * resolve against the base key.
12
+ */
13
+
14
+ /** @type {Record<string, { inputPerMTok: number, outputPerMTok: number }>} */
15
+ const EXPERIMENT_PRICING = {
16
+ // Claude 4.x
17
+ 'claude-opus-4-8': { inputPerMTok: 5, outputPerMTok: 25 },
18
+ 'claude-opus-4-7': { inputPerMTok: 5, outputPerMTok: 25 },
19
+ 'claude-opus-4-6': { inputPerMTok: 5, outputPerMTok: 25 },
20
+ 'claude-sonnet-4-6': { inputPerMTok: 3, outputPerMTok: 15 },
21
+ 'claude-haiku-4-5': { inputPerMTok: 1, outputPerMTok: 5 },
22
+ // GPT / Codex
23
+ 'gpt-5': { inputPerMTok: 10, outputPerMTok: 40 },
24
+ 'gpt-5.4': { inputPerMTok: 10, outputPerMTok: 40 },
25
+ 'gpt-4.1': { inputPerMTok: 2, outputPerMTok: 8 },
26
+ 'gpt-4o': { inputPerMTok: 2.5, outputPerMTok: 10 },
27
+ 'o3': { inputPerMTok: 10, outputPerMTok: 40 },
28
+ 'o4-mini': { inputPerMTok: 1.1, outputPerMTok: 4.4 },
29
+ };
30
+
31
+ /**
32
+ * Look up pricing for a model ID by exact match then prefix match.
33
+ *
34
+ * @param {string|null|undefined} modelID
35
+ * @returns {{ inputPerMTok: number, outputPerMTok: number } | null}
36
+ */
37
+ export function lookupExperimentPricing(modelID) {
38
+ if (!modelID) return null;
39
+ if (EXPERIMENT_PRICING[modelID]) return EXPERIMENT_PRICING[modelID];
40
+ for (const [key, pricing] of Object.entries(EXPERIMENT_PRICING)) {
41
+ if (modelID.startsWith(key)) return pricing;
42
+ }
43
+ return null;
44
+ }
45
+
46
+ /**
47
+ * Derive a USD cost from token counts using the static pricing table.
48
+ *
49
+ * @param {string|null|undefined} modelID
50
+ * @param {number} tokensIn
51
+ * @param {number} tokensOut
52
+ * @returns {number|null} USD cost, or null for unknown models
53
+ */
54
+ export function deriveUsd(modelID, tokensIn, tokensOut) {
55
+ const pricing = lookupExperimentPricing(modelID);
56
+ if (!pricing) return null;
57
+ const inputCost = ((tokensIn ?? 0) / 1_000_000) * pricing.inputPerMTok;
58
+ const outputCost = ((tokensOut ?? 0) / 1_000_000) * pricing.outputPerMTok;
59
+ return inputCost + outputCost;
60
+ }