@smartmemory/compose 0.2.51 → 0.2.53-beta
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude/skills/compose/SKILL.md +0 -8
- package/README.md +11 -0
- package/bin/compose.js +335 -3
- package/dist/assets/App-Bu9KMtTa.js +891 -0
- package/dist/assets/abnfDiagram-VRR7QNED-6zp3w9rx.js +1 -0
- package/dist/assets/arc-CnLxxuah.js +1 -0
- package/dist/assets/architectureDiagram-ZJ3FMSHR-COM46S9k.js +36 -0
- package/dist/assets/blockDiagram-677ZJIJ3-wKzgOwF8.js +132 -0
- package/dist/assets/{browser-CnKiSnlr.js → browser-Cr0recrN.js} +6 -6
- package/dist/assets/{c4Diagram-AHTNJAMY-nBIRLJTv.js → c4Diagram-LMCZKHZV-DIXC_mR6.js} +1 -1
- package/dist/assets/channel-DQJPWkb_.js +1 -0
- package/dist/assets/{chunk-QZHKN3VN-B4Wfybww.js → chunk-2Q5K7J3B-BqLqZX6m.js} +1 -1
- package/dist/assets/{chunk-YZCP3GAM-DE6_5aqh.js → chunk-32BRIVSS-CarSmVlX.js} +1 -1
- package/dist/assets/{chunk-FMBD7UC4-tYY8xcjr.js → chunk-5VM5RSS4-D-f_Jn3G.js} +1 -1
- package/dist/assets/chunk-EX3LRPZG-DT42xo8E.js +231 -0
- package/dist/assets/{chunk-4BX2VUAB-CNmGyhrp.js → chunk-JWPE2WC7-CwY1aZc4.js} +1 -1
- package/dist/assets/chunk-MOJQB5TN-DBbZU_KX.js +88 -0
- package/dist/assets/chunk-RYQCIY6F-DqxtcLJ3.js +1 -0
- package/dist/assets/chunk-V7JOEXUC-Cy4Ixww6.js +206 -0
- package/dist/assets/{chunk-EDXVE4YY-TPlt1bS2.js → chunk-VR4S4FIN-CFexfU_a.js} +1 -1
- package/dist/assets/{chunk-55IACEB6-D8KuKVLa.js → chunk-XXDRQBXY-uE-zN_vc.js} +1 -1
- package/dist/assets/classDiagram-OUVF2IWQ-CTLbAiUK.js +1 -0
- package/dist/assets/classDiagram-v2-EOCWNBFH-CTLbAiUK.js +1 -0
- package/dist/assets/{cose-bilkent-S5V4N54A-DzzRyJtB.js → cose-bilkent-JH36ORCC-DIjpekos.js} +1 -1
- package/dist/assets/cynefin-VYW2F7L2-Ba2gbQow.js +178 -0
- package/dist/assets/cynefinDiagram-TSTJHNR4-njGzb7Tg.js +62 -0
- package/dist/assets/dagre-VKFMJZFB-DuyZREZ3.js +4 -0
- package/dist/assets/diagram-FQU43EPY-Npq-5o3a.js +3 -0
- package/dist/assets/diagram-G47NLZAW-fm4k3axC.js +24 -0
- package/dist/assets/diagram-NH7WQ7WH-B8EFGrTG.js +24 -0
- package/dist/assets/diagram-OA4YK3LP-BC7UHx9Q.js +30 -0
- package/dist/assets/diagram-WEI45ONY-BqjHLcCX.js +41 -0
- package/dist/assets/ebnfDiagram-CCIWWBDH-D5zXd4Qg.js +1 -0
- package/dist/assets/erDiagram-Q63AITRT-DC2FMcra.js +85 -0
- package/dist/assets/flowDiagram-23GEKE2U-D7Dx7JU4.js +156 -0
- package/dist/assets/ganttDiagram-NO4QXBWP-8sQH2y5K.js +292 -0
- package/dist/assets/gitGraphDiagram-IHSO6WYX-Ci_rhxur.js +106 -0
- package/dist/assets/graph-C9eacEi8.js +1 -0
- package/dist/assets/graph-xkel59g2.js +331 -0
- package/dist/assets/index-B7-HQenC.js +119 -0
- package/dist/assets/index-BxRamj_i.css +1 -0
- package/dist/assets/infoDiagram-FWYZ7A6U-B9FGQ0Cb.js +2 -0
- package/dist/assets/{ishikawaDiagram-UXIWVN3A-DWx-7BBy.js → ishikawaDiagram-FXEZZL3T-BEKLyH6A.js} +5 -5
- package/dist/assets/{journeyDiagram-VCZTEJTY-C42hXNha.js → journeyDiagram-5HDEW3XC-BM-IDVRo.js} +1 -1
- package/dist/assets/{kanban-definition-6JOO6SKY-CEq930ew.js → kanban-definition-HUTT4EX6-Beb2k3pt.js} +7 -7
- package/dist/assets/katex-C5jXJg4s.js +257 -0
- package/dist/assets/layout-DEXfKzaS.js +1 -0
- package/dist/assets/{linear-D158F7OT.js → linear-Dwg7dTpz.js} +1 -1
- package/dist/assets/map-Czzmt4hB.js +1 -0
- package/dist/assets/{mindmap-definition-QFDTVHPH-d9SSh2nG.js → mindmap-definition-LN4V7U3C-CeaynfcE.js} +7 -7
- package/dist/assets/{mobile-Chw8RWyH.js → mobile-CNLMhdFP.js} +2 -2
- package/dist/assets/pegDiagram-2B236MQR-7SVdAVqj.js +1 -0
- package/dist/assets/pieDiagram-ENE6RG2P-CzqQ4TaE.js +39 -0
- package/dist/assets/quadrantDiagram-ABIIQ3AL-B9B6W80N.js +7 -0
- package/dist/assets/railroadDiagram-RFXS5EU6-BRNLawsr.js +1 -0
- package/dist/assets/{requirementDiagram-MS252O5E-B1ZFIotK.js → requirementDiagram-TGXJPOKE-CmASubeG.js} +3 -3
- package/dist/assets/sankeyDiagram-HTMAVEWB-DiReR0OB.js +40 -0
- package/dist/assets/sequenceDiagram-DBY2YBRQ-DVn5iZKR.js +162 -0
- package/dist/assets/sizeCapture-X5ZJPWSS-C8CuDOdp.js +1 -0
- package/dist/assets/stateDiagram-2N3HPSRC-BIdkHglY.js +1 -0
- package/dist/assets/stateDiagram-v2-6OUMAXLB-B9yje3Er.js +1 -0
- package/dist/assets/swimlanes-5IMT3BWC-BlVsYyyN.js +2 -0
- package/dist/assets/swimlanesDiagram-G3AALYLV-BTLpU45n.js +8 -0
- package/dist/assets/{timeline-definition-GMOUNBTQ-k_0vJQAK.js → timeline-definition-FHXFAJF6-CnTqJmQ2.js} +3 -3
- package/dist/assets/vennDiagram-L72KCM5P-D_qZxRhe.js +34 -0
- package/dist/assets/wardleyDiagram-EHGQE667-CT-WHFEO.js +78 -0
- package/dist/assets/xychartDiagram-FW5EYKEG-CVRFHwUQ.js +7 -0
- package/dist/index.html +3 -3
- package/lib/agent-string.js +34 -0
- package/lib/build.js +542 -148
- package/lib/experiment-judge.js +145 -0
- package/lib/experiment-metrics.js +265 -0
- package/lib/experiment-pricing.js +60 -0
- package/lib/experiment-report.js +306 -0
- package/lib/experiment-sandbox.js +147 -0
- package/lib/experiment.js +539 -0
- package/lib/feature-json.js +1 -1
- package/lib/feature-writer.js +30 -0
- package/lib/flow-state.js +36 -0
- package/lib/gate-prompt.js +66 -6
- package/lib/lifecycle-modes.js +213 -0
- package/lib/new.js +6 -2
- package/lib/roadmap-graph/index.js +26 -28
- package/lib/roadmap-graph/vision-adapter.js +126 -0
- package/lib/stratum-mcp-client.js +9 -0
- package/lib/triage.js +7 -1
- package/lib/vision-writer.js +30 -6
- package/package.json +1 -1
- package/pipelines/plan.stratum.yaml +161 -0
- package/scripts/release-manual.sh +98 -0
- package/server/artifact-manager.js +30 -3
- package/server/compose-mcp-tools.js +11 -9
- package/server/compose-mcp.js +4 -0
- package/server/feature-scan.js +87 -173
- package/server/file-watcher.js +33 -0
- package/server/graph-export.js +0 -0
- package/server/index.js +10 -5
- package/server/lifecycle-guard.js +62 -25
- package/server/pipeline-routes.js +773 -3
- package/server/roadmap-graph-vision.js +138 -0
- package/server/status-snapshot.js +8 -7
- package/server/vision-routes.js +64 -33
- package/server/vision-store.js +13 -4
- package/.claude/skills/compose/references/hermes-tools.md +0 -80
- package/dist/assets/App-CT0vXPgd.js +0 -724
- package/dist/assets/_baseUniq-CsOHc_iS.js +0 -1
- package/dist/assets/arc-BXht3LyE.js +0 -1
- package/dist/assets/architectureDiagram-Q4EWVU46-BV1r86-w.js +0 -36
- package/dist/assets/blockDiagram-DXYQGD6D-BgcMJjAR.js +0 -132
- package/dist/assets/channel-Dd4XaiYv.js +0 -1
- package/dist/assets/chunk-4TB4RGXK-MLT7I7w4.js +0 -206
- package/dist/assets/chunk-OYMX7WX6-yBPwT1AT.js +0 -231
- package/dist/assets/classDiagram-6PBFFD2Q-DshSQDF9.js +0 -1
- package/dist/assets/classDiagram-v2-HSJHXN6E-DshSQDF9.js +0 -1
- package/dist/assets/clone-CBsbNGAa.js +0 -1
- package/dist/assets/dagre-KV5264BT-CTcTVGdU.js +0 -4
- package/dist/assets/diagram-5BDNPKRD-qR6mc8kJ.js +0 -10
- package/dist/assets/diagram-G4DWMVQ6-CUcGXyNb.js +0 -24
- package/dist/assets/diagram-MMDJMWI5-B68cPonH.js +0 -43
- package/dist/assets/diagram-TYMM5635-q8Id8eFG.js +0 -24
- package/dist/assets/erDiagram-SMLLAGMA-CtxJi57K.js +0 -85
- package/dist/assets/flowDiagram-DWJPFMVM-BzzO4JYU.js +0 -162
- package/dist/assets/ganttDiagram-T4ZO3ILL-Dn0wIyhR.js +0 -292
- package/dist/assets/gitGraphDiagram-UUTBAWPF-tnNFXfMM.js +0 -106
- package/dist/assets/graph-DZe55uk8.js +0 -331
- package/dist/assets/graph-Tq_bs_r0.js +0 -1
- package/dist/assets/index-8UhRLbGq.js +0 -123
- package/dist/assets/index-CRqB9els.css +0 -1
- package/dist/assets/infoDiagram-42DDH7IO-BXLvLeS0.js +0 -2
- package/dist/assets/katex-DkKDou_j.js +0 -257
- package/dist/assets/layout-qtUgN9BC.js +0 -1
- package/dist/assets/min-hESxN-c0.js +0 -1
- package/dist/assets/pieDiagram-DEJITSTG-gglmeMav.js +0 -30
- package/dist/assets/quadrantDiagram-34T5L4WZ-CqFC_htU.js +0 -7
- package/dist/assets/sankeyDiagram-XADWPNL6-BvItvZt5.js +0 -10
- package/dist/assets/sequenceDiagram-FGHM5R23-CAIn5Z3U.js +0 -157
- package/dist/assets/stateDiagram-FHFEXIEX-CyJrGzib.js +0 -1
- package/dist/assets/stateDiagram-v2-QKLJ7IA2-DprhF_Ho.js +0 -1
- package/dist/assets/vennDiagram-DHZGUBPP-TZAW60b0.js +0 -34
- package/dist/assets/wardley-RL74JXVD-DzQEoG6X.js +0 -162
- package/dist/assets/wardleyDiagram-NUSXRM2D-BAoVVIjZ.js +0 -20
- package/dist/assets/xychartDiagram-5P7HB3ND-CynkZcUb.js +0 -7
- package/lib/roadmap-graph/collect.js +0 -178
|
@@ -0,0 +1,145 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* experiment-judge.js — LLM-judge rubric for COMP-MODEL-AB.
|
|
3
|
+
*
|
|
4
|
+
* Rates a build's produced diff against the goal on three axes (1–10 each):
|
|
5
|
+
* correctness — does the code solve the stated goal?
|
|
6
|
+
* clarity — is the code clear and idiomatic?
|
|
7
|
+
* idiomaticity — does it follow language/project conventions?
|
|
8
|
+
*
|
|
9
|
+
* Dispatched via the existing stratum agent runner pinned to judgeModel.
|
|
10
|
+
* Any failure (LLM error, JSON parse, schema mismatch) degrades to null —
|
|
11
|
+
* a judge failure never aborts the experiment.
|
|
12
|
+
*
|
|
13
|
+
* COMP-MODEL-AB design: the judge model is held constant across all configs
|
|
14
|
+
* under test. Caller is responsible for not using a config's implementer as
|
|
15
|
+
* the judge model (bias guard — warn in the orchestrator, not enforced here).
|
|
16
|
+
*/
|
|
17
|
+
|
|
18
|
+
import { resolveAgentConfig } from './agent-string.js';
|
|
19
|
+
import { injectSchema } from './inject-schema.js';
|
|
20
|
+
|
|
21
|
+
// ---------------------------------------------------------------------------
|
|
22
|
+
// Rubric schema injected into the judge prompt
|
|
23
|
+
// ---------------------------------------------------------------------------
|
|
24
|
+
|
|
25
|
+
const JUDGE_SCHEMA = {
|
|
26
|
+
type: 'object',
|
|
27
|
+
required: ['correctness', 'clarity', 'idiomaticity', 'rationale'],
|
|
28
|
+
properties: {
|
|
29
|
+
correctness: { type: 'integer', minimum: 1, maximum: 10 },
|
|
30
|
+
clarity: { type: 'integer', minimum: 1, maximum: 10 },
|
|
31
|
+
idiomaticity: { type: 'integer', minimum: 1, maximum: 10 },
|
|
32
|
+
rationale: { type: 'string' },
|
|
33
|
+
},
|
|
34
|
+
};
|
|
35
|
+
|
|
36
|
+
/**
|
|
37
|
+
* Build the judge prompt.
|
|
38
|
+
*
|
|
39
|
+
* @param {string} diff Full git diff of the build's produced changes.
|
|
40
|
+
* @param {string} goal The natural-language goal the build was given.
|
|
41
|
+
* @returns {string}
|
|
42
|
+
*/
|
|
43
|
+
function buildJudgePrompt(diff, goal) {
|
|
44
|
+
const base = [
|
|
45
|
+
'You are an expert code reviewer evaluating an AI-generated implementation.',
|
|
46
|
+
'',
|
|
47
|
+
`## Goal`,
|
|
48
|
+
goal,
|
|
49
|
+
'',
|
|
50
|
+
`## Implementation Diff`,
|
|
51
|
+
'```diff',
|
|
52
|
+
diff,
|
|
53
|
+
'```',
|
|
54
|
+
'',
|
|
55
|
+
'Rate the implementation on three axes, each on a scale of 1–10:',
|
|
56
|
+
'',
|
|
57
|
+
'- **correctness** (1–10): Does the code correctly solve the stated goal?',
|
|
58
|
+
' Consider: does it handle the described requirements, pass tests if present, and avoid obvious bugs?',
|
|
59
|
+
'- **clarity** (1–10): Is the code readable and well-structured?',
|
|
60
|
+
' Consider: naming, comments, function decomposition, absence of unnecessary complexity.',
|
|
61
|
+
'- **idiomaticity** (1–10): Does the code follow language and project conventions?',
|
|
62
|
+
' Consider: style, idiomatic patterns, appropriate use of language features.',
|
|
63
|
+
'',
|
|
64
|
+
'Then provide a one-line rationale summarising your overall assessment.',
|
|
65
|
+
].join('\n');
|
|
66
|
+
|
|
67
|
+
return injectSchema(base, JUDGE_SCHEMA);
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
/**
|
|
71
|
+
* Try to extract the judge result from agent text.
|
|
72
|
+
*
|
|
73
|
+
* @param {string} text
|
|
74
|
+
* @returns {{ correctness: number, clarity: number, idiomaticity: number, rationale: string }|null}
|
|
75
|
+
*/
|
|
76
|
+
function extractJudgeResult(text) {
|
|
77
|
+
// Find last ```json ... ``` block
|
|
78
|
+
const matches = [...text.matchAll(/```json\s*([\s\S]*?)```/g)];
|
|
79
|
+
if (!matches.length) return null;
|
|
80
|
+
const lastJson = matches[matches.length - 1][1].trim();
|
|
81
|
+
let parsed;
|
|
82
|
+
try { parsed = JSON.parse(lastJson); } catch { return null; }
|
|
83
|
+
|
|
84
|
+
// Validate required fields
|
|
85
|
+
const { correctness, clarity, idiomaticity, rationale } = parsed;
|
|
86
|
+
if (
|
|
87
|
+
typeof correctness !== 'number' || correctness < 1 || correctness > 10 ||
|
|
88
|
+
typeof clarity !== 'number' || clarity < 1 || clarity > 10 ||
|
|
89
|
+
typeof idiomaticity !== 'number' || idiomaticity < 1 || idiomaticity > 10 ||
|
|
90
|
+
typeof rationale !== 'string'
|
|
91
|
+
) {
|
|
92
|
+
return null;
|
|
93
|
+
}
|
|
94
|
+
|
|
95
|
+
return {
|
|
96
|
+
correctness: Math.round(correctness),
|
|
97
|
+
clarity: Math.round(clarity),
|
|
98
|
+
idiomaticity: Math.round(idiomaticity),
|
|
99
|
+
rationale: rationale.trim(),
|
|
100
|
+
};
|
|
101
|
+
}
|
|
102
|
+
|
|
103
|
+
// ---------------------------------------------------------------------------
|
|
104
|
+
// Public API
|
|
105
|
+
// ---------------------------------------------------------------------------
|
|
106
|
+
|
|
107
|
+
/**
|
|
108
|
+
* Run the LLM judge over a build's diff + goal.
|
|
109
|
+
*
|
|
110
|
+
* @param {object} args
|
|
111
|
+
* @param {string} args.diff Full git diff produced by the build.
|
|
112
|
+
* @param {string} args.goal Natural-language goal string.
|
|
113
|
+
* @param {string} args.judgeModel Agent string for the judge (e.g. "claude::critical").
|
|
114
|
+
* @param {object} args.stratum Connected StratumMcpClient.
|
|
115
|
+
* @param {string} [args.cwd] Working directory for the agent call.
|
|
116
|
+
*
|
|
117
|
+
* @returns {Promise<{ correctness: number, clarity: number, idiomaticity: number, rationale: string }|null>}
|
|
118
|
+
* Structured scores, or null on any failure (degrade, never throw).
|
|
119
|
+
*/
|
|
120
|
+
export async function judge({ diff, goal, judgeModel, stratum, cwd }) {
|
|
121
|
+
try {
|
|
122
|
+
const prompt = buildJudgePrompt(diff, goal);
|
|
123
|
+
const { provider, modelID, thinking, effort } = resolveAgentConfig(judgeModel);
|
|
124
|
+
|
|
125
|
+
let text;
|
|
126
|
+
if (typeof stratum.agentRun === 'function') {
|
|
127
|
+
// Use agentRun so we can pin the concrete model ID resolved from the tier.
|
|
128
|
+
const result = await stratum.agentRun(provider, prompt, {
|
|
129
|
+
modelID: modelID ?? undefined,
|
|
130
|
+
thinking: thinking ?? undefined,
|
|
131
|
+
effort: effort ?? undefined,
|
|
132
|
+
cwd: cwd ?? undefined,
|
|
133
|
+
});
|
|
134
|
+
text = result?.text ?? '';
|
|
135
|
+
} else {
|
|
136
|
+
// Fallback: runAgentText (no model pinning — test harnesses may use this).
|
|
137
|
+
text = await stratum.runAgentText(provider, prompt, { cwd: cwd ?? undefined });
|
|
138
|
+
}
|
|
139
|
+
|
|
140
|
+
return extractJudgeResult(text);
|
|
141
|
+
} catch {
|
|
142
|
+
// Any failure — network error, LLM refusal, JSON parse — degrades to null.
|
|
143
|
+
return null;
|
|
144
|
+
}
|
|
145
|
+
}
|
|
@@ -0,0 +1,265 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* experiment-metrics.js — Collect metrics from a completed sandbox run.
|
|
3
|
+
*
|
|
4
|
+
* COMP-MODEL-AB: four metric axes per run:
|
|
5
|
+
* cost — tokens in/out, call count, wall-clock, USD (derived via pricing table)
|
|
6
|
+
* outcome — completed, health score, test pass rate, files/lines changed
|
|
7
|
+
* process — review iterations, gate failures, retries, escalations
|
|
8
|
+
*
|
|
9
|
+
* Reads ONLY sandbox artifacts on disk; makes no LLM calls and runs no builds.
|
|
10
|
+
* A crashed build (exitCode ≠ 0, no history record) still yields a record with
|
|
11
|
+
* outcome.completed=false and whatever partial data exists.
|
|
12
|
+
*/
|
|
13
|
+
|
|
14
|
+
import { readFileSync, existsSync } from 'node:fs';
|
|
15
|
+
import { join } from 'node:path';
|
|
16
|
+
import { execSync } from 'node:child_process';
|
|
17
|
+
import { deriveUsd } from './experiment-pricing.js';
|
|
18
|
+
|
|
19
|
+
// ---------------------------------------------------------------------------
|
|
20
|
+
// Helpers
|
|
21
|
+
// ---------------------------------------------------------------------------
|
|
22
|
+
|
|
23
|
+
/**
|
|
24
|
+
* Read the last record from a JSONL file (most recent build history entry).
|
|
25
|
+
* @param {string} filePath
|
|
26
|
+
* @returns {object|null}
|
|
27
|
+
*/
|
|
28
|
+
function readLastJsonlRecord(filePath) {
|
|
29
|
+
if (!existsSync(filePath)) return null;
|
|
30
|
+
let raw;
|
|
31
|
+
try { raw = readFileSync(filePath, 'utf-8'); } catch { return null; }
|
|
32
|
+
const lines = raw.split('\n').filter(l => l.trim());
|
|
33
|
+
if (!lines.length) return null;
|
|
34
|
+
try { return JSON.parse(lines[lines.length - 1]); } catch { return null; }
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
/**
|
|
38
|
+
* Read all records from a JSONL file.
|
|
39
|
+
* @param {string} filePath
|
|
40
|
+
* @returns {object[]}
|
|
41
|
+
*/
|
|
42
|
+
function readAllJsonlRecords(filePath) {
|
|
43
|
+
if (!existsSync(filePath)) return [];
|
|
44
|
+
let raw;
|
|
45
|
+
try { raw = readFileSync(filePath, 'utf-8'); } catch { return []; }
|
|
46
|
+
const records = [];
|
|
47
|
+
for (const line of raw.split('\n')) {
|
|
48
|
+
const t = line.trim();
|
|
49
|
+
if (!t) continue;
|
|
50
|
+
try { records.push(JSON.parse(t)); } catch { /* skip malformed */ }
|
|
51
|
+
}
|
|
52
|
+
return records;
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
/**
|
|
56
|
+
* Run `git diff --stat` in the workspace and parse files/lines changed.
|
|
57
|
+
* When baselineSha is provided, diffs against that commit so changes committed
|
|
58
|
+
* in-process by the real build (ship step) are captured rather than returning
|
|
59
|
+
* 0/0 from a clean post-commit working tree.
|
|
60
|
+
* Returns { filesChanged: 0, linesChanged: 0 } on any error.
|
|
61
|
+
*
|
|
62
|
+
* @param {string} workspace
|
|
63
|
+
* @param {string|null} [baselineSha] SHA of the pre-build baseline commit (fix #2)
|
|
64
|
+
* @returns {{ filesChanged: number, linesChanged: number }}
|
|
65
|
+
*/
|
|
66
|
+
function gitDiffStat(workspace, baselineSha = null) {
|
|
67
|
+
try {
|
|
68
|
+
// Stage untracked files so new files created by the build appear in the stat.
|
|
69
|
+
// Idempotent — safe to call after executeRun already ran git add -A.
|
|
70
|
+
execSync('git add -A 2>/dev/null', { cwd: workspace, encoding: 'utf-8', timeout: 10_000 });
|
|
71
|
+
const diffCmd = baselineSha
|
|
72
|
+
? `git diff --stat ${baselineSha} 2>/dev/null`
|
|
73
|
+
: 'git diff --stat HEAD 2>/dev/null';
|
|
74
|
+
const out = execSync(diffCmd, {
|
|
75
|
+
cwd: workspace,
|
|
76
|
+
encoding: 'utf-8',
|
|
77
|
+
timeout: 10_000,
|
|
78
|
+
});
|
|
79
|
+
// Last line: "N files changed, M insertions(+), K deletions(-)"
|
|
80
|
+
const summary = out.split('\n').filter(Boolean).pop() ?? '';
|
|
81
|
+
const filesMatch = summary.match(/(\d+)\s+file/);
|
|
82
|
+
const insertMatch = summary.match(/(\d+)\s+insertion/);
|
|
83
|
+
const deleteMatch = summary.match(/(\d+)\s+deletion/);
|
|
84
|
+
const filesChanged = filesMatch ? parseInt(filesMatch[1], 10) : 0;
|
|
85
|
+
const linesChanged = (insertMatch ? parseInt(insertMatch[1], 10) : 0)
|
|
86
|
+
+ (deleteMatch ? parseInt(deleteMatch[1], 10) : 0);
|
|
87
|
+
return { filesChanged, linesChanged };
|
|
88
|
+
} catch {
|
|
89
|
+
return { filesChanged: 0, linesChanged: 0 };
|
|
90
|
+
}
|
|
91
|
+
}
|
|
92
|
+
|
|
93
|
+
/**
|
|
94
|
+
* Extract process friction metrics from build-stream.jsonl events.
|
|
95
|
+
*
|
|
96
|
+
* Signal mapping (verified against lib/build.js):
|
|
97
|
+
* retries — sum of `ev.retries` on `build_step_done` events (~line 1811).
|
|
98
|
+
* The build does NOT emit step_retry / build_retry events.
|
|
99
|
+
* gateFailures — count of `build_gate_resolved` events with outcome 'revise' or
|
|
100
|
+
* 'kill' (~lines 1877–1988). Auto-approvals (skip/flag policy modes)
|
|
101
|
+
* always emit outcome='approve' and are not counted. The old
|
|
102
|
+
* ensure_failed event does NOT exist in the real build stream.
|
|
103
|
+
* escalations — count of `build_error` events whose message matches /escalat/i
|
|
104
|
+
* (~line 1766). The 'escalation' event type is written only to the
|
|
105
|
+
* debug ledger, not the build stream.
|
|
106
|
+
* reviewIters — count of `build_step_done` events with stepId 'review' or
|
|
107
|
+
* 'codex_review' (unchanged — these stepIds do occur).
|
|
108
|
+
*
|
|
109
|
+
* @param {object[]} events All parsed build stream event objects
|
|
110
|
+
* @returns {{ reviewIters: number, gateFailures: number, retries: number, escalations: number }}
|
|
111
|
+
*/
|
|
112
|
+
function parseProcessFromStream(events) {
|
|
113
|
+
let reviewIters = 0;
|
|
114
|
+
let gateFailures = 0;
|
|
115
|
+
let retries = 0;
|
|
116
|
+
let escalations = 0;
|
|
117
|
+
|
|
118
|
+
for (const ev of events) {
|
|
119
|
+
const type = ev?.type ?? ev?.kind;
|
|
120
|
+
|
|
121
|
+
// review step completions count as review iterations; retries are a per-step
|
|
122
|
+
// field on the same event, not a separate event type.
|
|
123
|
+
// v1 limitation: retries is only non-zero for top-level steps; child-flow
|
|
124
|
+
// steps and parallel dispatch completions always emit retries:0 in their
|
|
125
|
+
// build_step_done payloads, so the sum undercounts multi-flow runs.
|
|
126
|
+
if (type === 'build_step_done') {
|
|
127
|
+
const stepId = ev?.stepId ?? ev?.step_id ?? '';
|
|
128
|
+
if (stepId === 'review' || stepId === 'codex_review') reviewIters++;
|
|
129
|
+
retries += typeof ev?.retries === 'number' ? ev.retries : 0;
|
|
130
|
+
}
|
|
131
|
+
|
|
132
|
+
// Gate failures: human gates resolved as 'revise' (rejected, needs rework)
|
|
133
|
+
// or 'kill' (terminated). Policy-auto-approved gates emit outcome='approve'.
|
|
134
|
+
if (type === 'build_gate_resolved') {
|
|
135
|
+
const outcome = ev?.outcome ?? '';
|
|
136
|
+
if (outcome === 'revise' || outcome === 'kill') gateFailures++;
|
|
137
|
+
}
|
|
138
|
+
|
|
139
|
+
// Escalations are signalled via build_error (not a separate 'escalation' event).
|
|
140
|
+
if (type === 'build_error') {
|
|
141
|
+
if (/escalat/i.test(ev?.message ?? '')) escalations++;
|
|
142
|
+
}
|
|
143
|
+
}
|
|
144
|
+
return { reviewIters, gateFailures, retries, escalations };
|
|
145
|
+
}
|
|
146
|
+
|
|
147
|
+
/**
|
|
148
|
+
* Extract health score from build-stream events.
|
|
149
|
+
*
|
|
150
|
+
* @param {object[]} events
|
|
151
|
+
* @returns {number|null}
|
|
152
|
+
*/
|
|
153
|
+
function parseHealthFromStream(events) {
|
|
154
|
+
for (let i = events.length - 1; i >= 0; i--) {
|
|
155
|
+
const ev = events[i];
|
|
156
|
+
if ((ev?.type === 'health_score' || ev?.kind === 'health_score') && typeof ev?.score === 'number') {
|
|
157
|
+
return ev.score;
|
|
158
|
+
}
|
|
159
|
+
// health_score embedded in kind/metadata envelope format
|
|
160
|
+
if (ev?.kind === 'health_score' && typeof ev?.metadata?.score === 'number') {
|
|
161
|
+
return ev.metadata.score;
|
|
162
|
+
}
|
|
163
|
+
}
|
|
164
|
+
return null;
|
|
165
|
+
}
|
|
166
|
+
|
|
167
|
+
// ---------------------------------------------------------------------------
|
|
168
|
+
// Public API
|
|
169
|
+
// ---------------------------------------------------------------------------
|
|
170
|
+
|
|
171
|
+
/**
|
|
172
|
+
* Collect all four metric axes from a sandbox's build artifacts.
|
|
173
|
+
*
|
|
174
|
+
* @param {object} args
|
|
175
|
+
* @param {{ workspace: string, runDir: string }} args.sandbox
|
|
176
|
+
* workspace — the git workspace dir (for git diff --stat)
|
|
177
|
+
* runDir — the run's output dir (contains manifest.json, build artifacts)
|
|
178
|
+
* @param {{ exitCode?: number, stdout?: string, wallMs?: number }} [args.buildResult]
|
|
179
|
+
* Optional output from the build process. `wallMs` is used as a cost fallback
|
|
180
|
+
* when the history record has no durationMs (e.g. crash before history write).
|
|
181
|
+
* `stdout` is retained for backward-compat but is no longer parsed for tests.
|
|
182
|
+
* @param {string|null} [args.baselineSha]
|
|
183
|
+
* SHA of the pre-build baseline commit. When provided, gitDiffStat diffs against
|
|
184
|
+
* this commit so in-process ship commits are counted (fix #2). Pass null / omit
|
|
185
|
+
* to fall back to `git diff --stat HEAD` (backward-compatible, greenfield fakes).
|
|
186
|
+
*
|
|
187
|
+
* @returns {{ cost: object, outcome: object, process: object }}
|
|
188
|
+
*/
|
|
189
|
+
export function collect({ sandbox, buildResult = {}, baselineSha = null }) {
|
|
190
|
+
const dataDir = join(sandbox.workspace, '.compose', 'data');
|
|
191
|
+
const composeDir = join(sandbox.workspace, '.compose');
|
|
192
|
+
|
|
193
|
+
// ---------------------------------------------------------------------------
|
|
194
|
+
// 1. Build-history record (terminal record — most recent)
|
|
195
|
+
// ---------------------------------------------------------------------------
|
|
196
|
+
const historyRecord = readLastJsonlRecord(join(dataDir, 'build-history.jsonl'));
|
|
197
|
+
|
|
198
|
+
// ---------------------------------------------------------------------------
|
|
199
|
+
// 2. Build stream events
|
|
200
|
+
// ---------------------------------------------------------------------------
|
|
201
|
+
const streamEvents = readAllJsonlRecords(join(composeDir, 'build-stream.jsonl'));
|
|
202
|
+
|
|
203
|
+
// ---------------------------------------------------------------------------
|
|
204
|
+
// 3. Cost axis
|
|
205
|
+
// ---------------------------------------------------------------------------
|
|
206
|
+
const tokensIn = historyRecord?.input_tokens ?? 0;
|
|
207
|
+
const tokensOut = historyRecord?.output_tokens ?? 0;
|
|
208
|
+
// calls = stepCount (step records in history, including gate records). This is
|
|
209
|
+
// NOT raw model invocations — one step can make multiple LLM calls internally.
|
|
210
|
+
const calls = historyRecord?.stepCount ?? 0;
|
|
211
|
+
const wallMs = historyRecord?.durationMs ?? (buildResult?.wallMs ?? 0);
|
|
212
|
+
|
|
213
|
+
// Try to derive USD from the history record's embedded cost first.
|
|
214
|
+
// Fall back to deriveUsd with a model from the stream if needed.
|
|
215
|
+
let usd = null;
|
|
216
|
+
if (typeof historyRecord?.cost_usd === 'number') {
|
|
217
|
+
usd = historyRecord.cost_usd;
|
|
218
|
+
} else {
|
|
219
|
+
// Find any step_model event to get the model ID for pricing
|
|
220
|
+
const modelEv = streamEvents.find(
|
|
221
|
+
e => (e?.type === 'step_model' || e?.kind === 'step_model') && (e?.modelID ?? e?.metadata?.modelID)
|
|
222
|
+
);
|
|
223
|
+
const modelID = modelEv?.modelID ?? modelEv?.metadata?.modelID ?? null;
|
|
224
|
+
if (modelID) usd = deriveUsd(modelID, tokensIn, tokensOut);
|
|
225
|
+
}
|
|
226
|
+
|
|
227
|
+
const cost = { tokensIn, tokensOut, calls, wallMs, usd };
|
|
228
|
+
|
|
229
|
+
// ---------------------------------------------------------------------------
|
|
230
|
+
// 4. Outcome axis
|
|
231
|
+
// ---------------------------------------------------------------------------
|
|
232
|
+
const completed = historyRecord?.status === 'complete';
|
|
233
|
+
const health = parseHealthFromStream(streamEvents);
|
|
234
|
+
|
|
235
|
+
// Test pass rate: the build's ship step runs tests via execSync and persists
|
|
236
|
+
// structured counts to build-history.jsonl as test_count and pass_rate fields
|
|
237
|
+
// (COMP-MODEL-AB fix B). Read directly from the history record — the ship step's
|
|
238
|
+
// execSync output never reaches realHeadlessBuild's stdout buffer (only the
|
|
239
|
+
// outer `node compose build` process's own stdout is captured there).
|
|
240
|
+
//
|
|
241
|
+
// v1 limitation: testsTotal/testsPass are null on failed, aborted, or thrown
|
|
242
|
+
// builds even if tests ran before the failure. _extractShipTestMetrics only
|
|
243
|
+
// fires on the success path (ship step completes + testSummary.parsed=true);
|
|
244
|
+
// terminalizeThrownBuild and the early-abort path never carry test_count/pass_rate.
|
|
245
|
+
// outcome.completed=false already signals the failure; null test metrics on those
|
|
246
|
+
// paths are expected and should not be treated as a data gap.
|
|
247
|
+
const testsTotal = historyRecord?.test_count ?? null;
|
|
248
|
+
let testsPass = null;
|
|
249
|
+
if (testsTotal !== null && historyRecord?.pass_rate != null) {
|
|
250
|
+
testsPass = historyRecord.pass_rate === 100
|
|
251
|
+
? testsTotal
|
|
252
|
+
: Math.round((historyRecord.pass_rate / 100) * testsTotal);
|
|
253
|
+
}
|
|
254
|
+
|
|
255
|
+
const { filesChanged, linesChanged } = gitDiffStat(sandbox.workspace, baselineSha);
|
|
256
|
+
|
|
257
|
+
const outcome = { completed, health, testsPass, testsTotal, filesChanged, linesChanged };
|
|
258
|
+
|
|
259
|
+
// ---------------------------------------------------------------------------
|
|
260
|
+
// 5. Process axis
|
|
261
|
+
// ---------------------------------------------------------------------------
|
|
262
|
+
const process = parseProcessFromStream(streamEvents);
|
|
263
|
+
|
|
264
|
+
return { cost, outcome, process };
|
|
265
|
+
}
|
|
@@ -0,0 +1,60 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* experiment-pricing.js — Static model→$/MTok table for COMP-MODEL-AB.
|
|
3
|
+
*
|
|
4
|
+
* Used by experiment-metrics.js to derive a USD cost estimate from raw token
|
|
5
|
+
* counts when build artifacts don't already carry a cost field. Unknown
|
|
6
|
+
* model IDs degrade to usd:null rather than crashing — a crashed / future
|
|
7
|
+
* model still yields a record with partial metrics.
|
|
8
|
+
*
|
|
9
|
+
* Price source: Anthropic public pricing page + OpenAI pricing (as of 2026-06).
|
|
10
|
+
* Keys are prefix-matched so dated variants (e.g. claude-sonnet-4-6-20250514)
|
|
11
|
+
* resolve against the base key.
|
|
12
|
+
*/
|
|
13
|
+
|
|
14
|
+
/** @type {Record<string, { inputPerMTok: number, outputPerMTok: number }>} */
|
|
15
|
+
const EXPERIMENT_PRICING = {
|
|
16
|
+
// Claude 4.x
|
|
17
|
+
'claude-opus-4-8': { inputPerMTok: 5, outputPerMTok: 25 },
|
|
18
|
+
'claude-opus-4-7': { inputPerMTok: 5, outputPerMTok: 25 },
|
|
19
|
+
'claude-opus-4-6': { inputPerMTok: 5, outputPerMTok: 25 },
|
|
20
|
+
'claude-sonnet-4-6': { inputPerMTok: 3, outputPerMTok: 15 },
|
|
21
|
+
'claude-haiku-4-5': { inputPerMTok: 1, outputPerMTok: 5 },
|
|
22
|
+
// GPT / Codex
|
|
23
|
+
'gpt-5': { inputPerMTok: 10, outputPerMTok: 40 },
|
|
24
|
+
'gpt-5.4': { inputPerMTok: 10, outputPerMTok: 40 },
|
|
25
|
+
'gpt-4.1': { inputPerMTok: 2, outputPerMTok: 8 },
|
|
26
|
+
'gpt-4o': { inputPerMTok: 2.5, outputPerMTok: 10 },
|
|
27
|
+
'o3': { inputPerMTok: 10, outputPerMTok: 40 },
|
|
28
|
+
'o4-mini': { inputPerMTok: 1.1, outputPerMTok: 4.4 },
|
|
29
|
+
};
|
|
30
|
+
|
|
31
|
+
/**
|
|
32
|
+
* Look up pricing for a model ID by exact match then prefix match.
|
|
33
|
+
*
|
|
34
|
+
* @param {string|null|undefined} modelID
|
|
35
|
+
* @returns {{ inputPerMTok: number, outputPerMTok: number } | null}
|
|
36
|
+
*/
|
|
37
|
+
export function lookupExperimentPricing(modelID) {
|
|
38
|
+
if (!modelID) return null;
|
|
39
|
+
if (EXPERIMENT_PRICING[modelID]) return EXPERIMENT_PRICING[modelID];
|
|
40
|
+
for (const [key, pricing] of Object.entries(EXPERIMENT_PRICING)) {
|
|
41
|
+
if (modelID.startsWith(key)) return pricing;
|
|
42
|
+
}
|
|
43
|
+
return null;
|
|
44
|
+
}
|
|
45
|
+
|
|
46
|
+
/**
|
|
47
|
+
* Derive a USD cost from token counts using the static pricing table.
|
|
48
|
+
*
|
|
49
|
+
* @param {string|null|undefined} modelID
|
|
50
|
+
* @param {number} tokensIn
|
|
51
|
+
* @param {number} tokensOut
|
|
52
|
+
* @returns {number|null} USD cost, or null for unknown models
|
|
53
|
+
*/
|
|
54
|
+
export function deriveUsd(modelID, tokensIn, tokensOut) {
|
|
55
|
+
const pricing = lookupExperimentPricing(modelID);
|
|
56
|
+
if (!pricing) return null;
|
|
57
|
+
const inputCost = ((tokensIn ?? 0) / 1_000_000) * pricing.inputPerMTok;
|
|
58
|
+
const outputCost = ((tokensOut ?? 0) / 1_000_000) * pricing.outputPerMTok;
|
|
59
|
+
return inputCost + outputCost;
|
|
60
|
+
}
|