@kodax-ai/kodax 0.7.95 → 0.7.96-alpha.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (163) hide show
  1. package/CHANGELOG.md +178 -1
  2. package/LICENSE +158 -158
  3. package/README.md +149 -24
  4. package/README_CN.md +117 -24
  5. package/config-templates/config.example.jsonc +2 -1
  6. package/config-templates/integrations/a2a.example.jsonc +91 -91
  7. package/config-templates/integrations/extensions.example.jsonc +7 -7
  8. package/config-templates/integrations/mcp.example.jsonc +16 -16
  9. package/dist/builtin/code-review/SKILL.md +82 -82
  10. package/dist/builtin/git-workflow/SKILL.md +84 -84
  11. package/dist/builtin/skill-creator/SKILL.md +127 -127
  12. package/dist/builtin/skill-creator/agents/analyzer.md +12 -12
  13. package/dist/builtin/skill-creator/agents/comparator.md +13 -13
  14. package/dist/builtin/skill-creator/agents/grader.md +13 -13
  15. package/dist/builtin/skill-creator/references/schemas.md +227 -227
  16. package/dist/builtin/skill-creator/scripts/aggregate-benchmark.d.ts +46 -46
  17. package/dist/builtin/skill-creator/scripts/aggregate-benchmark.js +208 -208
  18. package/dist/builtin/skill-creator/scripts/analyze-benchmark.d.ts +46 -46
  19. package/dist/builtin/skill-creator/scripts/analyze-benchmark.js +286 -286
  20. package/dist/builtin/skill-creator/scripts/compare-runs.d.ts +62 -62
  21. package/dist/builtin/skill-creator/scripts/compare-runs.js +330 -330
  22. package/dist/builtin/skill-creator/scripts/generate-review.d.ts +33 -33
  23. package/dist/builtin/skill-creator/scripts/generate-review.js +414 -414
  24. package/dist/builtin/skill-creator/scripts/grade-evals.d.ts +73 -73
  25. package/dist/builtin/skill-creator/scripts/grade-evals.js +402 -402
  26. package/dist/builtin/skill-creator/scripts/improve-description.d.ts +23 -23
  27. package/dist/builtin/skill-creator/scripts/improve-description.js +160 -160
  28. package/dist/builtin/skill-creator/scripts/init-skill.d.ts +14 -14
  29. package/dist/builtin/skill-creator/scripts/init-skill.js +155 -155
  30. package/dist/builtin/skill-creator/scripts/install-skill.d.ts +29 -29
  31. package/dist/builtin/skill-creator/scripts/install-skill.js +173 -173
  32. package/dist/builtin/skill-creator/scripts/package-skill.d.ts +38 -38
  33. package/dist/builtin/skill-creator/scripts/package-skill.js +121 -121
  34. package/dist/builtin/skill-creator/scripts/quick-validate.d.ts +8 -8
  35. package/dist/builtin/skill-creator/scripts/quick-validate.js +163 -163
  36. package/dist/builtin/skill-creator/scripts/run-eval.d.ts +66 -66
  37. package/dist/builtin/skill-creator/scripts/run-eval.js +353 -353
  38. package/dist/builtin/skill-creator/scripts/run-loop.d.ts +49 -49
  39. package/dist/builtin/skill-creator/scripts/run-loop.js +242 -242
  40. package/dist/builtin/skill-creator/scripts/run-trigger-eval.d.ts +58 -58
  41. package/dist/builtin/skill-creator/scripts/run-trigger-eval.js +224 -224
  42. package/dist/builtin/tdd/SKILL.md +56 -56
  43. package/dist/chunks/agent-4FABF6WK.js +2 -0
  44. package/dist/chunks/argument-completer-72VHB2VL.js +2 -0
  45. package/dist/chunks/{chunk-7JXA3533.js → chunk-4OIQZWA5.js} +223 -226
  46. package/dist/chunks/{chunk-BNVSKIAB.js → chunk-6BVF3HZ7.js} +136 -136
  47. package/dist/chunks/chunk-7JZZIFZH.js +501 -0
  48. package/dist/chunks/chunk-AC7Z2OIY.js +2 -0
  49. package/dist/chunks/{chunk-JS452J2F.js → chunk-C4KYZAFU.js} +2 -2
  50. package/dist/chunks/chunk-CHUPIWIF.js +251 -0
  51. package/dist/chunks/{chunk-C4DTUKTR.js → chunk-GYUU7XLH.js} +5 -4
  52. package/dist/chunks/{chunk-OK4AXDC5.js → chunk-HAX55GOQ.js} +1 -1
  53. package/dist/chunks/chunk-JKTJCGO2.js +418 -0
  54. package/dist/chunks/{chunk-O743AJ5V.js → chunk-LQYETRFK.js} +171 -175
  55. package/dist/chunks/{chunk-2PEDBGKN.js → chunk-LTUZTQ6W.js} +1 -1
  56. package/dist/chunks/chunk-LYWDUSKO.js +2 -0
  57. package/dist/chunks/chunk-NGH6T4FD.js +92 -0
  58. package/dist/chunks/{chunk-SRIJX5ZP.js → chunk-NVRWCSA5.js} +1 -1
  59. package/dist/chunks/chunk-O4DS7RIQ.js +384 -0
  60. package/dist/chunks/chunk-ORXU6OWZ.js +2 -0
  61. package/dist/chunks/chunk-POTW3O65.js +124 -0
  62. package/dist/chunks/{chunk-2TM3W3GK.js → chunk-RIQSS56Y.js} +1 -1
  63. package/dist/chunks/chunk-U27XKEN5.js +407 -0
  64. package/dist/chunks/{chunk-6ABQJUK6.js → chunk-XUX6OUCA.js} +2 -2
  65. package/dist/chunks/{client-JDOS3YJR.js → client-4C456K4O.js} +1 -1
  66. package/dist/chunks/compaction-config-3VQRDFF4.js +2 -0
  67. package/dist/chunks/{construction-bootstrap-YCWHHUYO.js → construction-bootstrap-AHN7EDGG.js} +1 -1
  68. package/dist/chunks/dist-BCBQXAGI.js +2 -0
  69. package/dist/chunks/dist-QK2YDWU6.js +2 -0
  70. package/dist/chunks/host-VIHV7ADG.js +2 -0
  71. package/dist/chunks/run-manager-NBVMU3KV.js +2 -0
  72. package/dist/chunks/utils-MEJZ2IGI.js +2 -0
  73. package/dist/index.d.ts +22 -21
  74. package/dist/index.js +7 -7
  75. package/dist/kodax_bootstrap.js +26 -26
  76. package/dist/kodax_cli.js +1905 -1882
  77. package/dist/native/darwin-arm64/LICENSE-APACHE.txt +13 -0
  78. package/dist/native/darwin-arm64/kodax-text-transaction.node +0 -0
  79. package/dist/native/darwin-arm64/manifest.json +16 -0
  80. package/dist/native/darwin-x64/LICENSE-APACHE.txt +13 -0
  81. package/dist/native/darwin-x64/kodax-text-transaction.node +0 -0
  82. package/dist/native/darwin-x64/manifest.json +16 -0
  83. package/dist/native/linux-arm64/LICENSE-APACHE.txt +13 -0
  84. package/dist/native/linux-arm64/kodax-text-transaction.node +0 -0
  85. package/dist/native/linux-arm64/manifest.json +16 -0
  86. package/dist/native/linux-x64/LICENSE-APACHE.txt +13 -0
  87. package/dist/native/linux-x64/kodax-text-transaction.node +0 -0
  88. package/dist/native/linux-x64/manifest.json +16 -0
  89. package/dist/native/win32-x64/LICENSE-APACHE.txt +13 -0
  90. package/dist/native/win32-x64/NOTICE-windows-sandbox.txt +8 -0
  91. package/dist/native/win32-x64/kodax-windows-sandbox.exe +0 -0
  92. package/dist/native/win32-x64/kodax-windows-text-transaction.node +0 -0
  93. package/dist/native/win32-x64/manifest.json +30 -0
  94. package/dist/provider-capabilities.json +108 -2
  95. package/dist/runtime-worker.js +1745 -1727
  96. package/dist/sandbox-network-broker.js +1715 -0
  97. package/dist/sdk-a2a.d.ts +13 -12
  98. package/dist/sdk-a2a.js +1 -1
  99. package/dist/sdk-agent.d.ts +27 -18
  100. package/dist/sdk-agent.js +1 -1
  101. package/dist/sdk-coding.d.ts +107 -58
  102. package/dist/sdk-coding.js +1 -1
  103. package/dist/sdk-experimental-memory.d.ts +4 -4
  104. package/dist/sdk-experimental-memory.js +1 -1
  105. package/dist/sdk-llm.d.ts +9 -10
  106. package/dist/sdk-llm.js +1 -1
  107. package/dist/sdk-mcp.js +1 -1
  108. package/dist/sdk-media.d.ts +1 -1
  109. package/dist/sdk-media.js +1 -1
  110. package/dist/sdk-repl.d.ts +25 -17
  111. package/dist/sdk-repl.js +1 -1
  112. package/dist/sdk-runtime.d.ts +41 -17
  113. package/dist/sdk-runtime.js +1 -1
  114. package/dist/sdk-sandbox.d.ts +17 -3
  115. package/dist/sdk-sandbox.js +1 -1
  116. package/dist/sdk-session.d.ts +8 -7
  117. package/dist/sdk-session.js +1 -1
  118. package/dist/sdk-skills.js +1 -1
  119. package/dist/semantic-worker.js +61 -61
  120. package/dist/types-chunks/{base.d-vopO1rml.d.ts → base.d-COo8E2dk.d.ts} +18 -27
  121. package/dist/types-chunks/{bash-prefix-extractor.d-B2rxNI4L.d.ts → bash-prefix-extractor.d-DQOlA8cy.d.ts} +72 -52
  122. package/dist/types-chunks/{capability-learning.d-CE1tm29C.d.ts → capability-learning.d-lRPgA3Or.d.ts} +1 -1
  123. package/dist/types-chunks/{capsule.d-Y7QRS4J2.d.ts → capsule.d-0zxsgXg_.d.ts} +3 -3
  124. package/dist/types-chunks/{controller.d-BR-UVeUR.d.ts → controller.d-BTtyIIC1.d.ts} +1 -1
  125. package/dist/types-chunks/{controller.d-C5noq5-w.d.ts → controller.d-C7tXUcXo.d.ts} +2 -2
  126. package/dist/types-chunks/{guardrail.d-DqHD-61O.d.ts → guardrail.d-DKTyEHJa.d.ts} +5 -5
  127. package/dist/types-chunks/{history-retrieval.d-CKU8zmt_.d.ts → history-retrieval.d-CHnL99S7.d.ts} +2 -2
  128. package/dist/types-chunks/{public-api.d-D7fXaWW8.d.ts → public-api.d-Dx3L3W4Z.d.ts} +7 -4
  129. package/dist/types-chunks/{repl.d-BYDb14s6.d.ts → repl.d-mO01olIa.d.ts} +5 -5
  130. package/dist/types-chunks/{resolver.d-CwHUwbdE.d.ts → resolver.d-DlXPjeyx.d.ts} +38 -6
  131. package/dist/types-chunks/{review-inbox.d-BSgFewyj.d.ts → review-inbox.d-WDSe1nze.d.ts} +1 -1
  132. package/dist/types-chunks/{run-manager.d-BVJSntpS.d.ts → run-manager.d-DENbbzgR.d.ts} +1 -1
  133. package/dist/types-chunks/{sdk-session-CqEG84oP.d.ts → sdk-session-DYHGXXH4.d.ts} +3 -3
  134. package/dist/types-chunks/{shell-command-sets.d-DqaaBwhU.d.ts → shell-command-sets.d-CjFqS4dp.d.ts} +1 -1
  135. package/dist/types-chunks/{side-query.d-iH0P9ASy.d.ts → side-query.d-j8A3sFv_.d.ts} +2 -2
  136. package/dist/types-chunks/{types.d-rUOjXych.d.ts → types.d-BbZjR0Wr.d.ts} +1 -1
  137. package/dist/types-chunks/{types.d-CaTXAW_Q.d.ts → types.d-BkTIKdH6.d.ts} +2 -2
  138. package/dist/types-chunks/{types.d-C5xQEFcj.d.ts → types.d-DncLrpu_.d.ts} +4 -4
  139. package/dist/types-chunks/{types.d-CK9A9rpP.d.ts → types.d-Xy0f3x4i.d.ts} +9 -0
  140. package/dist/types-chunks/{utils.d-CWUyLiXo.d.ts → utils.d-CnxefMos.d.ts} +6 -6
  141. package/package.json +10 -3
  142. package/public_docs/README.md +130 -0
  143. package/public_docs/configuration/sandbox.md +289 -0
  144. package/public_docs/sdk/embedder-guide.md +6674 -0
  145. package/scripts/kodax-bin.cjs +28 -28
  146. package/scripts/production-env.cjs +25 -25
  147. package/dist/chunks/agent-XXZMXKKA.js +0 -2
  148. package/dist/chunks/argument-completer-KCYDGTQW.js +0 -2
  149. package/dist/chunks/chunk-77PNP27P.js +0 -92
  150. package/dist/chunks/chunk-7QBWHCIZ.js +0 -2
  151. package/dist/chunks/chunk-BHE66SP3.js +0 -123
  152. package/dist/chunks/chunk-BR2OSB2I.js +0 -722
  153. package/dist/chunks/chunk-DSMONSVB.js +0 -407
  154. package/dist/chunks/chunk-Q5VRWSKG.js +0 -418
  155. package/dist/chunks/chunk-XWRPOGJN.js +0 -4
  156. package/dist/chunks/chunk-Y5VIQBT6.js +0 -383
  157. package/dist/chunks/compaction-config-EGS4577X.js +0 -2
  158. package/dist/chunks/dist-ITTXQHRJ.js +0 -2
  159. package/dist/chunks/dist-KFYN3OKD.js +0 -2
  160. package/dist/chunks/host-VAYADIPW.js +0 -2
  161. package/dist/chunks/run-manager-52IJZMZJ.js +0 -2
  162. package/dist/chunks/utils-P7PBSL3R.js +0 -2
  163. package/dist/sandbox-workspace-session.js +0 -1705
@@ -1,290 +1,290 @@
1
- #!/usr/bin/env node
2
-
3
- import { readFile, writeFile } from 'node:fs/promises';
4
- import path from 'node:path';
5
- import { fileURLToPath } from 'node:url';
6
- import {
7
- buildBenchmarkDocument,
8
- loadRunResults,
9
- } from './aggregate-benchmark.js';
10
- import {
11
- extractJsonObject,
12
- loadKodaXSDK,
13
- loadRelativeText,
14
- readJsonFile,
15
- truncateText,
16
- } from './utils.js';
17
-
18
- function normalizeStringArray(value) {
19
- if (!Array.isArray(value)) {
20
- return [];
21
- }
22
- return value
23
- .map((item) => String(item ?? '').trim())
24
- .filter(Boolean);
25
- }
26
-
27
- function summarizeFailureClusters(configRuns) {
28
- const clusters = {};
29
-
30
- for (const [configName, runs] of Object.entries(configRuns)) {
31
- const failureCounts = new Map();
32
- const notes = [];
33
-
34
- for (const run of runs) {
35
- for (const expectation of run.expectations ?? []) {
36
- if (expectation?.passed === true) {
37
- continue;
38
- }
39
- const text = String(expectation?.text ?? '').trim();
40
- if (!text) {
41
- continue;
42
- }
43
- failureCounts.set(text, (failureCounts.get(text) ?? 0) + 1);
44
- }
45
-
46
- for (const note of run.notes ?? []) {
47
- const normalized = String(note ?? '').trim();
48
- if (normalized) {
49
- notes.push(normalized);
50
- }
51
- }
52
- }
53
-
54
- clusters[configName] = {
55
- repeated_failures: Array.from(failureCounts.entries())
56
- .sort((left, right) => right[1] - left[1])
57
- .slice(0, 10)
58
- .map(([text, count]) => ({ text, count })),
59
- notes: notes.slice(0, 10),
60
- };
61
- }
62
-
63
- return clusters;
64
- }
65
-
66
- function normalizeAnalysisResult(rawText, benchmark, failureClusters) {
67
- const parsed = extractJsonObject(rawText) ?? {};
68
-
69
- return {
70
- skill_name: benchmark.skill_name,
71
- generated_at: new Date().toISOString(),
72
- workspace: benchmark.workspace,
73
- verdict: ['improves', 'regresses', 'mixed', 'inconclusive'].includes(parsed.verdict)
74
- ? parsed.verdict
75
- : 'inconclusive',
76
- release_readiness: ['ready', 'needs_iteration', 'needs_manual_review'].includes(parsed.release_readiness)
77
- ? parsed.release_readiness
78
- : 'needs_manual_review',
79
- recommendation: String(parsed.recommendation ?? '').trim(),
80
- key_findings: normalizeStringArray(parsed.key_findings),
81
- variance_hotspots: normalizeStringArray(parsed.variance_hotspots),
82
- suggested_actions: normalizeStringArray(parsed.suggested_actions),
83
- watchouts: normalizeStringArray(parsed.watchouts),
84
- supporting_metrics: {
85
- pass_rate_delta: benchmark.delta?.pass_rate ?? 'n/a',
86
- time_seconds_delta: benchmark.delta?.time_seconds ?? 'n/a',
87
- tokens_delta: benchmark.delta?.tokens ?? 'n/a',
88
- },
89
- failure_clusters: failureClusters,
90
- };
91
- }
92
-
93
- export function buildAnalysisPrompt(input) {
94
- return `${input.agentInstructions.trim()}
95
-
96
- Return JSON with this shape:
97
- {
98
- "verdict": "improves | regresses | mixed | inconclusive",
99
- "release_readiness": "ready | needs_iteration | needs_manual_review",
100
- "recommendation": "short recommendation",
101
- "key_findings": [],
102
- "variance_hotspots": [],
103
- "suggested_actions": [],
104
- "watchouts": []
105
- }
106
-
107
- ## Benchmark Summary
108
- ${truncateText(JSON.stringify({
109
- skill_name: input.benchmark.skill_name,
110
- configs: input.benchmark.configs,
111
- delta: input.benchmark.delta,
112
- }, null, 2), 12000)}
113
-
114
- ## Failure Clusters
115
- ${truncateText(JSON.stringify(input.failureClusters, null, 2), 8000)}
116
- `;
117
- }
118
-
119
- async function defaultRunAnalyst(prompt, options) {
120
- const { runKodaX } = await loadKodaXSDK();
121
- const result = await runKodaX(
122
- {
123
- provider: options.provider ?? 'anthropic',
124
- model: options.model,
125
- maxIter: options.maxIter ?? 20,
126
- reasoningMode: options.reasoningMode ?? 'balanced',
127
- thinking: options.reasoningMode ? options.reasoningMode !== 'off' : true,
128
- context: {
129
- gitRoot: path.resolve(options.cwd ?? options.workspaceDir ?? process.cwd()),
130
- },
131
- },
132
- prompt
133
- );
134
- return result.lastText;
135
- }
136
-
137
- export function renderAnalysisMarkdown(analysis) {
138
- const lines = [
139
- `# Benchmark Analysis: ${analysis.skill_name}`,
140
- '',
141
- `Generated: ${analysis.generated_at}`,
142
- '',
143
- `- Verdict: ${analysis.verdict}`,
144
- `- Release readiness: ${analysis.release_readiness}`,
145
- `- Recommendation: ${analysis.recommendation || 'n/a'}`,
146
- '',
147
- ];
148
-
149
- const sections = [
150
- ['key_findings', 'Key Findings'],
151
- ['variance_hotspots', 'Variance Hotspots'],
152
- ['suggested_actions', 'Suggested Actions'],
153
- ['watchouts', 'Watchouts'],
154
- ];
155
-
156
- for (const [field, title] of sections) {
157
- lines.push(`## ${title}`);
158
- lines.push('');
159
- const items = Array.isArray(analysis[field]) ? analysis[field] : [];
160
- if (items.length === 0) {
161
- lines.push('- None');
162
- } else {
163
- for (const item of items) {
164
- lines.push(`- ${item}`);
165
- }
166
- }
167
- lines.push('');
168
- }
169
-
170
- lines.push('## Supporting Metrics');
171
- lines.push('');
172
- lines.push(`- Pass rate delta: ${analysis.supporting_metrics.pass_rate_delta}`);
173
- lines.push(`- Time delta: ${analysis.supporting_metrics.time_seconds_delta}`);
174
- lines.push(`- Tokens delta: ${analysis.supporting_metrics.tokens_delta}`);
175
-
176
- return `${lines.join('\n')}\n`;
177
- }
178
-
179
- export async function analyzeBenchmark(
180
- options,
181
- runner = defaultRunAnalyst
182
- ) {
183
- const workspaceDir = path.resolve(options.workspaceDir);
184
- const benchmarkPath = path.resolve(options.benchmarkPath ?? path.join(workspaceDir, 'benchmark.json'));
185
- let benchmark = await readJsonFile(benchmarkPath, null);
186
-
187
- if (!benchmark) {
188
- const configRuns = await loadRunResults(workspaceDir);
189
- if (Object.keys(configRuns).length === 0) {
190
- throw new Error(`No benchmark data found in ${workspaceDir}`);
191
- }
192
- benchmark = buildBenchmarkDocument(workspaceDir, options.skillName ?? path.basename(workspaceDir), configRuns);
193
- await writeFile(benchmarkPath, `${JSON.stringify(benchmark, null, 2)}\n`, 'utf8');
194
- }
195
-
196
- const configRuns = await loadRunResults(workspaceDir);
197
- const failureClusters = summarizeFailureClusters(configRuns);
198
- const agentInstructions = await loadRelativeText(import.meta.url, '../agents/analyzer.md');
199
- const prompt = buildAnalysisPrompt({
200
- agentInstructions,
201
- benchmark,
202
- failureClusters,
203
- });
204
- const rawResponse = await runner(prompt, {
205
- ...options,
206
- workspaceDir,
207
- benchmarkPath,
208
- benchmark,
209
- });
210
- const analysis = normalizeAnalysisResult(rawResponse, benchmark, failureClusters);
211
- const analysisJsonPath = path.resolve(options.outputPath ?? path.join(workspaceDir, 'analysis.json'));
212
- const analysisMdPath = path.resolve(options.markdownPath ?? path.join(workspaceDir, 'analysis.md'));
213
-
214
- await writeFile(analysisJsonPath, `${JSON.stringify(analysis, null, 2)}\n`, 'utf8');
215
- await writeFile(analysisMdPath, renderAnalysisMarkdown(analysis), 'utf8');
216
-
217
- return {
218
- analysis,
219
- prompt,
220
- rawResponse,
221
- analysisJsonPath,
222
- analysisMdPath,
223
- };
224
- }
225
-
226
- function parseArgs(argv) {
227
- const args = {
228
- workspaceDir: argv[2] ?? '',
229
- benchmarkPath: undefined,
230
- outputPath: undefined,
231
- markdownPath: undefined,
232
- skillName: undefined,
233
- provider: 'anthropic',
234
- model: undefined,
235
- reasoningMode: 'balanced',
236
- maxIter: 20,
237
- cwd: process.cwd(),
238
- };
239
-
240
- for (let index = 3; index < argv.length; index += 1) {
241
- const token = argv[index];
242
- if (token === '--benchmark' && argv[index + 1]) {
243
- args.benchmarkPath = argv[++index];
244
- } else if (token === '--output' && argv[index + 1]) {
245
- args.outputPath = argv[++index];
246
- } else if (token === '--markdown' && argv[index + 1]) {
247
- args.markdownPath = argv[++index];
248
- } else if (token === '--skill-name' && argv[index + 1]) {
249
- args.skillName = argv[++index];
250
- } else if (token === '--provider' && argv[index + 1]) {
251
- args.provider = argv[++index];
252
- } else if (token === '--model' && argv[index + 1]) {
253
- args.model = argv[++index];
254
- } else if (token === '--reasoning' && argv[index + 1]) {
255
- args.reasoningMode = argv[++index];
256
- } else if (token === '--max-iter' && argv[index + 1]) {
257
- args.maxIter = Number(argv[++index]);
258
- } else if (token === '--cwd' && argv[index + 1]) {
259
- args.cwd = argv[++index];
260
- }
261
- }
262
-
263
- return args;
264
- }
265
-
1
+ #!/usr/bin/env node
2
+
3
+ import { readFile, writeFile } from 'node:fs/promises';
4
+ import path from 'node:path';
5
+ import { fileURLToPath } from 'node:url';
6
+ import {
7
+ buildBenchmarkDocument,
8
+ loadRunResults,
9
+ } from './aggregate-benchmark.js';
10
+ import {
11
+ extractJsonObject,
12
+ loadKodaXSDK,
13
+ loadRelativeText,
14
+ readJsonFile,
15
+ truncateText,
16
+ } from './utils.js';
17
+
18
+ function normalizeStringArray(value) {
19
+ if (!Array.isArray(value)) {
20
+ return [];
21
+ }
22
+ return value
23
+ .map((item) => String(item ?? '').trim())
24
+ .filter(Boolean);
25
+ }
26
+
27
+ function summarizeFailureClusters(configRuns) {
28
+ const clusters = {};
29
+
30
+ for (const [configName, runs] of Object.entries(configRuns)) {
31
+ const failureCounts = new Map();
32
+ const notes = [];
33
+
34
+ for (const run of runs) {
35
+ for (const expectation of run.expectations ?? []) {
36
+ if (expectation?.passed === true) {
37
+ continue;
38
+ }
39
+ const text = String(expectation?.text ?? '').trim();
40
+ if (!text) {
41
+ continue;
42
+ }
43
+ failureCounts.set(text, (failureCounts.get(text) ?? 0) + 1);
44
+ }
45
+
46
+ for (const note of run.notes ?? []) {
47
+ const normalized = String(note ?? '').trim();
48
+ if (normalized) {
49
+ notes.push(normalized);
50
+ }
51
+ }
52
+ }
53
+
54
+ clusters[configName] = {
55
+ repeated_failures: Array.from(failureCounts.entries())
56
+ .sort((left, right) => right[1] - left[1])
57
+ .slice(0, 10)
58
+ .map(([text, count]) => ({ text, count })),
59
+ notes: notes.slice(0, 10),
60
+ };
61
+ }
62
+
63
+ return clusters;
64
+ }
65
+
66
+ function normalizeAnalysisResult(rawText, benchmark, failureClusters) {
67
+ const parsed = extractJsonObject(rawText) ?? {};
68
+
69
+ return {
70
+ skill_name: benchmark.skill_name,
71
+ generated_at: new Date().toISOString(),
72
+ workspace: benchmark.workspace,
73
+ verdict: ['improves', 'regresses', 'mixed', 'inconclusive'].includes(parsed.verdict)
74
+ ? parsed.verdict
75
+ : 'inconclusive',
76
+ release_readiness: ['ready', 'needs_iteration', 'needs_manual_review'].includes(parsed.release_readiness)
77
+ ? parsed.release_readiness
78
+ : 'needs_manual_review',
79
+ recommendation: String(parsed.recommendation ?? '').trim(),
80
+ key_findings: normalizeStringArray(parsed.key_findings),
81
+ variance_hotspots: normalizeStringArray(parsed.variance_hotspots),
82
+ suggested_actions: normalizeStringArray(parsed.suggested_actions),
83
+ watchouts: normalizeStringArray(parsed.watchouts),
84
+ supporting_metrics: {
85
+ pass_rate_delta: benchmark.delta?.pass_rate ?? 'n/a',
86
+ time_seconds_delta: benchmark.delta?.time_seconds ?? 'n/a',
87
+ tokens_delta: benchmark.delta?.tokens ?? 'n/a',
88
+ },
89
+ failure_clusters: failureClusters,
90
+ };
91
+ }
92
+
93
+ export function buildAnalysisPrompt(input) {
94
+ return `${input.agentInstructions.trim()}
95
+
96
+ Return JSON with this shape:
97
+ {
98
+ "verdict": "improves | regresses | mixed | inconclusive",
99
+ "release_readiness": "ready | needs_iteration | needs_manual_review",
100
+ "recommendation": "short recommendation",
101
+ "key_findings": [],
102
+ "variance_hotspots": [],
103
+ "suggested_actions": [],
104
+ "watchouts": []
105
+ }
106
+
107
+ ## Benchmark Summary
108
+ ${truncateText(JSON.stringify({
109
+ skill_name: input.benchmark.skill_name,
110
+ configs: input.benchmark.configs,
111
+ delta: input.benchmark.delta,
112
+ }, null, 2), 12000)}
113
+
114
+ ## Failure Clusters
115
+ ${truncateText(JSON.stringify(input.failureClusters, null, 2), 8000)}
116
+ `;
117
+ }
118
+
119
+ async function defaultRunAnalyst(prompt, options) {
120
+ const { runKodaX } = await loadKodaXSDK();
121
+ const result = await runKodaX(
122
+ {
123
+ provider: options.provider ?? 'anthropic',
124
+ model: options.model,
125
+ maxIter: options.maxIter ?? 20,
126
+ reasoningMode: options.reasoningMode ?? 'balanced',
127
+ thinking: options.reasoningMode ? options.reasoningMode !== 'off' : true,
128
+ context: {
129
+ gitRoot: path.resolve(options.cwd ?? options.workspaceDir ?? process.cwd()),
130
+ },
131
+ },
132
+ prompt
133
+ );
134
+ return result.lastText;
135
+ }
136
+
137
+ export function renderAnalysisMarkdown(analysis) {
138
+ const lines = [
139
+ `# Benchmark Analysis: ${analysis.skill_name}`,
140
+ '',
141
+ `Generated: ${analysis.generated_at}`,
142
+ '',
143
+ `- Verdict: ${analysis.verdict}`,
144
+ `- Release readiness: ${analysis.release_readiness}`,
145
+ `- Recommendation: ${analysis.recommendation || 'n/a'}`,
146
+ '',
147
+ ];
148
+
149
+ const sections = [
150
+ ['key_findings', 'Key Findings'],
151
+ ['variance_hotspots', 'Variance Hotspots'],
152
+ ['suggested_actions', 'Suggested Actions'],
153
+ ['watchouts', 'Watchouts'],
154
+ ];
155
+
156
+ for (const [field, title] of sections) {
157
+ lines.push(`## ${title}`);
158
+ lines.push('');
159
+ const items = Array.isArray(analysis[field]) ? analysis[field] : [];
160
+ if (items.length === 0) {
161
+ lines.push('- None');
162
+ } else {
163
+ for (const item of items) {
164
+ lines.push(`- ${item}`);
165
+ }
166
+ }
167
+ lines.push('');
168
+ }
169
+
170
+ lines.push('## Supporting Metrics');
171
+ lines.push('');
172
+ lines.push(`- Pass rate delta: ${analysis.supporting_metrics.pass_rate_delta}`);
173
+ lines.push(`- Time delta: ${analysis.supporting_metrics.time_seconds_delta}`);
174
+ lines.push(`- Tokens delta: ${analysis.supporting_metrics.tokens_delta}`);
175
+
176
+ return `${lines.join('\n')}\n`;
177
+ }
178
+
179
+ export async function analyzeBenchmark(
180
+ options,
181
+ runner = defaultRunAnalyst
182
+ ) {
183
+ const workspaceDir = path.resolve(options.workspaceDir);
184
+ const benchmarkPath = path.resolve(options.benchmarkPath ?? path.join(workspaceDir, 'benchmark.json'));
185
+ let benchmark = await readJsonFile(benchmarkPath, null);
186
+
187
+ if (!benchmark) {
188
+ const configRuns = await loadRunResults(workspaceDir);
189
+ if (Object.keys(configRuns).length === 0) {
190
+ throw new Error(`No benchmark data found in ${workspaceDir}`);
191
+ }
192
+ benchmark = buildBenchmarkDocument(workspaceDir, options.skillName ?? path.basename(workspaceDir), configRuns);
193
+ await writeFile(benchmarkPath, `${JSON.stringify(benchmark, null, 2)}\n`, 'utf8');
194
+ }
195
+
196
+ const configRuns = await loadRunResults(workspaceDir);
197
+ const failureClusters = summarizeFailureClusters(configRuns);
198
+ const agentInstructions = await loadRelativeText(import.meta.url, '../agents/analyzer.md');
199
+ const prompt = buildAnalysisPrompt({
200
+ agentInstructions,
201
+ benchmark,
202
+ failureClusters,
203
+ });
204
+ const rawResponse = await runner(prompt, {
205
+ ...options,
206
+ workspaceDir,
207
+ benchmarkPath,
208
+ benchmark,
209
+ });
210
+ const analysis = normalizeAnalysisResult(rawResponse, benchmark, failureClusters);
211
+ const analysisJsonPath = path.resolve(options.outputPath ?? path.join(workspaceDir, 'analysis.json'));
212
+ const analysisMdPath = path.resolve(options.markdownPath ?? path.join(workspaceDir, 'analysis.md'));
213
+
214
+ await writeFile(analysisJsonPath, `${JSON.stringify(analysis, null, 2)}\n`, 'utf8');
215
+ await writeFile(analysisMdPath, renderAnalysisMarkdown(analysis), 'utf8');
216
+
217
+ return {
218
+ analysis,
219
+ prompt,
220
+ rawResponse,
221
+ analysisJsonPath,
222
+ analysisMdPath,
223
+ };
224
+ }
225
+
226
+ function parseArgs(argv) {
227
+ const args = {
228
+ workspaceDir: argv[2] ?? '',
229
+ benchmarkPath: undefined,
230
+ outputPath: undefined,
231
+ markdownPath: undefined,
232
+ skillName: undefined,
233
+ provider: 'anthropic',
234
+ model: undefined,
235
+ reasoningMode: 'balanced',
236
+ maxIter: 20,
237
+ cwd: process.cwd(),
238
+ };
239
+
240
+ for (let index = 3; index < argv.length; index += 1) {
241
+ const token = argv[index];
242
+ if (token === '--benchmark' && argv[index + 1]) {
243
+ args.benchmarkPath = argv[++index];
244
+ } else if (token === '--output' && argv[index + 1]) {
245
+ args.outputPath = argv[++index];
246
+ } else if (token === '--markdown' && argv[index + 1]) {
247
+ args.markdownPath = argv[++index];
248
+ } else if (token === '--skill-name' && argv[index + 1]) {
249
+ args.skillName = argv[++index];
250
+ } else if (token === '--provider' && argv[index + 1]) {
251
+ args.provider = argv[++index];
252
+ } else if (token === '--model' && argv[index + 1]) {
253
+ args.model = argv[++index];
254
+ } else if (token === '--reasoning' && argv[index + 1]) {
255
+ args.reasoningMode = argv[++index];
256
+ } else if (token === '--max-iter' && argv[index + 1]) {
257
+ args.maxIter = Number(argv[++index]);
258
+ } else if (token === '--cwd' && argv[index + 1]) {
259
+ args.cwd = argv[++index];
260
+ }
261
+ }
262
+
263
+ return args;
264
+ }
265
+
266
266
  export async function main(argv = process.argv) {
267
267
  const args = parseArgs(argv);
268
- if (!args.workspaceDir) {
269
- console.error('Usage: node scripts/analyze-benchmark.js <workspace> [--benchmark benchmark.json] [--output analysis.json] [--markdown analysis.md]');
270
- process.exit(1);
271
- }
272
-
273
- const result = await analyzeBenchmark(args);
274
- process.stdout.write(`${JSON.stringify({
275
- analysis: result.analysis,
276
- analysis_json: result.analysisJsonPath,
277
- analysis_md: result.analysisMdPath,
278
- }, null, 2)}\n`);
279
- }
280
-
268
+ if (!args.workspaceDir) {
269
+ console.error('Usage: node scripts/analyze-benchmark.js <workspace> [--benchmark benchmark.json] [--output analysis.json] [--markdown analysis.md]');
270
+ process.exit(1);
271
+ }
272
+
273
+ const result = await analyzeBenchmark(args);
274
+ process.stdout.write(`${JSON.stringify({
275
+ analysis: result.analysis,
276
+ analysis_json: result.analysisJsonPath,
277
+ analysis_md: result.analysisMdPath,
278
+ }, null, 2)}\n`);
279
+ }
280
+
281
281
  const isDirectRun = process.env.KODAX_MODULE_BUNDLE !== 'true'
282
282
  && process.argv[1]
283
- && fileURLToPath(import.meta.url) === path.resolve(process.argv[1]);
284
-
285
- if (isDirectRun) {
286
- main().catch((error) => {
287
- console.error(error instanceof Error ? error.message : String(error));
288
- process.exit(1);
289
- });
290
- }
283
+ && fileURLToPath(import.meta.url) === path.resolve(process.argv[1]);
284
+
285
+ if (isDirectRun) {
286
+ main().catch((error) => {
287
+ console.error(error instanceof Error ? error.message : String(error));
288
+ process.exit(1);
289
+ });
290
+ }