@kodax-ai/kodax 0.7.95 → 0.7.96-alpha.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (163) hide show
  1. package/CHANGELOG.md +178 -1
  2. package/LICENSE +158 -158
  3. package/README.md +149 -24
  4. package/README_CN.md +117 -24
  5. package/config-templates/config.example.jsonc +2 -1
  6. package/config-templates/integrations/a2a.example.jsonc +91 -91
  7. package/config-templates/integrations/extensions.example.jsonc +7 -7
  8. package/config-templates/integrations/mcp.example.jsonc +16 -16
  9. package/dist/builtin/code-review/SKILL.md +82 -82
  10. package/dist/builtin/git-workflow/SKILL.md +84 -84
  11. package/dist/builtin/skill-creator/SKILL.md +127 -127
  12. package/dist/builtin/skill-creator/agents/analyzer.md +12 -12
  13. package/dist/builtin/skill-creator/agents/comparator.md +13 -13
  14. package/dist/builtin/skill-creator/agents/grader.md +13 -13
  15. package/dist/builtin/skill-creator/references/schemas.md +227 -227
  16. package/dist/builtin/skill-creator/scripts/aggregate-benchmark.d.ts +46 -46
  17. package/dist/builtin/skill-creator/scripts/aggregate-benchmark.js +208 -208
  18. package/dist/builtin/skill-creator/scripts/analyze-benchmark.d.ts +46 -46
  19. package/dist/builtin/skill-creator/scripts/analyze-benchmark.js +286 -286
  20. package/dist/builtin/skill-creator/scripts/compare-runs.d.ts +62 -62
  21. package/dist/builtin/skill-creator/scripts/compare-runs.js +330 -330
  22. package/dist/builtin/skill-creator/scripts/generate-review.d.ts +33 -33
  23. package/dist/builtin/skill-creator/scripts/generate-review.js +414 -414
  24. package/dist/builtin/skill-creator/scripts/grade-evals.d.ts +73 -73
  25. package/dist/builtin/skill-creator/scripts/grade-evals.js +402 -402
  26. package/dist/builtin/skill-creator/scripts/improve-description.d.ts +23 -23
  27. package/dist/builtin/skill-creator/scripts/improve-description.js +160 -160
  28. package/dist/builtin/skill-creator/scripts/init-skill.d.ts +14 -14
  29. package/dist/builtin/skill-creator/scripts/init-skill.js +155 -155
  30. package/dist/builtin/skill-creator/scripts/install-skill.d.ts +29 -29
  31. package/dist/builtin/skill-creator/scripts/install-skill.js +173 -173
  32. package/dist/builtin/skill-creator/scripts/package-skill.d.ts +38 -38
  33. package/dist/builtin/skill-creator/scripts/package-skill.js +121 -121
  34. package/dist/builtin/skill-creator/scripts/quick-validate.d.ts +8 -8
  35. package/dist/builtin/skill-creator/scripts/quick-validate.js +163 -163
  36. package/dist/builtin/skill-creator/scripts/run-eval.d.ts +66 -66
  37. package/dist/builtin/skill-creator/scripts/run-eval.js +353 -353
  38. package/dist/builtin/skill-creator/scripts/run-loop.d.ts +49 -49
  39. package/dist/builtin/skill-creator/scripts/run-loop.js +242 -242
  40. package/dist/builtin/skill-creator/scripts/run-trigger-eval.d.ts +58 -58
  41. package/dist/builtin/skill-creator/scripts/run-trigger-eval.js +224 -224
  42. package/dist/builtin/tdd/SKILL.md +56 -56
  43. package/dist/chunks/agent-4FABF6WK.js +2 -0
  44. package/dist/chunks/argument-completer-72VHB2VL.js +2 -0
  45. package/dist/chunks/{chunk-7JXA3533.js → chunk-4OIQZWA5.js} +223 -226
  46. package/dist/chunks/{chunk-BNVSKIAB.js → chunk-6BVF3HZ7.js} +136 -136
  47. package/dist/chunks/chunk-7JZZIFZH.js +501 -0
  48. package/dist/chunks/chunk-AC7Z2OIY.js +2 -0
  49. package/dist/chunks/{chunk-JS452J2F.js → chunk-C4KYZAFU.js} +2 -2
  50. package/dist/chunks/chunk-CHUPIWIF.js +251 -0
  51. package/dist/chunks/{chunk-C4DTUKTR.js → chunk-GYUU7XLH.js} +5 -4
  52. package/dist/chunks/{chunk-OK4AXDC5.js → chunk-HAX55GOQ.js} +1 -1
  53. package/dist/chunks/chunk-JKTJCGO2.js +418 -0
  54. package/dist/chunks/{chunk-O743AJ5V.js → chunk-LQYETRFK.js} +171 -175
  55. package/dist/chunks/{chunk-2PEDBGKN.js → chunk-LTUZTQ6W.js} +1 -1
  56. package/dist/chunks/chunk-LYWDUSKO.js +2 -0
  57. package/dist/chunks/chunk-NGH6T4FD.js +92 -0
  58. package/dist/chunks/{chunk-SRIJX5ZP.js → chunk-NVRWCSA5.js} +1 -1
  59. package/dist/chunks/chunk-O4DS7RIQ.js +384 -0
  60. package/dist/chunks/chunk-ORXU6OWZ.js +2 -0
  61. package/dist/chunks/chunk-POTW3O65.js +124 -0
  62. package/dist/chunks/{chunk-2TM3W3GK.js → chunk-RIQSS56Y.js} +1 -1
  63. package/dist/chunks/chunk-U27XKEN5.js +407 -0
  64. package/dist/chunks/{chunk-6ABQJUK6.js → chunk-XUX6OUCA.js} +2 -2
  65. package/dist/chunks/{client-JDOS3YJR.js → client-4C456K4O.js} +1 -1
  66. package/dist/chunks/compaction-config-3VQRDFF4.js +2 -0
  67. package/dist/chunks/{construction-bootstrap-YCWHHUYO.js → construction-bootstrap-AHN7EDGG.js} +1 -1
  68. package/dist/chunks/dist-BCBQXAGI.js +2 -0
  69. package/dist/chunks/dist-QK2YDWU6.js +2 -0
  70. package/dist/chunks/host-VIHV7ADG.js +2 -0
  71. package/dist/chunks/run-manager-NBVMU3KV.js +2 -0
  72. package/dist/chunks/utils-MEJZ2IGI.js +2 -0
  73. package/dist/index.d.ts +22 -21
  74. package/dist/index.js +7 -7
  75. package/dist/kodax_bootstrap.js +26 -26
  76. package/dist/kodax_cli.js +1905 -1882
  77. package/dist/native/darwin-arm64/LICENSE-APACHE.txt +13 -0
  78. package/dist/native/darwin-arm64/kodax-text-transaction.node +0 -0
  79. package/dist/native/darwin-arm64/manifest.json +16 -0
  80. package/dist/native/darwin-x64/LICENSE-APACHE.txt +13 -0
  81. package/dist/native/darwin-x64/kodax-text-transaction.node +0 -0
  82. package/dist/native/darwin-x64/manifest.json +16 -0
  83. package/dist/native/linux-arm64/LICENSE-APACHE.txt +13 -0
  84. package/dist/native/linux-arm64/kodax-text-transaction.node +0 -0
  85. package/dist/native/linux-arm64/manifest.json +16 -0
  86. package/dist/native/linux-x64/LICENSE-APACHE.txt +13 -0
  87. package/dist/native/linux-x64/kodax-text-transaction.node +0 -0
  88. package/dist/native/linux-x64/manifest.json +16 -0
  89. package/dist/native/win32-x64/LICENSE-APACHE.txt +13 -0
  90. package/dist/native/win32-x64/NOTICE-windows-sandbox.txt +8 -0
  91. package/dist/native/win32-x64/kodax-windows-sandbox.exe +0 -0
  92. package/dist/native/win32-x64/kodax-windows-text-transaction.node +0 -0
  93. package/dist/native/win32-x64/manifest.json +30 -0
  94. package/dist/provider-capabilities.json +108 -2
  95. package/dist/runtime-worker.js +1745 -1727
  96. package/dist/sandbox-network-broker.js +1715 -0
  97. package/dist/sdk-a2a.d.ts +13 -12
  98. package/dist/sdk-a2a.js +1 -1
  99. package/dist/sdk-agent.d.ts +27 -18
  100. package/dist/sdk-agent.js +1 -1
  101. package/dist/sdk-coding.d.ts +107 -58
  102. package/dist/sdk-coding.js +1 -1
  103. package/dist/sdk-experimental-memory.d.ts +4 -4
  104. package/dist/sdk-experimental-memory.js +1 -1
  105. package/dist/sdk-llm.d.ts +9 -10
  106. package/dist/sdk-llm.js +1 -1
  107. package/dist/sdk-mcp.js +1 -1
  108. package/dist/sdk-media.d.ts +1 -1
  109. package/dist/sdk-media.js +1 -1
  110. package/dist/sdk-repl.d.ts +25 -17
  111. package/dist/sdk-repl.js +1 -1
  112. package/dist/sdk-runtime.d.ts +41 -17
  113. package/dist/sdk-runtime.js +1 -1
  114. package/dist/sdk-sandbox.d.ts +17 -3
  115. package/dist/sdk-sandbox.js +1 -1
  116. package/dist/sdk-session.d.ts +8 -7
  117. package/dist/sdk-session.js +1 -1
  118. package/dist/sdk-skills.js +1 -1
  119. package/dist/semantic-worker.js +61 -61
  120. package/dist/types-chunks/{base.d-vopO1rml.d.ts → base.d-COo8E2dk.d.ts} +18 -27
  121. package/dist/types-chunks/{bash-prefix-extractor.d-B2rxNI4L.d.ts → bash-prefix-extractor.d-DQOlA8cy.d.ts} +72 -52
  122. package/dist/types-chunks/{capability-learning.d-CE1tm29C.d.ts → capability-learning.d-lRPgA3Or.d.ts} +1 -1
  123. package/dist/types-chunks/{capsule.d-Y7QRS4J2.d.ts → capsule.d-0zxsgXg_.d.ts} +3 -3
  124. package/dist/types-chunks/{controller.d-BR-UVeUR.d.ts → controller.d-BTtyIIC1.d.ts} +1 -1
  125. package/dist/types-chunks/{controller.d-C5noq5-w.d.ts → controller.d-C7tXUcXo.d.ts} +2 -2
  126. package/dist/types-chunks/{guardrail.d-DqHD-61O.d.ts → guardrail.d-DKTyEHJa.d.ts} +5 -5
  127. package/dist/types-chunks/{history-retrieval.d-CKU8zmt_.d.ts → history-retrieval.d-CHnL99S7.d.ts} +2 -2
  128. package/dist/types-chunks/{public-api.d-D7fXaWW8.d.ts → public-api.d-Dx3L3W4Z.d.ts} +7 -4
  129. package/dist/types-chunks/{repl.d-BYDb14s6.d.ts → repl.d-mO01olIa.d.ts} +5 -5
  130. package/dist/types-chunks/{resolver.d-CwHUwbdE.d.ts → resolver.d-DlXPjeyx.d.ts} +38 -6
  131. package/dist/types-chunks/{review-inbox.d-BSgFewyj.d.ts → review-inbox.d-WDSe1nze.d.ts} +1 -1
  132. package/dist/types-chunks/{run-manager.d-BVJSntpS.d.ts → run-manager.d-DENbbzgR.d.ts} +1 -1
  133. package/dist/types-chunks/{sdk-session-CqEG84oP.d.ts → sdk-session-DYHGXXH4.d.ts} +3 -3
  134. package/dist/types-chunks/{shell-command-sets.d-DqaaBwhU.d.ts → shell-command-sets.d-CjFqS4dp.d.ts} +1 -1
  135. package/dist/types-chunks/{side-query.d-iH0P9ASy.d.ts → side-query.d-j8A3sFv_.d.ts} +2 -2
  136. package/dist/types-chunks/{types.d-rUOjXych.d.ts → types.d-BbZjR0Wr.d.ts} +1 -1
  137. package/dist/types-chunks/{types.d-CaTXAW_Q.d.ts → types.d-BkTIKdH6.d.ts} +2 -2
  138. package/dist/types-chunks/{types.d-C5xQEFcj.d.ts → types.d-DncLrpu_.d.ts} +4 -4
  139. package/dist/types-chunks/{types.d-CK9A9rpP.d.ts → types.d-Xy0f3x4i.d.ts} +9 -0
  140. package/dist/types-chunks/{utils.d-CWUyLiXo.d.ts → utils.d-CnxefMos.d.ts} +6 -6
  141. package/package.json +10 -3
  142. package/public_docs/README.md +130 -0
  143. package/public_docs/configuration/sandbox.md +289 -0
  144. package/public_docs/sdk/embedder-guide.md +6674 -0
  145. package/scripts/kodax-bin.cjs +28 -28
  146. package/scripts/production-env.cjs +25 -25
  147. package/dist/chunks/agent-XXZMXKKA.js +0 -2
  148. package/dist/chunks/argument-completer-KCYDGTQW.js +0 -2
  149. package/dist/chunks/chunk-77PNP27P.js +0 -92
  150. package/dist/chunks/chunk-7QBWHCIZ.js +0 -2
  151. package/dist/chunks/chunk-BHE66SP3.js +0 -123
  152. package/dist/chunks/chunk-BR2OSB2I.js +0 -722
  153. package/dist/chunks/chunk-DSMONSVB.js +0 -407
  154. package/dist/chunks/chunk-Q5VRWSKG.js +0 -418
  155. package/dist/chunks/chunk-XWRPOGJN.js +0 -4
  156. package/dist/chunks/chunk-Y5VIQBT6.js +0 -383
  157. package/dist/chunks/compaction-config-EGS4577X.js +0 -2
  158. package/dist/chunks/dist-ITTXQHRJ.js +0 -2
  159. package/dist/chunks/dist-KFYN3OKD.js +0 -2
  160. package/dist/chunks/host-VAYADIPW.js +0 -2
  161. package/dist/chunks/run-manager-52IJZMZJ.js +0 -2
  162. package/dist/chunks/utils-P7PBSL3R.js +0 -2
  163. package/dist/sandbox-workspace-session.js +0 -1705
@@ -1,227 +1,227 @@
1
- # Skill Creator Schemas
2
-
3
- 这份参考文档定义 KodaX 版 `skill-creator` 默认使用的评测文件格式。它不是强制协议,但建议优先沿用,方便后续聚合、review 和自动分析。
4
-
5
- ## `evals/evals.json`
6
-
7
- 用于保存测试提示集合。
8
-
9
- ```json
10
- {
11
- "skill_name": "example-skill",
12
- "evals": [
13
- {
14
- "id": 1,
15
- "prompt": "User task prompt",
16
- "expected_output": "What a good result should achieve",
17
- "files": [],
18
- "assertions": []
19
- }
20
- ]
21
- }
22
- ```
23
-
24
- 字段说明:
25
- - `skill_name`: skill 名称。
26
- - `evals`: 测试用例数组。
27
- - `id`: 用例唯一标识。
28
- - `prompt`: 给 agent 的任务文本。
29
- - `expected_output`: 对预期结果的简短说明。
30
- - `files`: 需要作为输入提供的文件列表。
31
- - `assertions`: 可选,后续 grading 用的断言定义。
32
-
33
- ## `eval_metadata.json`
34
-
35
- 用于单个 eval 目录,帮助 review 工具识别 prompt 和断言。
36
-
37
- ```json
38
- {
39
- "eval_id": 1,
40
- "eval_name": "handles-empty-input",
41
- "prompt": "Implement validation for empty input",
42
- "expected_output": "Reject empty input with a clear message",
43
- "assertions": [
44
- {
45
- "text": "rejects empty input with a clear message"
46
- }
47
- ]
48
- }
49
- ```
50
-
51
- ## `grading.json`
52
-
53
- 由 `grade-evals.js` 生成,用于保存单次运行后的断言判定结果。
54
-
55
- ```json
56
- {
57
- "summary": {
58
- "passed": 2,
59
- "failed": 1,
60
- "total": 3,
61
- "pass_rate": 0.6667
62
- },
63
- "expectations": [
64
- {
65
- "text": "rejects empty input with a clear message",
66
- "passed": true,
67
- "evidence": "Observed in outputs/result.md"
68
- }
69
- ],
70
- "execution_metrics": {
71
- "total_tool_calls": 4,
72
- "errors_encountered": 0,
73
- "output_chars": 5120
74
- },
75
- "user_notes_summary": {
76
- "uncertainties": [],
77
- "needs_review": [],
78
- "workarounds": []
79
- },
80
- "overall_summary": "Mostly correct, but edge cases need review.",
81
- "timing": {
82
- "total_tokens": 84852,
83
- "total_duration_seconds": 23.3
84
- },
85
- "meta": {
86
- "generated_at": "2026-03-17T12:00:00.000Z",
87
- "eval_id": 1,
88
- "eval_name": "handles-empty-input",
89
- "config": "with_skill",
90
- "run_id": "run-1"
91
- }
92
- }
93
- ```
94
-
95
- 要求:
96
- - `expectations` 里的字段名固定为 `text`、`passed`、`evidence`。
97
- - `pass_rate` 建议是 `0..1` 之间的小数。
98
-
99
- ## `timing.json`
100
-
101
- 用于保存一次运行的耗时与 token 信息。
102
-
103
- ```json
104
- {
105
- "total_tokens": 84852,
106
- "duration_ms": 23332,
107
- "total_duration_seconds": 23.3
108
- }
109
- ```
110
-
111
- ## `benchmark.json`
112
-
113
- 由 `aggregate-benchmark.js` 生成,用于总览不同配置的表现。
114
-
115
- ```json
116
- {
117
- "skill_name": "example-skill",
118
- "generated_at": "2026-03-17T12:00:00.000Z",
119
- "workspace": "/abs/path/to/iteration-1",
120
- "configs": {
121
- "with_skill": {
122
- "pass_rate": { "mean": 0.9, "stddev": 0.1, "min": 0.8, "max": 1.0 },
123
- "time_seconds": { "mean": 12.4, "stddev": 1.1, "min": 11.2, "max": 13.5 },
124
- "tokens": { "mean": 4200, "stddev": 380, "min": 3900, "max": 4700 }
125
- },
126
- "without_skill": {
127
- "pass_rate": { "mean": 0.6, "stddev": 0.2, "min": 0.4, "max": 0.8 },
128
- "time_seconds": { "mean": 9.5, "stddev": 0.7, "min": 8.9, "max": 10.2 },
129
- "tokens": { "mean": 3100, "stddev": 240, "min": 2900, "max": 3400 }
130
- }
131
- },
132
- "delta": {
133
- "pass_rate": "+0.3000",
134
- "time_seconds": "+2.9000",
135
- "tokens": "+1100.0000"
136
- },
137
- "runs": {
138
- "with_skill": [],
139
- "without_skill": []
140
- }
141
- }
142
- ```
143
-
144
- ## `analysis.json`
145
-
146
- 由 `analyze-benchmark.js` 生成,用于总结 benchmark 的稳定收益、方差热点和下一步建议。
147
-
148
- ```json
149
- {
150
- "skill_name": "example-skill",
151
- "generated_at": "2026-03-17T12:15:00.000Z",
152
- "workspace": "/abs/path/to/iteration-1",
153
- "verdict": "improves",
154
- "release_readiness": "needs_iteration",
155
- "recommendation": "Keep the skill, but reduce variance before release.",
156
- "key_findings": [
157
- "with_skill materially improves pass rate"
158
- ],
159
- "variance_hotspots": [
160
- "baseline repeatedly misses billing details"
161
- ],
162
- "suggested_actions": [
163
- "tighten assertions around billing coverage"
164
- ],
165
- "watchouts": [
166
- "token cost increased"
167
- ],
168
- "supporting_metrics": {
169
- "pass_rate_delta": "+0.3000",
170
- "time_seconds_delta": "+2.9000",
171
- "tokens_delta": "+1100.0000"
172
- },
173
- "failure_clusters": {}
174
- }
175
- ```
176
-
177
- ## `comparison.json`
178
-
179
- 由 `compare-runs.js` 生成,用于 blind comparison 两个 config 的输出质量。
180
-
181
- ```json
182
- {
183
- "workspace": "/abs/path/to/iteration-1",
184
- "generated_at": "2026-03-17T12:20:00.000Z",
185
- "config_a": "with_skill",
186
- "config_b": "without_skill",
187
- "summary": {
188
- "total_pairs": 3,
189
- "config_a_wins": 2,
190
- "config_b_wins": 0,
191
- "ties": 1,
192
- "inconclusive": 0
193
- },
194
- "comparisons": [
195
- {
196
- "eval_id": 1,
197
- "winner_label": "A",
198
- "winner_config": "with_skill",
199
- "confidence": 0.9,
200
- "rationale": "Candidate A is more complete and specific."
201
- }
202
- ]
203
- }
204
- ```
205
-
206
- ## 推荐目录结构
207
-
208
- ```text
209
- my-skill-workspace/
210
- └── iteration-1/
211
- ├── eval-0/
212
- │ ├── eval_metadata.json
213
- │ ├── with_skill/
214
- │ │ ├── outputs/
215
- │ │ ├── grading.json
216
- │ │ └── timing.json
217
- │ └── without_skill/
218
- │ ├── outputs/
219
- │ ├── grading.json
220
- │ └── timing.json
221
- ├── benchmark.json
222
- ├── benchmark.md
223
- ├── analysis.json
224
- ├── analysis.md
225
- ├── comparison.json
226
- └── comparison.md
227
- ```
1
+ # Skill Creator Schemas
2
+
3
+ 这份参考文档定义 KodaX 版 `skill-creator` 默认使用的评测文件格式。它不是强制协议,但建议优先沿用,方便后续聚合、review 和自动分析。
4
+
5
+ ## `evals/evals.json`
6
+
7
+ 用于保存测试提示集合。
8
+
9
+ ```json
10
+ {
11
+ "skill_name": "example-skill",
12
+ "evals": [
13
+ {
14
+ "id": 1,
15
+ "prompt": "User task prompt",
16
+ "expected_output": "What a good result should achieve",
17
+ "files": [],
18
+ "assertions": []
19
+ }
20
+ ]
21
+ }
22
+ ```
23
+
24
+ 字段说明:
25
+ - `skill_name`: skill 名称。
26
+ - `evals`: 测试用例数组。
27
+ - `id`: 用例唯一标识。
28
+ - `prompt`: 给 agent 的任务文本。
29
+ - `expected_output`: 对预期结果的简短说明。
30
+ - `files`: 需要作为输入提供的文件列表。
31
+ - `assertions`: 可选,后续 grading 用的断言定义。
32
+
33
+ ## `eval_metadata.json`
34
+
35
+ 用于单个 eval 目录,帮助 review 工具识别 prompt 和断言。
36
+
37
+ ```json
38
+ {
39
+ "eval_id": 1,
40
+ "eval_name": "handles-empty-input",
41
+ "prompt": "Implement validation for empty input",
42
+ "expected_output": "Reject empty input with a clear message",
43
+ "assertions": [
44
+ {
45
+ "text": "rejects empty input with a clear message"
46
+ }
47
+ ]
48
+ }
49
+ ```
50
+
51
+ ## `grading.json`
52
+
53
+ 由 `grade-evals.js` 生成,用于保存单次运行后的断言判定结果。
54
+
55
+ ```json
56
+ {
57
+ "summary": {
58
+ "passed": 2,
59
+ "failed": 1,
60
+ "total": 3,
61
+ "pass_rate": 0.6667
62
+ },
63
+ "expectations": [
64
+ {
65
+ "text": "rejects empty input with a clear message",
66
+ "passed": true,
67
+ "evidence": "Observed in outputs/result.md"
68
+ }
69
+ ],
70
+ "execution_metrics": {
71
+ "total_tool_calls": 4,
72
+ "errors_encountered": 0,
73
+ "output_chars": 5120
74
+ },
75
+ "user_notes_summary": {
76
+ "uncertainties": [],
77
+ "needs_review": [],
78
+ "workarounds": []
79
+ },
80
+ "overall_summary": "Mostly correct, but edge cases need review.",
81
+ "timing": {
82
+ "total_tokens": 84852,
83
+ "total_duration_seconds": 23.3
84
+ },
85
+ "meta": {
86
+ "generated_at": "2026-03-17T12:00:00.000Z",
87
+ "eval_id": 1,
88
+ "eval_name": "handles-empty-input",
89
+ "config": "with_skill",
90
+ "run_id": "run-1"
91
+ }
92
+ }
93
+ ```
94
+
95
+ 要求:
96
+ - `expectations` 里的字段名固定为 `text`、`passed`、`evidence`。
97
+ - `pass_rate` 建议是 `0..1` 之间的小数。
98
+
99
+ ## `timing.json`
100
+
101
+ 用于保存一次运行的耗时与 token 信息。
102
+
103
+ ```json
104
+ {
105
+ "total_tokens": 84852,
106
+ "duration_ms": 23332,
107
+ "total_duration_seconds": 23.3
108
+ }
109
+ ```
110
+
111
+ ## `benchmark.json`
112
+
113
+ 由 `aggregate-benchmark.js` 生成,用于总览不同配置的表现。
114
+
115
+ ```json
116
+ {
117
+ "skill_name": "example-skill",
118
+ "generated_at": "2026-03-17T12:00:00.000Z",
119
+ "workspace": "/abs/path/to/iteration-1",
120
+ "configs": {
121
+ "with_skill": {
122
+ "pass_rate": { "mean": 0.9, "stddev": 0.1, "min": 0.8, "max": 1.0 },
123
+ "time_seconds": { "mean": 12.4, "stddev": 1.1, "min": 11.2, "max": 13.5 },
124
+ "tokens": { "mean": 4200, "stddev": 380, "min": 3900, "max": 4700 }
125
+ },
126
+ "without_skill": {
127
+ "pass_rate": { "mean": 0.6, "stddev": 0.2, "min": 0.4, "max": 0.8 },
128
+ "time_seconds": { "mean": 9.5, "stddev": 0.7, "min": 8.9, "max": 10.2 },
129
+ "tokens": { "mean": 3100, "stddev": 240, "min": 2900, "max": 3400 }
130
+ }
131
+ },
132
+ "delta": {
133
+ "pass_rate": "+0.3000",
134
+ "time_seconds": "+2.9000",
135
+ "tokens": "+1100.0000"
136
+ },
137
+ "runs": {
138
+ "with_skill": [],
139
+ "without_skill": []
140
+ }
141
+ }
142
+ ```
143
+
144
+ ## `analysis.json`
145
+
146
+ 由 `analyze-benchmark.js` 生成,用于总结 benchmark 的稳定收益、方差热点和下一步建议。
147
+
148
+ ```json
149
+ {
150
+ "skill_name": "example-skill",
151
+ "generated_at": "2026-03-17T12:15:00.000Z",
152
+ "workspace": "/abs/path/to/iteration-1",
153
+ "verdict": "improves",
154
+ "release_readiness": "needs_iteration",
155
+ "recommendation": "Keep the skill, but reduce variance before release.",
156
+ "key_findings": [
157
+ "with_skill materially improves pass rate"
158
+ ],
159
+ "variance_hotspots": [
160
+ "baseline repeatedly misses billing details"
161
+ ],
162
+ "suggested_actions": [
163
+ "tighten assertions around billing coverage"
164
+ ],
165
+ "watchouts": [
166
+ "token cost increased"
167
+ ],
168
+ "supporting_metrics": {
169
+ "pass_rate_delta": "+0.3000",
170
+ "time_seconds_delta": "+2.9000",
171
+ "tokens_delta": "+1100.0000"
172
+ },
173
+ "failure_clusters": {}
174
+ }
175
+ ```
176
+
177
+ ## `comparison.json`
178
+
179
+ 由 `compare-runs.js` 生成,用于 blind comparison 两个 config 的输出质量。
180
+
181
+ ```json
182
+ {
183
+ "workspace": "/abs/path/to/iteration-1",
184
+ "generated_at": "2026-03-17T12:20:00.000Z",
185
+ "config_a": "with_skill",
186
+ "config_b": "without_skill",
187
+ "summary": {
188
+ "total_pairs": 3,
189
+ "config_a_wins": 2,
190
+ "config_b_wins": 0,
191
+ "ties": 1,
192
+ "inconclusive": 0
193
+ },
194
+ "comparisons": [
195
+ {
196
+ "eval_id": 1,
197
+ "winner_label": "A",
198
+ "winner_config": "with_skill",
199
+ "confidence": 0.9,
200
+ "rationale": "Candidate A is more complete and specific."
201
+ }
202
+ ]
203
+ }
204
+ ```
205
+
206
+ ## 推荐目录结构
207
+
208
+ ```text
209
+ my-skill-workspace/
210
+ └── iteration-1/
211
+ ├── eval-0/
212
+ │ ├── eval_metadata.json
213
+ │ ├── with_skill/
214
+ │ │ ├── outputs/
215
+ │ │ ├── grading.json
216
+ │ │ └── timing.json
217
+ │ └── without_skill/
218
+ │ ├── outputs/
219
+ │ ├── grading.json
220
+ │ └── timing.json
221
+ ├── benchmark.json
222
+ ├── benchmark.md
223
+ ├── analysis.json
224
+ ├── analysis.md
225
+ ├── comparison.json
226
+ └── comparison.md
227
+ ```
@@ -1,46 +1,46 @@
1
- export interface BenchmarkRun {
2
- eval_id: string | number;
3
- run_id: string;
4
- pass_rate: number;
5
- passed: number;
6
- failed: number;
7
- total: number;
8
- time_seconds: number;
9
- tokens: number;
10
- tool_calls: number;
11
- errors: number;
12
- expectations: Array<Record<string, unknown>>;
13
- notes: string[];
14
- }
15
-
16
- export interface StatsSummary {
17
- mean: number;
18
- stddev: number;
19
- min: number;
20
- max: number;
21
- }
22
-
23
- export interface BenchmarkDocument {
24
- skill_name: string;
25
- generated_at: string;
26
- workspace: string;
27
- configs: Record<string, {
28
- pass_rate: StatsSummary;
29
- time_seconds: StatsSummary;
30
- tokens: StatsSummary;
31
- }>;
32
- delta: {
33
- pass_rate: string;
34
- time_seconds: string;
35
- tokens: string;
36
- };
37
- runs: Record<string, BenchmarkRun[]>;
38
- }
39
-
40
- export function loadRunResults(iterationDir: string): Promise<Record<string, BenchmarkRun[]>>;
41
- export function buildBenchmarkDocument(
42
- iterationDir: string,
43
- skillName: string,
44
- configRuns: Record<string, BenchmarkRun[]>
45
- ): BenchmarkDocument;
46
- export function renderBenchmarkMarkdown(benchmark: BenchmarkDocument): string;
1
+ export interface BenchmarkRun {
2
+ eval_id: string | number;
3
+ run_id: string;
4
+ pass_rate: number;
5
+ passed: number;
6
+ failed: number;
7
+ total: number;
8
+ time_seconds: number;
9
+ tokens: number;
10
+ tool_calls: number;
11
+ errors: number;
12
+ expectations: Array<Record<string, unknown>>;
13
+ notes: string[];
14
+ }
15
+
16
+ export interface StatsSummary {
17
+ mean: number;
18
+ stddev: number;
19
+ min: number;
20
+ max: number;
21
+ }
22
+
23
+ export interface BenchmarkDocument {
24
+ skill_name: string;
25
+ generated_at: string;
26
+ workspace: string;
27
+ configs: Record<string, {
28
+ pass_rate: StatsSummary;
29
+ time_seconds: StatsSummary;
30
+ tokens: StatsSummary;
31
+ }>;
32
+ delta: {
33
+ pass_rate: string;
34
+ time_seconds: string;
35
+ tokens: string;
36
+ };
37
+ runs: Record<string, BenchmarkRun[]>;
38
+ }
39
+
40
+ export function loadRunResults(iterationDir: string): Promise<Record<string, BenchmarkRun[]>>;
41
+ export function buildBenchmarkDocument(
42
+ iterationDir: string,
43
+ skillName: string,
44
+ configRuns: Record<string, BenchmarkRun[]>
45
+ ): BenchmarkDocument;
46
+ export function renderBenchmarkMarkdown(benchmark: BenchmarkDocument): string;