@kodax-ai/kodax 0.7.95 → 0.7.96-alpha.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +178 -1
- package/LICENSE +158 -158
- package/README.md +149 -24
- package/README_CN.md +117 -24
- package/config-templates/config.example.jsonc +2 -1
- package/config-templates/integrations/a2a.example.jsonc +91 -91
- package/config-templates/integrations/extensions.example.jsonc +7 -7
- package/config-templates/integrations/mcp.example.jsonc +16 -16
- package/dist/builtin/code-review/SKILL.md +82 -82
- package/dist/builtin/git-workflow/SKILL.md +84 -84
- package/dist/builtin/skill-creator/SKILL.md +127 -127
- package/dist/builtin/skill-creator/agents/analyzer.md +12 -12
- package/dist/builtin/skill-creator/agents/comparator.md +13 -13
- package/dist/builtin/skill-creator/agents/grader.md +13 -13
- package/dist/builtin/skill-creator/references/schemas.md +227 -227
- package/dist/builtin/skill-creator/scripts/aggregate-benchmark.d.ts +46 -46
- package/dist/builtin/skill-creator/scripts/aggregate-benchmark.js +208 -208
- package/dist/builtin/skill-creator/scripts/analyze-benchmark.d.ts +46 -46
- package/dist/builtin/skill-creator/scripts/analyze-benchmark.js +286 -286
- package/dist/builtin/skill-creator/scripts/compare-runs.d.ts +62 -62
- package/dist/builtin/skill-creator/scripts/compare-runs.js +330 -330
- package/dist/builtin/skill-creator/scripts/generate-review.d.ts +33 -33
- package/dist/builtin/skill-creator/scripts/generate-review.js +414 -414
- package/dist/builtin/skill-creator/scripts/grade-evals.d.ts +73 -73
- package/dist/builtin/skill-creator/scripts/grade-evals.js +402 -402
- package/dist/builtin/skill-creator/scripts/improve-description.d.ts +23 -23
- package/dist/builtin/skill-creator/scripts/improve-description.js +160 -160
- package/dist/builtin/skill-creator/scripts/init-skill.d.ts +14 -14
- package/dist/builtin/skill-creator/scripts/init-skill.js +155 -155
- package/dist/builtin/skill-creator/scripts/install-skill.d.ts +29 -29
- package/dist/builtin/skill-creator/scripts/install-skill.js +173 -173
- package/dist/builtin/skill-creator/scripts/package-skill.d.ts +38 -38
- package/dist/builtin/skill-creator/scripts/package-skill.js +121 -121
- package/dist/builtin/skill-creator/scripts/quick-validate.d.ts +8 -8
- package/dist/builtin/skill-creator/scripts/quick-validate.js +163 -163
- package/dist/builtin/skill-creator/scripts/run-eval.d.ts +66 -66
- package/dist/builtin/skill-creator/scripts/run-eval.js +353 -353
- package/dist/builtin/skill-creator/scripts/run-loop.d.ts +49 -49
- package/dist/builtin/skill-creator/scripts/run-loop.js +242 -242
- package/dist/builtin/skill-creator/scripts/run-trigger-eval.d.ts +58 -58
- package/dist/builtin/skill-creator/scripts/run-trigger-eval.js +224 -224
- package/dist/builtin/tdd/SKILL.md +56 -56
- package/dist/chunks/agent-4FABF6WK.js +2 -0
- package/dist/chunks/argument-completer-72VHB2VL.js +2 -0
- package/dist/chunks/{chunk-7JXA3533.js → chunk-4OIQZWA5.js} +223 -226
- package/dist/chunks/{chunk-BNVSKIAB.js → chunk-6BVF3HZ7.js} +136 -136
- package/dist/chunks/chunk-7JZZIFZH.js +501 -0
- package/dist/chunks/chunk-AC7Z2OIY.js +2 -0
- package/dist/chunks/{chunk-JS452J2F.js → chunk-C4KYZAFU.js} +2 -2
- package/dist/chunks/chunk-CHUPIWIF.js +251 -0
- package/dist/chunks/{chunk-C4DTUKTR.js → chunk-GYUU7XLH.js} +5 -4
- package/dist/chunks/{chunk-OK4AXDC5.js → chunk-HAX55GOQ.js} +1 -1
- package/dist/chunks/chunk-JKTJCGO2.js +418 -0
- package/dist/chunks/{chunk-O743AJ5V.js → chunk-LQYETRFK.js} +171 -175
- package/dist/chunks/{chunk-2PEDBGKN.js → chunk-LTUZTQ6W.js} +1 -1
- package/dist/chunks/chunk-LYWDUSKO.js +2 -0
- package/dist/chunks/chunk-NGH6T4FD.js +92 -0
- package/dist/chunks/{chunk-SRIJX5ZP.js → chunk-NVRWCSA5.js} +1 -1
- package/dist/chunks/chunk-O4DS7RIQ.js +384 -0
- package/dist/chunks/chunk-ORXU6OWZ.js +2 -0
- package/dist/chunks/chunk-POTW3O65.js +124 -0
- package/dist/chunks/{chunk-2TM3W3GK.js → chunk-RIQSS56Y.js} +1 -1
- package/dist/chunks/chunk-U27XKEN5.js +407 -0
- package/dist/chunks/{chunk-6ABQJUK6.js → chunk-XUX6OUCA.js} +2 -2
- package/dist/chunks/{client-JDOS3YJR.js → client-4C456K4O.js} +1 -1
- package/dist/chunks/compaction-config-3VQRDFF4.js +2 -0
- package/dist/chunks/{construction-bootstrap-YCWHHUYO.js → construction-bootstrap-AHN7EDGG.js} +1 -1
- package/dist/chunks/dist-BCBQXAGI.js +2 -0
- package/dist/chunks/dist-QK2YDWU6.js +2 -0
- package/dist/chunks/host-VIHV7ADG.js +2 -0
- package/dist/chunks/run-manager-NBVMU3KV.js +2 -0
- package/dist/chunks/utils-MEJZ2IGI.js +2 -0
- package/dist/index.d.ts +22 -21
- package/dist/index.js +7 -7
- package/dist/kodax_bootstrap.js +26 -26
- package/dist/kodax_cli.js +1905 -1882
- package/dist/native/darwin-arm64/LICENSE-APACHE.txt +13 -0
- package/dist/native/darwin-arm64/kodax-text-transaction.node +0 -0
- package/dist/native/darwin-arm64/manifest.json +16 -0
- package/dist/native/darwin-x64/LICENSE-APACHE.txt +13 -0
- package/dist/native/darwin-x64/kodax-text-transaction.node +0 -0
- package/dist/native/darwin-x64/manifest.json +16 -0
- package/dist/native/linux-arm64/LICENSE-APACHE.txt +13 -0
- package/dist/native/linux-arm64/kodax-text-transaction.node +0 -0
- package/dist/native/linux-arm64/manifest.json +16 -0
- package/dist/native/linux-x64/LICENSE-APACHE.txt +13 -0
- package/dist/native/linux-x64/kodax-text-transaction.node +0 -0
- package/dist/native/linux-x64/manifest.json +16 -0
- package/dist/native/win32-x64/LICENSE-APACHE.txt +13 -0
- package/dist/native/win32-x64/NOTICE-windows-sandbox.txt +8 -0
- package/dist/native/win32-x64/kodax-windows-sandbox.exe +0 -0
- package/dist/native/win32-x64/kodax-windows-text-transaction.node +0 -0
- package/dist/native/win32-x64/manifest.json +30 -0
- package/dist/provider-capabilities.json +108 -2
- package/dist/runtime-worker.js +1745 -1727
- package/dist/sandbox-network-broker.js +1715 -0
- package/dist/sdk-a2a.d.ts +13 -12
- package/dist/sdk-a2a.js +1 -1
- package/dist/sdk-agent.d.ts +27 -18
- package/dist/sdk-agent.js +1 -1
- package/dist/sdk-coding.d.ts +107 -58
- package/dist/sdk-coding.js +1 -1
- package/dist/sdk-experimental-memory.d.ts +4 -4
- package/dist/sdk-experimental-memory.js +1 -1
- package/dist/sdk-llm.d.ts +9 -10
- package/dist/sdk-llm.js +1 -1
- package/dist/sdk-mcp.js +1 -1
- package/dist/sdk-media.d.ts +1 -1
- package/dist/sdk-media.js +1 -1
- package/dist/sdk-repl.d.ts +25 -17
- package/dist/sdk-repl.js +1 -1
- package/dist/sdk-runtime.d.ts +41 -17
- package/dist/sdk-runtime.js +1 -1
- package/dist/sdk-sandbox.d.ts +17 -3
- package/dist/sdk-sandbox.js +1 -1
- package/dist/sdk-session.d.ts +8 -7
- package/dist/sdk-session.js +1 -1
- package/dist/sdk-skills.js +1 -1
- package/dist/semantic-worker.js +61 -61
- package/dist/types-chunks/{base.d-vopO1rml.d.ts → base.d-COo8E2dk.d.ts} +18 -27
- package/dist/types-chunks/{bash-prefix-extractor.d-B2rxNI4L.d.ts → bash-prefix-extractor.d-DQOlA8cy.d.ts} +72 -52
- package/dist/types-chunks/{capability-learning.d-CE1tm29C.d.ts → capability-learning.d-lRPgA3Or.d.ts} +1 -1
- package/dist/types-chunks/{capsule.d-Y7QRS4J2.d.ts → capsule.d-0zxsgXg_.d.ts} +3 -3
- package/dist/types-chunks/{controller.d-BR-UVeUR.d.ts → controller.d-BTtyIIC1.d.ts} +1 -1
- package/dist/types-chunks/{controller.d-C5noq5-w.d.ts → controller.d-C7tXUcXo.d.ts} +2 -2
- package/dist/types-chunks/{guardrail.d-DqHD-61O.d.ts → guardrail.d-DKTyEHJa.d.ts} +5 -5
- package/dist/types-chunks/{history-retrieval.d-CKU8zmt_.d.ts → history-retrieval.d-CHnL99S7.d.ts} +2 -2
- package/dist/types-chunks/{public-api.d-D7fXaWW8.d.ts → public-api.d-Dx3L3W4Z.d.ts} +7 -4
- package/dist/types-chunks/{repl.d-BYDb14s6.d.ts → repl.d-mO01olIa.d.ts} +5 -5
- package/dist/types-chunks/{resolver.d-CwHUwbdE.d.ts → resolver.d-DlXPjeyx.d.ts} +38 -6
- package/dist/types-chunks/{review-inbox.d-BSgFewyj.d.ts → review-inbox.d-WDSe1nze.d.ts} +1 -1
- package/dist/types-chunks/{run-manager.d-BVJSntpS.d.ts → run-manager.d-DENbbzgR.d.ts} +1 -1
- package/dist/types-chunks/{sdk-session-CqEG84oP.d.ts → sdk-session-DYHGXXH4.d.ts} +3 -3
- package/dist/types-chunks/{shell-command-sets.d-DqaaBwhU.d.ts → shell-command-sets.d-CjFqS4dp.d.ts} +1 -1
- package/dist/types-chunks/{side-query.d-iH0P9ASy.d.ts → side-query.d-j8A3sFv_.d.ts} +2 -2
- package/dist/types-chunks/{types.d-rUOjXych.d.ts → types.d-BbZjR0Wr.d.ts} +1 -1
- package/dist/types-chunks/{types.d-CaTXAW_Q.d.ts → types.d-BkTIKdH6.d.ts} +2 -2
- package/dist/types-chunks/{types.d-C5xQEFcj.d.ts → types.d-DncLrpu_.d.ts} +4 -4
- package/dist/types-chunks/{types.d-CK9A9rpP.d.ts → types.d-Xy0f3x4i.d.ts} +9 -0
- package/dist/types-chunks/{utils.d-CWUyLiXo.d.ts → utils.d-CnxefMos.d.ts} +6 -6
- package/package.json +10 -3
- package/public_docs/README.md +130 -0
- package/public_docs/configuration/sandbox.md +289 -0
- package/public_docs/sdk/embedder-guide.md +6674 -0
- package/scripts/kodax-bin.cjs +28 -28
- package/scripts/production-env.cjs +25 -25
- package/dist/chunks/agent-XXZMXKKA.js +0 -2
- package/dist/chunks/argument-completer-KCYDGTQW.js +0 -2
- package/dist/chunks/chunk-77PNP27P.js +0 -92
- package/dist/chunks/chunk-7QBWHCIZ.js +0 -2
- package/dist/chunks/chunk-BHE66SP3.js +0 -123
- package/dist/chunks/chunk-BR2OSB2I.js +0 -722
- package/dist/chunks/chunk-DSMONSVB.js +0 -407
- package/dist/chunks/chunk-Q5VRWSKG.js +0 -418
- package/dist/chunks/chunk-XWRPOGJN.js +0 -4
- package/dist/chunks/chunk-Y5VIQBT6.js +0 -383
- package/dist/chunks/compaction-config-EGS4577X.js +0 -2
- package/dist/chunks/dist-ITTXQHRJ.js +0 -2
- package/dist/chunks/dist-KFYN3OKD.js +0 -2
- package/dist/chunks/host-VAYADIPW.js +0 -2
- package/dist/chunks/run-manager-52IJZMZJ.js +0 -2
- package/dist/chunks/utils-P7PBSL3R.js +0 -2
- package/dist/sandbox-workspace-session.js +0 -1705
|
@@ -1,227 +1,227 @@
|
|
|
1
|
-
# Skill Creator Schemas
|
|
2
|
-
|
|
3
|
-
这份参考文档定义 KodaX 版 `skill-creator` 默认使用的评测文件格式。它不是强制协议,但建议优先沿用,方便后续聚合、review 和自动分析。
|
|
4
|
-
|
|
5
|
-
## `evals/evals.json`
|
|
6
|
-
|
|
7
|
-
用于保存测试提示集合。
|
|
8
|
-
|
|
9
|
-
```json
|
|
10
|
-
{
|
|
11
|
-
"skill_name": "example-skill",
|
|
12
|
-
"evals": [
|
|
13
|
-
{
|
|
14
|
-
"id": 1,
|
|
15
|
-
"prompt": "User task prompt",
|
|
16
|
-
"expected_output": "What a good result should achieve",
|
|
17
|
-
"files": [],
|
|
18
|
-
"assertions": []
|
|
19
|
-
}
|
|
20
|
-
]
|
|
21
|
-
}
|
|
22
|
-
```
|
|
23
|
-
|
|
24
|
-
字段说明:
|
|
25
|
-
- `skill_name`: skill 名称。
|
|
26
|
-
- `evals`: 测试用例数组。
|
|
27
|
-
- `id`: 用例唯一标识。
|
|
28
|
-
- `prompt`: 给 agent 的任务文本。
|
|
29
|
-
- `expected_output`: 对预期结果的简短说明。
|
|
30
|
-
- `files`: 需要作为输入提供的文件列表。
|
|
31
|
-
- `assertions`: 可选,后续 grading 用的断言定义。
|
|
32
|
-
|
|
33
|
-
## `eval_metadata.json`
|
|
34
|
-
|
|
35
|
-
用于单个 eval 目录,帮助 review 工具识别 prompt 和断言。
|
|
36
|
-
|
|
37
|
-
```json
|
|
38
|
-
{
|
|
39
|
-
"eval_id": 1,
|
|
40
|
-
"eval_name": "handles-empty-input",
|
|
41
|
-
"prompt": "Implement validation for empty input",
|
|
42
|
-
"expected_output": "Reject empty input with a clear message",
|
|
43
|
-
"assertions": [
|
|
44
|
-
{
|
|
45
|
-
"text": "rejects empty input with a clear message"
|
|
46
|
-
}
|
|
47
|
-
]
|
|
48
|
-
}
|
|
49
|
-
```
|
|
50
|
-
|
|
51
|
-
## `grading.json`
|
|
52
|
-
|
|
53
|
-
由 `grade-evals.js` 生成,用于保存单次运行后的断言判定结果。
|
|
54
|
-
|
|
55
|
-
```json
|
|
56
|
-
{
|
|
57
|
-
"summary": {
|
|
58
|
-
"passed": 2,
|
|
59
|
-
"failed": 1,
|
|
60
|
-
"total": 3,
|
|
61
|
-
"pass_rate": 0.6667
|
|
62
|
-
},
|
|
63
|
-
"expectations": [
|
|
64
|
-
{
|
|
65
|
-
"text": "rejects empty input with a clear message",
|
|
66
|
-
"passed": true,
|
|
67
|
-
"evidence": "Observed in outputs/result.md"
|
|
68
|
-
}
|
|
69
|
-
],
|
|
70
|
-
"execution_metrics": {
|
|
71
|
-
"total_tool_calls": 4,
|
|
72
|
-
"errors_encountered": 0,
|
|
73
|
-
"output_chars": 5120
|
|
74
|
-
},
|
|
75
|
-
"user_notes_summary": {
|
|
76
|
-
"uncertainties": [],
|
|
77
|
-
"needs_review": [],
|
|
78
|
-
"workarounds": []
|
|
79
|
-
},
|
|
80
|
-
"overall_summary": "Mostly correct, but edge cases need review.",
|
|
81
|
-
"timing": {
|
|
82
|
-
"total_tokens": 84852,
|
|
83
|
-
"total_duration_seconds": 23.3
|
|
84
|
-
},
|
|
85
|
-
"meta": {
|
|
86
|
-
"generated_at": "2026-03-17T12:00:00.000Z",
|
|
87
|
-
"eval_id": 1,
|
|
88
|
-
"eval_name": "handles-empty-input",
|
|
89
|
-
"config": "with_skill",
|
|
90
|
-
"run_id": "run-1"
|
|
91
|
-
}
|
|
92
|
-
}
|
|
93
|
-
```
|
|
94
|
-
|
|
95
|
-
要求:
|
|
96
|
-
- `expectations` 里的字段名固定为 `text`、`passed`、`evidence`。
|
|
97
|
-
- `pass_rate` 建议是 `0..1` 之间的小数。
|
|
98
|
-
|
|
99
|
-
## `timing.json`
|
|
100
|
-
|
|
101
|
-
用于保存一次运行的耗时与 token 信息。
|
|
102
|
-
|
|
103
|
-
```json
|
|
104
|
-
{
|
|
105
|
-
"total_tokens": 84852,
|
|
106
|
-
"duration_ms": 23332,
|
|
107
|
-
"total_duration_seconds": 23.3
|
|
108
|
-
}
|
|
109
|
-
```
|
|
110
|
-
|
|
111
|
-
## `benchmark.json`
|
|
112
|
-
|
|
113
|
-
由 `aggregate-benchmark.js` 生成,用于总览不同配置的表现。
|
|
114
|
-
|
|
115
|
-
```json
|
|
116
|
-
{
|
|
117
|
-
"skill_name": "example-skill",
|
|
118
|
-
"generated_at": "2026-03-17T12:00:00.000Z",
|
|
119
|
-
"workspace": "/abs/path/to/iteration-1",
|
|
120
|
-
"configs": {
|
|
121
|
-
"with_skill": {
|
|
122
|
-
"pass_rate": { "mean": 0.9, "stddev": 0.1, "min": 0.8, "max": 1.0 },
|
|
123
|
-
"time_seconds": { "mean": 12.4, "stddev": 1.1, "min": 11.2, "max": 13.5 },
|
|
124
|
-
"tokens": { "mean": 4200, "stddev": 380, "min": 3900, "max": 4700 }
|
|
125
|
-
},
|
|
126
|
-
"without_skill": {
|
|
127
|
-
"pass_rate": { "mean": 0.6, "stddev": 0.2, "min": 0.4, "max": 0.8 },
|
|
128
|
-
"time_seconds": { "mean": 9.5, "stddev": 0.7, "min": 8.9, "max": 10.2 },
|
|
129
|
-
"tokens": { "mean": 3100, "stddev": 240, "min": 2900, "max": 3400 }
|
|
130
|
-
}
|
|
131
|
-
},
|
|
132
|
-
"delta": {
|
|
133
|
-
"pass_rate": "+0.3000",
|
|
134
|
-
"time_seconds": "+2.9000",
|
|
135
|
-
"tokens": "+1100.0000"
|
|
136
|
-
},
|
|
137
|
-
"runs": {
|
|
138
|
-
"with_skill": [],
|
|
139
|
-
"without_skill": []
|
|
140
|
-
}
|
|
141
|
-
}
|
|
142
|
-
```
|
|
143
|
-
|
|
144
|
-
## `analysis.json`
|
|
145
|
-
|
|
146
|
-
由 `analyze-benchmark.js` 生成,用于总结 benchmark 的稳定收益、方差热点和下一步建议。
|
|
147
|
-
|
|
148
|
-
```json
|
|
149
|
-
{
|
|
150
|
-
"skill_name": "example-skill",
|
|
151
|
-
"generated_at": "2026-03-17T12:15:00.000Z",
|
|
152
|
-
"workspace": "/abs/path/to/iteration-1",
|
|
153
|
-
"verdict": "improves",
|
|
154
|
-
"release_readiness": "needs_iteration",
|
|
155
|
-
"recommendation": "Keep the skill, but reduce variance before release.",
|
|
156
|
-
"key_findings": [
|
|
157
|
-
"with_skill materially improves pass rate"
|
|
158
|
-
],
|
|
159
|
-
"variance_hotspots": [
|
|
160
|
-
"baseline repeatedly misses billing details"
|
|
161
|
-
],
|
|
162
|
-
"suggested_actions": [
|
|
163
|
-
"tighten assertions around billing coverage"
|
|
164
|
-
],
|
|
165
|
-
"watchouts": [
|
|
166
|
-
"token cost increased"
|
|
167
|
-
],
|
|
168
|
-
"supporting_metrics": {
|
|
169
|
-
"pass_rate_delta": "+0.3000",
|
|
170
|
-
"time_seconds_delta": "+2.9000",
|
|
171
|
-
"tokens_delta": "+1100.0000"
|
|
172
|
-
},
|
|
173
|
-
"failure_clusters": {}
|
|
174
|
-
}
|
|
175
|
-
```
|
|
176
|
-
|
|
177
|
-
## `comparison.json`
|
|
178
|
-
|
|
179
|
-
由 `compare-runs.js` 生成,用于 blind comparison 两个 config 的输出质量。
|
|
180
|
-
|
|
181
|
-
```json
|
|
182
|
-
{
|
|
183
|
-
"workspace": "/abs/path/to/iteration-1",
|
|
184
|
-
"generated_at": "2026-03-17T12:20:00.000Z",
|
|
185
|
-
"config_a": "with_skill",
|
|
186
|
-
"config_b": "without_skill",
|
|
187
|
-
"summary": {
|
|
188
|
-
"total_pairs": 3,
|
|
189
|
-
"config_a_wins": 2,
|
|
190
|
-
"config_b_wins": 0,
|
|
191
|
-
"ties": 1,
|
|
192
|
-
"inconclusive": 0
|
|
193
|
-
},
|
|
194
|
-
"comparisons": [
|
|
195
|
-
{
|
|
196
|
-
"eval_id": 1,
|
|
197
|
-
"winner_label": "A",
|
|
198
|
-
"winner_config": "with_skill",
|
|
199
|
-
"confidence": 0.9,
|
|
200
|
-
"rationale": "Candidate A is more complete and specific."
|
|
201
|
-
}
|
|
202
|
-
]
|
|
203
|
-
}
|
|
204
|
-
```
|
|
205
|
-
|
|
206
|
-
## 推荐目录结构
|
|
207
|
-
|
|
208
|
-
```text
|
|
209
|
-
my-skill-workspace/
|
|
210
|
-
└── iteration-1/
|
|
211
|
-
├── eval-0/
|
|
212
|
-
│ ├── eval_metadata.json
|
|
213
|
-
│ ├── with_skill/
|
|
214
|
-
│ │ ├── outputs/
|
|
215
|
-
│ │ ├── grading.json
|
|
216
|
-
│ │ └── timing.json
|
|
217
|
-
│ └── without_skill/
|
|
218
|
-
│ ├── outputs/
|
|
219
|
-
│ ├── grading.json
|
|
220
|
-
│ └── timing.json
|
|
221
|
-
├── benchmark.json
|
|
222
|
-
├── benchmark.md
|
|
223
|
-
├── analysis.json
|
|
224
|
-
├── analysis.md
|
|
225
|
-
├── comparison.json
|
|
226
|
-
└── comparison.md
|
|
227
|
-
```
|
|
1
|
+
# Skill Creator Schemas
|
|
2
|
+
|
|
3
|
+
这份参考文档定义 KodaX 版 `skill-creator` 默认使用的评测文件格式。它不是强制协议,但建议优先沿用,方便后续聚合、review 和自动分析。
|
|
4
|
+
|
|
5
|
+
## `evals/evals.json`
|
|
6
|
+
|
|
7
|
+
用于保存测试提示集合。
|
|
8
|
+
|
|
9
|
+
```json
|
|
10
|
+
{
|
|
11
|
+
"skill_name": "example-skill",
|
|
12
|
+
"evals": [
|
|
13
|
+
{
|
|
14
|
+
"id": 1,
|
|
15
|
+
"prompt": "User task prompt",
|
|
16
|
+
"expected_output": "What a good result should achieve",
|
|
17
|
+
"files": [],
|
|
18
|
+
"assertions": []
|
|
19
|
+
}
|
|
20
|
+
]
|
|
21
|
+
}
|
|
22
|
+
```
|
|
23
|
+
|
|
24
|
+
字段说明:
|
|
25
|
+
- `skill_name`: skill 名称。
|
|
26
|
+
- `evals`: 测试用例数组。
|
|
27
|
+
- `id`: 用例唯一标识。
|
|
28
|
+
- `prompt`: 给 agent 的任务文本。
|
|
29
|
+
- `expected_output`: 对预期结果的简短说明。
|
|
30
|
+
- `files`: 需要作为输入提供的文件列表。
|
|
31
|
+
- `assertions`: 可选,后续 grading 用的断言定义。
|
|
32
|
+
|
|
33
|
+
## `eval_metadata.json`
|
|
34
|
+
|
|
35
|
+
用于单个 eval 目录,帮助 review 工具识别 prompt 和断言。
|
|
36
|
+
|
|
37
|
+
```json
|
|
38
|
+
{
|
|
39
|
+
"eval_id": 1,
|
|
40
|
+
"eval_name": "handles-empty-input",
|
|
41
|
+
"prompt": "Implement validation for empty input",
|
|
42
|
+
"expected_output": "Reject empty input with a clear message",
|
|
43
|
+
"assertions": [
|
|
44
|
+
{
|
|
45
|
+
"text": "rejects empty input with a clear message"
|
|
46
|
+
}
|
|
47
|
+
]
|
|
48
|
+
}
|
|
49
|
+
```
|
|
50
|
+
|
|
51
|
+
## `grading.json`
|
|
52
|
+
|
|
53
|
+
由 `grade-evals.js` 生成,用于保存单次运行后的断言判定结果。
|
|
54
|
+
|
|
55
|
+
```json
|
|
56
|
+
{
|
|
57
|
+
"summary": {
|
|
58
|
+
"passed": 2,
|
|
59
|
+
"failed": 1,
|
|
60
|
+
"total": 3,
|
|
61
|
+
"pass_rate": 0.6667
|
|
62
|
+
},
|
|
63
|
+
"expectations": [
|
|
64
|
+
{
|
|
65
|
+
"text": "rejects empty input with a clear message",
|
|
66
|
+
"passed": true,
|
|
67
|
+
"evidence": "Observed in outputs/result.md"
|
|
68
|
+
}
|
|
69
|
+
],
|
|
70
|
+
"execution_metrics": {
|
|
71
|
+
"total_tool_calls": 4,
|
|
72
|
+
"errors_encountered": 0,
|
|
73
|
+
"output_chars": 5120
|
|
74
|
+
},
|
|
75
|
+
"user_notes_summary": {
|
|
76
|
+
"uncertainties": [],
|
|
77
|
+
"needs_review": [],
|
|
78
|
+
"workarounds": []
|
|
79
|
+
},
|
|
80
|
+
"overall_summary": "Mostly correct, but edge cases need review.",
|
|
81
|
+
"timing": {
|
|
82
|
+
"total_tokens": 84852,
|
|
83
|
+
"total_duration_seconds": 23.3
|
|
84
|
+
},
|
|
85
|
+
"meta": {
|
|
86
|
+
"generated_at": "2026-03-17T12:00:00.000Z",
|
|
87
|
+
"eval_id": 1,
|
|
88
|
+
"eval_name": "handles-empty-input",
|
|
89
|
+
"config": "with_skill",
|
|
90
|
+
"run_id": "run-1"
|
|
91
|
+
}
|
|
92
|
+
}
|
|
93
|
+
```
|
|
94
|
+
|
|
95
|
+
要求:
|
|
96
|
+
- `expectations` 里的字段名固定为 `text`、`passed`、`evidence`。
|
|
97
|
+
- `pass_rate` 建议是 `0..1` 之间的小数。
|
|
98
|
+
|
|
99
|
+
## `timing.json`
|
|
100
|
+
|
|
101
|
+
用于保存一次运行的耗时与 token 信息。
|
|
102
|
+
|
|
103
|
+
```json
|
|
104
|
+
{
|
|
105
|
+
"total_tokens": 84852,
|
|
106
|
+
"duration_ms": 23332,
|
|
107
|
+
"total_duration_seconds": 23.3
|
|
108
|
+
}
|
|
109
|
+
```
|
|
110
|
+
|
|
111
|
+
## `benchmark.json`
|
|
112
|
+
|
|
113
|
+
由 `aggregate-benchmark.js` 生成,用于总览不同配置的表现。
|
|
114
|
+
|
|
115
|
+
```json
|
|
116
|
+
{
|
|
117
|
+
"skill_name": "example-skill",
|
|
118
|
+
"generated_at": "2026-03-17T12:00:00.000Z",
|
|
119
|
+
"workspace": "/abs/path/to/iteration-1",
|
|
120
|
+
"configs": {
|
|
121
|
+
"with_skill": {
|
|
122
|
+
"pass_rate": { "mean": 0.9, "stddev": 0.1, "min": 0.8, "max": 1.0 },
|
|
123
|
+
"time_seconds": { "mean": 12.4, "stddev": 1.1, "min": 11.2, "max": 13.5 },
|
|
124
|
+
"tokens": { "mean": 4200, "stddev": 380, "min": 3900, "max": 4700 }
|
|
125
|
+
},
|
|
126
|
+
"without_skill": {
|
|
127
|
+
"pass_rate": { "mean": 0.6, "stddev": 0.2, "min": 0.4, "max": 0.8 },
|
|
128
|
+
"time_seconds": { "mean": 9.5, "stddev": 0.7, "min": 8.9, "max": 10.2 },
|
|
129
|
+
"tokens": { "mean": 3100, "stddev": 240, "min": 2900, "max": 3400 }
|
|
130
|
+
}
|
|
131
|
+
},
|
|
132
|
+
"delta": {
|
|
133
|
+
"pass_rate": "+0.3000",
|
|
134
|
+
"time_seconds": "+2.9000",
|
|
135
|
+
"tokens": "+1100.0000"
|
|
136
|
+
},
|
|
137
|
+
"runs": {
|
|
138
|
+
"with_skill": [],
|
|
139
|
+
"without_skill": []
|
|
140
|
+
}
|
|
141
|
+
}
|
|
142
|
+
```
|
|
143
|
+
|
|
144
|
+
## `analysis.json`
|
|
145
|
+
|
|
146
|
+
由 `analyze-benchmark.js` 生成,用于总结 benchmark 的稳定收益、方差热点和下一步建议。
|
|
147
|
+
|
|
148
|
+
```json
|
|
149
|
+
{
|
|
150
|
+
"skill_name": "example-skill",
|
|
151
|
+
"generated_at": "2026-03-17T12:15:00.000Z",
|
|
152
|
+
"workspace": "/abs/path/to/iteration-1",
|
|
153
|
+
"verdict": "improves",
|
|
154
|
+
"release_readiness": "needs_iteration",
|
|
155
|
+
"recommendation": "Keep the skill, but reduce variance before release.",
|
|
156
|
+
"key_findings": [
|
|
157
|
+
"with_skill materially improves pass rate"
|
|
158
|
+
],
|
|
159
|
+
"variance_hotspots": [
|
|
160
|
+
"baseline repeatedly misses billing details"
|
|
161
|
+
],
|
|
162
|
+
"suggested_actions": [
|
|
163
|
+
"tighten assertions around billing coverage"
|
|
164
|
+
],
|
|
165
|
+
"watchouts": [
|
|
166
|
+
"token cost increased"
|
|
167
|
+
],
|
|
168
|
+
"supporting_metrics": {
|
|
169
|
+
"pass_rate_delta": "+0.3000",
|
|
170
|
+
"time_seconds_delta": "+2.9000",
|
|
171
|
+
"tokens_delta": "+1100.0000"
|
|
172
|
+
},
|
|
173
|
+
"failure_clusters": {}
|
|
174
|
+
}
|
|
175
|
+
```
|
|
176
|
+
|
|
177
|
+
## `comparison.json`
|
|
178
|
+
|
|
179
|
+
由 `compare-runs.js` 生成,用于 blind comparison 两个 config 的输出质量。
|
|
180
|
+
|
|
181
|
+
```json
|
|
182
|
+
{
|
|
183
|
+
"workspace": "/abs/path/to/iteration-1",
|
|
184
|
+
"generated_at": "2026-03-17T12:20:00.000Z",
|
|
185
|
+
"config_a": "with_skill",
|
|
186
|
+
"config_b": "without_skill",
|
|
187
|
+
"summary": {
|
|
188
|
+
"total_pairs": 3,
|
|
189
|
+
"config_a_wins": 2,
|
|
190
|
+
"config_b_wins": 0,
|
|
191
|
+
"ties": 1,
|
|
192
|
+
"inconclusive": 0
|
|
193
|
+
},
|
|
194
|
+
"comparisons": [
|
|
195
|
+
{
|
|
196
|
+
"eval_id": 1,
|
|
197
|
+
"winner_label": "A",
|
|
198
|
+
"winner_config": "with_skill",
|
|
199
|
+
"confidence": 0.9,
|
|
200
|
+
"rationale": "Candidate A is more complete and specific."
|
|
201
|
+
}
|
|
202
|
+
]
|
|
203
|
+
}
|
|
204
|
+
```
|
|
205
|
+
|
|
206
|
+
## 推荐目录结构
|
|
207
|
+
|
|
208
|
+
```text
|
|
209
|
+
my-skill-workspace/
|
|
210
|
+
└── iteration-1/
|
|
211
|
+
├── eval-0/
|
|
212
|
+
│ ├── eval_metadata.json
|
|
213
|
+
│ ├── with_skill/
|
|
214
|
+
│ │ ├── outputs/
|
|
215
|
+
│ │ ├── grading.json
|
|
216
|
+
│ │ └── timing.json
|
|
217
|
+
│ └── without_skill/
|
|
218
|
+
│ ├── outputs/
|
|
219
|
+
│ ├── grading.json
|
|
220
|
+
│ └── timing.json
|
|
221
|
+
├── benchmark.json
|
|
222
|
+
├── benchmark.md
|
|
223
|
+
├── analysis.json
|
|
224
|
+
├── analysis.md
|
|
225
|
+
├── comparison.json
|
|
226
|
+
└── comparison.md
|
|
227
|
+
```
|
|
@@ -1,46 +1,46 @@
|
|
|
1
|
-
export interface BenchmarkRun {
|
|
2
|
-
eval_id: string | number;
|
|
3
|
-
run_id: string;
|
|
4
|
-
pass_rate: number;
|
|
5
|
-
passed: number;
|
|
6
|
-
failed: number;
|
|
7
|
-
total: number;
|
|
8
|
-
time_seconds: number;
|
|
9
|
-
tokens: number;
|
|
10
|
-
tool_calls: number;
|
|
11
|
-
errors: number;
|
|
12
|
-
expectations: Array<Record<string, unknown>>;
|
|
13
|
-
notes: string[];
|
|
14
|
-
}
|
|
15
|
-
|
|
16
|
-
export interface StatsSummary {
|
|
17
|
-
mean: number;
|
|
18
|
-
stddev: number;
|
|
19
|
-
min: number;
|
|
20
|
-
max: number;
|
|
21
|
-
}
|
|
22
|
-
|
|
23
|
-
export interface BenchmarkDocument {
|
|
24
|
-
skill_name: string;
|
|
25
|
-
generated_at: string;
|
|
26
|
-
workspace: string;
|
|
27
|
-
configs: Record<string, {
|
|
28
|
-
pass_rate: StatsSummary;
|
|
29
|
-
time_seconds: StatsSummary;
|
|
30
|
-
tokens: StatsSummary;
|
|
31
|
-
}>;
|
|
32
|
-
delta: {
|
|
33
|
-
pass_rate: string;
|
|
34
|
-
time_seconds: string;
|
|
35
|
-
tokens: string;
|
|
36
|
-
};
|
|
37
|
-
runs: Record<string, BenchmarkRun[]>;
|
|
38
|
-
}
|
|
39
|
-
|
|
40
|
-
export function loadRunResults(iterationDir: string): Promise<Record<string, BenchmarkRun[]>>;
|
|
41
|
-
export function buildBenchmarkDocument(
|
|
42
|
-
iterationDir: string,
|
|
43
|
-
skillName: string,
|
|
44
|
-
configRuns: Record<string, BenchmarkRun[]>
|
|
45
|
-
): BenchmarkDocument;
|
|
46
|
-
export function renderBenchmarkMarkdown(benchmark: BenchmarkDocument): string;
|
|
1
|
+
export interface BenchmarkRun {
|
|
2
|
+
eval_id: string | number;
|
|
3
|
+
run_id: string;
|
|
4
|
+
pass_rate: number;
|
|
5
|
+
passed: number;
|
|
6
|
+
failed: number;
|
|
7
|
+
total: number;
|
|
8
|
+
time_seconds: number;
|
|
9
|
+
tokens: number;
|
|
10
|
+
tool_calls: number;
|
|
11
|
+
errors: number;
|
|
12
|
+
expectations: Array<Record<string, unknown>>;
|
|
13
|
+
notes: string[];
|
|
14
|
+
}
|
|
15
|
+
|
|
16
|
+
export interface StatsSummary {
|
|
17
|
+
mean: number;
|
|
18
|
+
stddev: number;
|
|
19
|
+
min: number;
|
|
20
|
+
max: number;
|
|
21
|
+
}
|
|
22
|
+
|
|
23
|
+
export interface BenchmarkDocument {
|
|
24
|
+
skill_name: string;
|
|
25
|
+
generated_at: string;
|
|
26
|
+
workspace: string;
|
|
27
|
+
configs: Record<string, {
|
|
28
|
+
pass_rate: StatsSummary;
|
|
29
|
+
time_seconds: StatsSummary;
|
|
30
|
+
tokens: StatsSummary;
|
|
31
|
+
}>;
|
|
32
|
+
delta: {
|
|
33
|
+
pass_rate: string;
|
|
34
|
+
time_seconds: string;
|
|
35
|
+
tokens: string;
|
|
36
|
+
};
|
|
37
|
+
runs: Record<string, BenchmarkRun[]>;
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
export function loadRunResults(iterationDir: string): Promise<Record<string, BenchmarkRun[]>>;
|
|
41
|
+
export function buildBenchmarkDocument(
|
|
42
|
+
iterationDir: string,
|
|
43
|
+
skillName: string,
|
|
44
|
+
configRuns: Record<string, BenchmarkRun[]>
|
|
45
|
+
): BenchmarkDocument;
|
|
46
|
+
export function renderBenchmarkMarkdown(benchmark: BenchmarkDocument): string;
|