oh-my-knowledge 0.26.0 → 0.27.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +109 -347
- package/README.zh.md +146 -376
- package/dist/src/analysis/report-diagnostics.d.ts +1 -1
- package/dist/src/analysis/report-diagnostics.js +1 -1
- package/dist/src/analysis/sample-diagnostics.d.ts +2 -2
- package/dist/src/analysis/sample-diagnostics.js +2 -2
- package/dist/src/authoring/generator.d.ts.map +1 -1
- package/dist/src/authoring/generator.js +9 -8
- package/dist/src/authoring/generator.js.map +1 -1
- package/dist/src/cli/cli-exit.d.ts +15 -0
- package/dist/src/cli/cli-exit.d.ts.map +1 -0
- package/dist/src/cli/cli-exit.js +19 -0
- package/dist/src/cli/cli-exit.js.map +1 -0
- package/dist/src/cli/commands/_shared.d.ts +12 -0
- package/dist/src/cli/commands/_shared.d.ts.map +1 -0
- package/dist/src/cli/commands/_shared.js +26 -0
- package/dist/src/cli/commands/_shared.js.map +1 -0
- package/dist/src/cli/commands/doctor.d.ts +2 -0
- package/dist/src/cli/commands/doctor.d.ts.map +1 -0
- package/dist/src/cli/commands/doctor.js +137 -0
- package/dist/src/cli/commands/doctor.js.map +1 -0
- package/dist/src/cli/commands/eval-debias.d.ts +2 -0
- package/dist/src/cli/commands/eval-debias.d.ts.map +1 -0
- package/dist/src/cli/commands/eval-debias.js +88 -0
- package/dist/src/cli/commands/eval-debias.js.map +1 -0
- package/dist/src/cli/commands/eval-gold.d.ts +2 -0
- package/dist/src/cli/commands/eval-gold.d.ts.map +1 -0
- package/dist/src/cli/commands/eval-gold.js +137 -0
- package/dist/src/cli/commands/eval-gold.js.map +1 -0
- package/dist/src/cli/commands/eval-runner.d.ts +2 -0
- package/dist/src/cli/commands/eval-runner.d.ts.map +1 -0
- package/dist/src/cli/commands/eval-runner.js +299 -0
- package/dist/src/cli/commands/eval-runner.js.map +1 -0
- package/dist/src/cli/commands/eval.d.ts +2 -0
- package/dist/src/cli/commands/eval.d.ts.map +1 -0
- package/dist/src/cli/commands/eval.js +16 -0
- package/dist/src/cli/commands/eval.js.map +1 -0
- package/dist/src/cli/commands/export-diff.d.ts +2 -0
- package/dist/src/cli/commands/export-diff.d.ts.map +1 -0
- package/dist/src/cli/commands/export-diff.js +177 -0
- package/dist/src/cli/commands/export-diff.js.map +1 -0
- package/dist/src/cli/commands/export-saturation.d.ts +2 -0
- package/dist/src/cli/commands/export-saturation.d.ts.map +1 -0
- package/dist/src/cli/commands/export-saturation.js +59 -0
- package/dist/src/cli/commands/export-saturation.js.map +1 -0
- package/dist/src/cli/commands/export-verdict.d.ts +2 -0
- package/dist/src/cli/commands/export-verdict.d.ts.map +1 -0
- package/dist/src/cli/commands/export-verdict.js +45 -0
- package/dist/src/cli/commands/export-verdict.js.map +1 -0
- package/dist/src/cli/commands/export.d.ts +2 -0
- package/dist/src/cli/commands/export.d.ts.map +1 -0
- package/dist/src/cli/commands/export.js +150 -0
- package/dist/src/cli/commands/export.js.map +1 -0
- package/dist/src/cli/commands/improve-failures.d.ts +2 -0
- package/dist/src/cli/commands/improve-failures.d.ts.map +1 -0
- package/dist/src/cli/commands/improve-failures.js +59 -0
- package/dist/src/cli/commands/improve-failures.js.map +1 -0
- package/dist/src/cli/commands/improve-plan.d.ts +2 -0
- package/dist/src/cli/commands/improve-plan.d.ts.map +1 -0
- package/dist/src/cli/commands/improve-plan.js +75 -0
- package/dist/src/cli/commands/improve-plan.js.map +1 -0
- package/dist/src/cli/commands/improve-samples.d.ts +2 -0
- package/dist/src/cli/commands/improve-samples.d.ts.map +1 -0
- package/dist/src/cli/commands/improve-samples.js +120 -0
- package/dist/src/cli/commands/improve-samples.js.map +1 -0
- package/dist/src/cli/commands/improve-skill.d.ts +2 -0
- package/dist/src/cli/commands/improve-skill.d.ts.map +1 -0
- package/dist/src/cli/commands/improve-skill.js +115 -0
- package/dist/src/cli/commands/improve-skill.js.map +1 -0
- package/dist/src/cli/commands/improve.d.ts +2 -0
- package/dist/src/cli/commands/improve.d.ts.map +1 -0
- package/dist/src/cli/commands/improve.js +34 -0
- package/dist/src/cli/commands/improve.js.map +1 -0
- package/dist/src/cli/commands/init.d.ts +2 -0
- package/dist/src/cli/commands/init.d.ts.map +1 -0
- package/dist/src/cli/commands/init.js +110 -0
- package/dist/src/cli/commands/init.js.map +1 -0
- package/dist/src/cli/commands/observe.d.ts +2 -0
- package/dist/src/cli/commands/observe.d.ts.map +1 -0
- package/dist/src/cli/commands/observe.js +77 -0
- package/dist/src/cli/commands/observe.js.map +1 -0
- package/dist/src/cli/commands/registry.d.ts +11 -0
- package/dist/src/cli/commands/registry.d.ts.map +1 -0
- package/dist/src/cli/commands/registry.js +42 -0
- package/dist/src/cli/commands/registry.js.map +1 -0
- package/dist/src/cli/commands/studio.d.ts +2 -0
- package/dist/src/cli/commands/studio.d.ts.map +1 -0
- package/dist/src/cli/commands/studio.js +76 -0
- package/dist/src/cli/commands/studio.js.map +1 -0
- package/dist/src/cli/coverage-renderer.d.ts +1 -1
- package/dist/src/cli/coverage-renderer.js +1 -1
- package/dist/src/cli/i18n-dict.d.ts +4 -4
- package/dist/src/cli/i18n-dict.d.ts.map +1 -1
- package/dist/src/cli/i18n-dict.js +558 -564
- package/dist/src/cli/i18n-dict.js.map +1 -1
- package/dist/src/cli/index.js +43 -1516
- package/dist/src/cli/index.js.map +1 -1
- package/dist/src/cli/parse-run-config.d.ts +3 -4
- package/dist/src/cli/parse-run-config.d.ts.map +1 -1
- package/dist/src/cli/parse-run-config.js +5 -5
- package/dist/src/cli/parse-run-config.js.map +1 -1
- package/dist/src/cli/parse-strict.d.ts +0 -13
- package/dist/src/cli/parse-strict.d.ts.map +1 -1
- package/dist/src/cli/parse-strict.js +2 -1
- package/dist/src/cli/parse-strict.js.map +1 -1
- package/dist/src/doctor/index.d.ts +1 -1
- package/dist/src/doctor/index.js +2 -2
- package/dist/src/doctor/index.js.map +1 -1
- package/dist/src/doctor/preflight.d.ts +2 -2
- package/dist/src/doctor/preflight.js +2 -2
- package/dist/src/eval-core/fact-checker.js +1 -1
- package/dist/src/eval-core/fact-checker.js.map +1 -1
- package/dist/src/eval-core/layer-gates.d.ts +1 -1
- package/dist/src/eval-core/layer-gates.js +1 -1
- package/dist/src/eval-core/verdict.d.ts +4 -4
- package/dist/src/eval-core/verdict.d.ts.map +1 -1
- package/dist/src/eval-core/verdict.js +2 -2
- package/dist/src/eval-workflows/batch-evaluation-workflow.d.ts +15 -2
- package/dist/src/eval-workflows/batch-evaluation-workflow.d.ts.map +1 -1
- package/dist/src/eval-workflows/batch-evaluation-workflow.js +13 -2
- package/dist/src/eval-workflows/batch-evaluation-workflow.js.map +1 -1
- package/dist/src/eval-workflows/evaluation-pipeline.d.ts +6 -3
- package/dist/src/eval-workflows/evaluation-pipeline.d.ts.map +1 -1
- package/dist/src/eval-workflows/evaluation-pipeline.js +10 -9
- package/dist/src/eval-workflows/evaluation-pipeline.js.map +1 -1
- package/dist/src/eval-workflows/run-evaluation.d.ts +1 -1
- package/dist/src/eval-workflows/run-evaluation.d.ts.map +1 -1
- package/dist/src/eval-workflows/run-evaluation.js +13 -6
- package/dist/src/eval-workflows/run-evaluation.js.map +1 -1
- package/dist/src/executors/script.d.ts.map +1 -1
- package/dist/src/executors/script.js +16 -3
- package/dist/src/executors/script.js.map +1 -1
- package/dist/src/grading/debias-validate.d.ts +2 -2
- package/dist/src/grading/debias-validate.js +2 -2
- package/dist/src/grading/gold-cli.d.ts +2 -5
- package/dist/src/grading/gold-cli.d.ts.map +1 -1
- package/dist/src/grading/gold-cli.js +4 -8
- package/dist/src/grading/gold-cli.js.map +1 -1
- package/dist/src/grading/judge.d.ts +1 -1
- package/dist/src/renderer/html-renderer.js +1 -1
- package/dist/src/renderer/layout.js +5 -5
- package/dist/src/renderer/layout.js.map +1 -1
- package/dist/src/renderer/skill-health-renderer.d.ts +1 -1
- package/dist/src/renderer/skill-health-renderer.js +1 -1
- package/dist/src/renderer/summary.js +8 -8
- package/dist/src/server/report-server.d.ts.map +1 -1
- package/dist/src/server/report-server.js +8 -7
- package/dist/src/server/report-server.js.map +1 -1
- package/dist/src/types/doctor.d.ts +3 -3
- package/dist/src/types/eval.d.ts +1 -1
- package/dist/src/types/report.d.ts +1 -1
- package/package.json +1 -1
|
@@ -16,9 +16,9 @@
|
|
|
16
16
|
* 2. **保留原文的白名单 (产品术语 / 命令 / 文件名)**
|
|
17
17
|
* 以下 token 在两种语言里都保留原文, 不翻译:
|
|
18
18
|
* - 产品名: omk, oh-my-knowledge, Claude, npm
|
|
19
|
-
* - 子命令空间和命令名:
|
|
20
|
-
*
|
|
21
|
-
*
|
|
19
|
+
* - 子命令空间和命令名: init, doctor, eval, observe, improve, export,
|
|
20
|
+
* studio, samples, skill, plan, failures, gold, debias, diff, verdict,
|
|
21
|
+
* saturation
|
|
22
22
|
* - omk 核心业务术语: skill, variant, sample, judge, executor (出现在产品
|
|
23
23
|
* UI 里时首字母可大写如 "Skill 评测", 描述句中保持小写)
|
|
24
24
|
* - 技术参数: --lang, --control, --treatment, --bootstrap, --judge-repeat,
|
|
@@ -48,21 +48,9 @@
|
|
|
48
48
|
* Record 类型自动强制每 key 加新语言版本。
|
|
49
49
|
*/
|
|
50
50
|
export const CLI_DICT = {
|
|
51
|
-
'cli.common.lang_invalid_silent': {
|
|
52
|
-
zh: '无效的语言代码: {value} (仅支持 zh / en, 已使用默认 zh)',
|
|
53
|
-
en: 'Invalid language code: {value} (supported: zh / en, using default zh)',
|
|
54
|
-
},
|
|
55
|
-
'cli.common.help_hint': {
|
|
56
|
-
zh: "运行 'omk --help' 查看用法",
|
|
57
|
-
en: "Run 'omk --help' to see usage",
|
|
58
|
-
},
|
|
59
51
|
'cli.common.unknown_domain': {
|
|
60
|
-
zh: "
|
|
61
|
-
en: "Unknown
|
|
62
|
-
},
|
|
63
|
-
'cli.common.unknown_bench_command': {
|
|
64
|
-
zh: "未知子命令: bench {command} (运行 'omk --help' 查看可用列表)",
|
|
65
|
-
en: "Unknown bench command: {command} (run 'omk --help' to see all commands)",
|
|
52
|
+
zh: "未知命令:{domain}。运行 'omk --help' 查看可用命令。",
|
|
53
|
+
en: "Unknown command: {domain}. Run 'omk --help' to see available commands.",
|
|
66
54
|
},
|
|
67
55
|
'cli.init.scaffolded': {
|
|
68
56
|
zh: '已初始化测评项目: {dir}',
|
|
@@ -73,7 +61,7 @@ export const CLI_DICT = {
|
|
|
73
61
|
en: 'Next steps:',
|
|
74
62
|
},
|
|
75
63
|
'cli.init.next_step_edit_samples': {
|
|
76
|
-
zh: ' 1. 编辑 eval-samples.json
|
|
64
|
+
zh: ' 1. 编辑 eval-samples.json,加入你要测的评测用例',
|
|
77
65
|
en: ' 1. Edit eval-samples.json to add your test cases',
|
|
78
66
|
},
|
|
79
67
|
'cli.init.next_step_edit_skills': {
|
|
@@ -81,8 +69,8 @@ export const CLI_DICT = {
|
|
|
81
69
|
en: ' 2. Edit skills/code-review-v1/SKILL.md and skills/code-review-v2/SKILL.md with your skill versions',
|
|
82
70
|
},
|
|
83
71
|
'cli.init.next_step_run': {
|
|
84
|
-
zh: ' 3. 运行: omk
|
|
85
|
-
en: ' 3. Run: omk
|
|
72
|
+
zh: ' 3. 运行: omk eval --control code-review-v1 --treatment code-review-v2',
|
|
73
|
+
en: ' 3. Run: omk eval --control code-review-v1 --treatment code-review-v2',
|
|
86
74
|
},
|
|
87
75
|
'cli.init.note_codex_executor': {
|
|
88
76
|
zh: '\n注: omk 评测时把 SKILL.md 整文(含 frontmatter)作为 system prompt 注入 — 跨 executor 一致(claude / codex / openai-api / gemini 都走同一条路径,不依赖任何 executor 的 native skill auto-discovery 或 Skill 工具机制)。frontmatter 在 prompt 头部对 model 行为无显著影响。\n模板带 Claude Code 兼容的 frontmatter(name + description)是为了让同一份 directory-skill 也能 deploy 到 Claude Code:把整个目录复制到 ~/.claude/skills/code-review-v1/(整目录,不是单个 SKILL.md),Claude SDK 才能识别。这是 omk 评测之外的 bonus,一份文件双向 dogfood。',
|
|
@@ -92,6 +80,22 @@ export const CLI_DICT = {
|
|
|
92
80
|
zh: '\n💡 新版本可用: {old} → {new}, 运行 npm update {pkg} -g 升级\n\n',
|
|
93
81
|
en: '\n💡 New version available: {old} → {new}, run npm update {pkg} -g to upgrade\n\n',
|
|
94
82
|
},
|
|
83
|
+
'cli.run.power_warning_tiny_n': {
|
|
84
|
+
zh: '⚠ N={n} < 5:仅适合探索,任何结论都不可靠,CI 会很宽。需要决策时建议 ≥20 条评测用例。',
|
|
85
|
+
en: '⚠ N={n} < 5 (exploration-only): any conclusion is unreliable, CI will be uselessly wide. Decisions need ≥20 cases.',
|
|
86
|
+
},
|
|
87
|
+
'cli.run.power_warning_small_n': {
|
|
88
|
+
zh: '⚠ N={n} < 20:只能识别很大的效果(Cohen\'s d > 0.8),中等效果(d ≈ 0.5)很难检出。要做可靠决策建议 ≥20 条评测用例。',
|
|
89
|
+
en: '⚠ N={n} < 20 (large-effect-only, Cohen\'s d > 0.8): medium effects (d ≈ 0.5) hard to detect. For confident decisions consider ≥20 cases.',
|
|
90
|
+
},
|
|
91
|
+
'cli.run.power_warning_repeat_one': {
|
|
92
|
+
zh: '⚠ --repeat=1:单轮评测无法测稳定性(CV 会标记为未测量)。用 --repeat 3+ 检测同一 variant 内部方差。',
|
|
93
|
+
en: '⚠ --repeat=1: single-run cannot measure stability (CV will be marked "not measured"). Use --repeat 3+ to detect within-variant variance.',
|
|
94
|
+
},
|
|
95
|
+
'cli.run.dry_run_no_scores': {
|
|
96
|
+
zh: 'eval dry-run:仅预览任务,不检查分数',
|
|
97
|
+
en: 'Eval dry-run: no scores to check',
|
|
98
|
+
},
|
|
95
99
|
'cli.progress.preflight_starting': {
|
|
96
100
|
zh: '⏳ 正在预检模型连通性...\n',
|
|
97
101
|
en: '⏳ Preflight: checking model connectivity...\n',
|
|
@@ -168,6 +172,14 @@ export const CLI_DICT = {
|
|
|
168
172
|
zh: '\n✅ 批量评测完成\n',
|
|
169
173
|
en: '\n✅ Batch evaluation done\n',
|
|
170
174
|
},
|
|
175
|
+
'cli.run.batch_verdict_header': {
|
|
176
|
+
zh: '批量评测结论:{status}({passed}/{total} 通过)',
|
|
177
|
+
en: 'Batch verdict: {status} ({passed}/{total} passed)',
|
|
178
|
+
},
|
|
179
|
+
'cli.run.batch_child_report_missing': {
|
|
180
|
+
zh: '⚠ 子报告缺失:{id},将按不可 ship 处理。\n',
|
|
181
|
+
en: '⚠ Child report missing: {id}; treating it as not shippable.\n',
|
|
182
|
+
},
|
|
171
183
|
'cli.run.eval_complete': {
|
|
172
184
|
zh: '\n✅ 评测完成\n',
|
|
173
185
|
en: '\n✅ Evaluation done\n',
|
|
@@ -180,6 +192,10 @@ export const CLI_DICT = {
|
|
|
180
192
|
zh: '📄 报告已保存到: {path}\n',
|
|
181
193
|
en: '📄 Report saved to: {path}\n',
|
|
182
194
|
},
|
|
195
|
+
'cli.run.report_only_gate_skipped': {
|
|
196
|
+
zh: 'ℹ 已启用 report-only 模式:保留 verdict 输出,但本次不使用 verdict 改写 exit code。\n',
|
|
197
|
+
en: 'ℹ Report-only mode enabled: verdict is still printed, but it will not affect the exit code.\n',
|
|
198
|
+
},
|
|
183
199
|
'cli.run.report_server_running': {
|
|
184
200
|
zh: '\n📊 报告服务已启动: {url}\n',
|
|
185
201
|
en: '\n📊 Report server running at {url}\n',
|
|
@@ -197,8 +213,8 @@ export const CLI_DICT = {
|
|
|
197
213
|
en: '\n💡 Non-interactive environment, skipping report server\n',
|
|
198
214
|
},
|
|
199
215
|
'cli.run.no_serve_view_hint': {
|
|
200
|
-
zh: '
|
|
201
|
-
en: '
|
|
216
|
+
zh: ' 导出报告:omk export {id} --reports-dir {dir}\n',
|
|
217
|
+
en: ' Export report: omk export {id} --reports-dir {dir}\n',
|
|
202
218
|
},
|
|
203
219
|
'cli.run.gold_load_failed': {
|
|
204
220
|
zh: '\n⚠ gold dataset 加载失败 ({dir}):\n',
|
|
@@ -216,9 +232,9 @@ export const CLI_DICT = {
|
|
|
216
232
|
zh: '❌ 错误: {message}',
|
|
217
233
|
en: '❌ Error: {message}',
|
|
218
234
|
},
|
|
219
|
-
'cli.
|
|
220
|
-
zh:
|
|
221
|
-
en:
|
|
235
|
+
'cli.observe.view_hint': {
|
|
236
|
+
zh: '分析 JSON 已写入 output-dir;后续可用 omk observe 持续生成日报。',
|
|
237
|
+
en: 'Analysis JSON written to output-dir; use omk observe to keep producing health reports.',
|
|
222
238
|
},
|
|
223
239
|
'cli.common.skill_dir_not_found': {
|
|
224
240
|
zh: '未找到 skill 目录: {path}',
|
|
@@ -237,23 +253,43 @@ export const CLI_DICT = {
|
|
|
237
253
|
en: 'No judge configured. Pass --judge-models <executor:model> or ensure the report has meta.judgeModels.',
|
|
238
254
|
},
|
|
239
255
|
'cli.common.judge_models_single_only': {
|
|
240
|
-
zh: '
|
|
241
|
-
en: '
|
|
242
|
-
},
|
|
243
|
-
'cli.common.usage_gold_validate': {
|
|
244
|
-
zh: '用法: omk bench gold validate <dir>',
|
|
245
|
-
en: 'Usage: omk bench gold validate <dir>',
|
|
256
|
+
zh: '{cmd} 仅支持单评委。--judge-models 只能传一个 executor:model entry。',
|
|
257
|
+
en: '{cmd} only supports a single judge. --judge-models accepts exactly one executor:model entry.',
|
|
246
258
|
},
|
|
247
259
|
'cli.common.warn_load_samples_failed': {
|
|
248
260
|
zh: '⚠ 加载 samples 文件失败 ({path}): {message}\n',
|
|
249
261
|
en: '⚠ Failed to load samples file ({path}): {message}\n',
|
|
250
262
|
},
|
|
263
|
+
'cli.export.unsupported_format': {
|
|
264
|
+
zh: '不支持的导出格式:{format}。可用格式:html / markdown / github-summary。',
|
|
265
|
+
en: 'Unsupported export format: {format}. Available: html / markdown / github-summary.',
|
|
266
|
+
},
|
|
267
|
+
'cli.export.html_done': {
|
|
268
|
+
zh: '已导出 HTML:{path}',
|
|
269
|
+
en: 'HTML exported to: {path}',
|
|
270
|
+
},
|
|
271
|
+
'cli.export.done': {
|
|
272
|
+
zh: '已导出:{path}',
|
|
273
|
+
en: 'Exported to: {path}',
|
|
274
|
+
},
|
|
275
|
+
'cli.studio.started': {
|
|
276
|
+
zh: 'studio 已启动:{url}',
|
|
277
|
+
en: 'Studio running at {url}',
|
|
278
|
+
},
|
|
279
|
+
'cli.studio.stop_hint': {
|
|
280
|
+
zh: '按 Ctrl+C 停止服务',
|
|
281
|
+
en: 'Press Ctrl+C to stop',
|
|
282
|
+
},
|
|
283
|
+
'cli.studio.open_failed': {
|
|
284
|
+
zh: '⚠ 无法自动打开浏览器({command}):{message}\n',
|
|
285
|
+
en: '⚠ Failed to open browser automatically ({command}): {message}\n',
|
|
286
|
+
},
|
|
251
287
|
'cli.gen.skill_skipped_existing': {
|
|
252
288
|
zh: '⏭️ {name}: eval-samples 已存在, 跳过\n',
|
|
253
289
|
en: '⏭️ {name}: eval-samples already exists, skipping\n',
|
|
254
290
|
},
|
|
255
291
|
'cli.gen.skill_generating': {
|
|
256
|
-
zh: '🔄 {name}: 正在生成 {count}
|
|
292
|
+
zh: '🔄 {name}: 正在生成 {count} 条评测用例...\n',
|
|
257
293
|
en: '🔄 {name}: generating {count} test cases...\n',
|
|
258
294
|
},
|
|
259
295
|
'cli.gen.skill_done': {
|
|
@@ -269,19 +305,19 @@ export const CLI_DICT = {
|
|
|
269
305
|
en: 'No eval-samples need generating (all skills already have paired files)',
|
|
270
306
|
},
|
|
271
307
|
'cli.gen.batch_summary': {
|
|
272
|
-
zh: '\n共生成 {n} 份 eval-samples, 请审查后运行: omk
|
|
273
|
-
en: '\nGenerated {n} eval-samples files. Review them, then run: omk
|
|
308
|
+
zh: '\n共生成 {n} 份 eval-samples, 请审查后运行: omk eval --batch',
|
|
309
|
+
en: '\nGenerated {n} eval-samples files. Review them, then run: omk eval --batch',
|
|
274
310
|
},
|
|
275
311
|
'cli.gen.specify_skill_path': {
|
|
276
|
-
zh: '请指定 skill 文件路径, 例如: omk
|
|
277
|
-
en: 'Please specify a skill file path, e.g.: omk
|
|
312
|
+
zh: '请指定 skill 文件路径, 例如: omk improve samples skills/my-skill.md',
|
|
313
|
+
en: 'Please specify a skill file path, e.g.: omk improve samples skills/my-skill.md',
|
|
278
314
|
},
|
|
279
315
|
'cli.gen.samples_already_exists': {
|
|
280
316
|
zh: 'eval-samples.json 已存在。如需覆盖请先删除该文件。',
|
|
281
317
|
en: 'eval-samples.json already exists. Delete it first if you want to overwrite.',
|
|
282
318
|
},
|
|
283
319
|
'cli.gen.single_generating': {
|
|
284
|
-
zh: '🔄 正在生成 {count}
|
|
320
|
+
zh: '🔄 正在生成 {count} 条评测用例...\n',
|
|
285
321
|
en: '🔄 Generating {count} test cases...\n',
|
|
286
322
|
},
|
|
287
323
|
'cli.gen.single_done': {
|
|
@@ -289,20 +325,20 @@ export const CLI_DICT = {
|
|
|
289
325
|
en: '✅ Generated {n} samples → {path}{cost}\n',
|
|
290
326
|
},
|
|
291
327
|
'cli.gen.review_hint': {
|
|
292
|
-
zh: '\n
|
|
293
|
-
en: '\nReview the generated test cases, then run: omk
|
|
328
|
+
zh: '\n请审查生成的评测用例后运行: omk eval',
|
|
329
|
+
en: '\nReview the generated test cases, then run: omk eval',
|
|
294
330
|
},
|
|
295
331
|
'cli.gen.failed': {
|
|
296
332
|
zh: '生成失败: {message}',
|
|
297
333
|
en: 'Generation failed: {message}',
|
|
298
334
|
},
|
|
299
335
|
'cli.evolve.specify_skill_path': {
|
|
300
|
-
zh: '请指定 skill 文件路径, 例如: omk
|
|
301
|
-
en: 'Please specify a skill file path, e.g.: omk
|
|
336
|
+
zh: '请指定 skill 文件路径, 例如: omk improve skill skills/my-skill.md',
|
|
337
|
+
en: 'Please specify a skill file path, e.g.: omk improve skill skills/my-skill.md',
|
|
302
338
|
},
|
|
303
339
|
'cli.evolve.section_header': {
|
|
304
|
-
zh: '\n===
|
|
305
|
-
en: '\n===
|
|
340
|
+
zh: '\n=== Improve skill: {path} ===\n',
|
|
341
|
+
en: '\n=== Improve skill: {path} ===\n',
|
|
306
342
|
},
|
|
307
343
|
'cli.evolve.round_baseline': {
|
|
308
344
|
zh: '第 0 轮 (基线): score={score} ({cost})\n',
|
|
@@ -329,544 +365,285 @@ export const CLI_DICT = {
|
|
|
329
365
|
en: 'All versions saved at: {dir}/\n',
|
|
330
366
|
},
|
|
331
367
|
'cli.evolve.report_link': {
|
|
332
|
-
zh: '📊
|
|
333
|
-
en: '📊 Report: omk
|
|
334
|
-
},
|
|
335
|
-
'cli.gold.created_files': {
|
|
336
|
-
zh: '已在 {dir} 创建 {n} 个文件:',
|
|
337
|
-
en: 'Created {n} files in {dir}:',
|
|
338
|
-
},
|
|
339
|
-
'cli.gold.next_step_edit_annotations': {
|
|
340
|
-
zh: '\n下一步: 编辑 annotations.yaml 加入真实标注 → 跑 omk bench gold validate',
|
|
341
|
-
en: '\nNext step: edit annotations.yaml with real annotations → run omk bench gold validate',
|
|
342
|
-
},
|
|
343
|
-
'cli.gold.validate_ok': {
|
|
344
|
-
zh: '✓ gold dataset OK — 共 {n} 条标注',
|
|
345
|
-
en: '✓ gold dataset OK — {n} annotations',
|
|
346
|
-
},
|
|
347
|
-
'cli.debias.warn_cost_doubles': {
|
|
348
|
-
zh: '\n⚠ debias-validate 会重判所有 (sample × variant), judge 成本大约翻倍。\n',
|
|
349
|
-
en: '\n⚠ debias-validate will re-judge all (sample × variant) pairs; judge cost will roughly double.\n',
|
|
350
|
-
},
|
|
351
|
-
'cli.saturation.no_data': {
|
|
352
|
-
zh: '该 report 没有 saturation 数据 (需要 --repeat ≥ 2 才会记录)。',
|
|
353
|
-
en: 'This report has no saturation data (requires --repeat ≥ 2 to record).',
|
|
354
|
-
},
|
|
355
|
-
'cli.saturation.verdict_header': {
|
|
356
|
-
zh: '\n Saturation verdict (复述持久化结果)\n',
|
|
357
|
-
en: '\n Saturation verdict (replaying persisted result)\n',
|
|
358
|
-
},
|
|
359
|
-
'cli.saturation.variant_no_trace': {
|
|
360
|
-
zh: ' {variant}: 没有 trace 数据',
|
|
361
|
-
en: ' {variant}: no trace data',
|
|
368
|
+
zh: '📊 评测报告:omk export {id} --format html\n',
|
|
369
|
+
en: '📊 Report: omk export {id} --format html\n',
|
|
362
370
|
},
|
|
363
|
-
'cli.
|
|
364
|
-
zh:
|
|
365
|
-
|
|
366
|
-
|
|
367
|
-
|
|
368
|
-
|
|
369
|
-
|
|
370
|
-
|
|
371
|
-
|
|
372
|
-
|
|
373
|
-
|
|
374
|
-
|
|
375
|
-
|
|
376
|
-
|
|
377
|
-
|
|
378
|
-
|
|
379
|
-
|
|
380
|
-
|
|
381
|
-
|
|
382
|
-
|
|
383
|
-
|
|
384
|
-
|
|
385
|
-
|
|
371
|
+
'cli.help.product_main': {
|
|
372
|
+
zh: `
|
|
373
|
+
oh-my-knowledge — 知识载体工作台
|
|
374
|
+
|
|
375
|
+
用法:
|
|
376
|
+
omk init [dir] 初始化一个 skill 评测项目
|
|
377
|
+
omk doctor [path] 静态健康检查:结构、依赖、样本、配置、污染风险
|
|
378
|
+
omk eval [options] 离线评测:比较版本,输出 verdict + report
|
|
379
|
+
omk observe <sessions-dir> 线上观测:真实 session、gap、失败率、inbox
|
|
380
|
+
omk improve <report-id> 改进建议:样本质量、失败聚类、skill patch 线索
|
|
381
|
+
omk export <report-id> [options] 证据导出:PR / CI / audit
|
|
382
|
+
omk studio 打开本地工作台
|
|
383
|
+
|
|
384
|
+
主路径:
|
|
385
|
+
omk doctor
|
|
386
|
+
omk eval --control code-review-v1 --treatment code-review-v2
|
|
387
|
+
omk observe ~/.claude/projects/<project>
|
|
388
|
+
omk improve <report-id>
|
|
389
|
+
omk export <report-id> --format github-summary
|
|
390
|
+
omk studio
|
|
391
|
+
|
|
392
|
+
通用选项:
|
|
393
|
+
--lang <zh|en> CLI 输出语言(默认:zh,也可设 OMK_LANG)
|
|
394
|
+
|
|
395
|
+
运行 'omk <command> --help' 查看单个命令的参数。
|
|
396
|
+
`,
|
|
397
|
+
en: `
|
|
398
|
+
oh-my-knowledge — Knowledge Artifact Workbench
|
|
399
|
+
|
|
400
|
+
Usage:
|
|
401
|
+
omk init [dir] Scaffold a skill evaluation project
|
|
402
|
+
omk doctor [path] Static health check: structure, deps, samples, config, contamination risk
|
|
403
|
+
omk eval [options] Offline evaluation: compare versions, emit verdict + report
|
|
404
|
+
omk observe <sessions-dir> Production observation: sessions, gaps, failure rate, inbox
|
|
405
|
+
omk improve <report-id> Improvement advice: sample quality, failure clusters, skill patch hints
|
|
406
|
+
omk export <report-id> [options] Evidence export for PR / CI / audit
|
|
407
|
+
omk studio Open the local workbench
|
|
408
|
+
|
|
409
|
+
Main workflow:
|
|
410
|
+
omk doctor
|
|
411
|
+
omk eval --control code-review-v1 --treatment code-review-v2
|
|
412
|
+
omk observe ~/.claude/projects/<project>
|
|
413
|
+
omk improve <report-id>
|
|
414
|
+
omk export <report-id> --format github-summary
|
|
415
|
+
omk studio
|
|
416
|
+
|
|
417
|
+
Common options:
|
|
418
|
+
--lang <zh|en> CLI output language (default: zh, or set OMK_LANG)
|
|
419
|
+
|
|
420
|
+
Run 'omk <command> --help' for command-specific options.
|
|
421
|
+
`,
|
|
386
422
|
},
|
|
387
|
-
'cli.
|
|
388
|
-
zh:
|
|
389
|
-
|
|
423
|
+
'cli.help.init_usage': {
|
|
424
|
+
zh: `
|
|
425
|
+
omk init — 初始化 skill 评测项目
|
|
426
|
+
|
|
427
|
+
用法:
|
|
428
|
+
omk init [dir]
|
|
429
|
+
|
|
430
|
+
生成内容:
|
|
431
|
+
eval-samples.json 示例评测用例
|
|
432
|
+
skills/code-review-v1/SKILL.md 基线 skill
|
|
433
|
+
skills/code-review-v2/SKILL.md 实验组 skill
|
|
434
|
+
|
|
435
|
+
下一步:
|
|
436
|
+
1. 编辑 eval-samples.json,替换成你的真实评测用例
|
|
437
|
+
2. 编辑两个 SKILL.md,填入要对比的 skill 版本
|
|
438
|
+
3. 运行 omk eval --control code-review-v1 --treatment code-review-v2
|
|
439
|
+
`,
|
|
440
|
+
en: `
|
|
441
|
+
omk init — scaffold a skill evaluation project
|
|
442
|
+
|
|
443
|
+
Usage:
|
|
444
|
+
omk init [dir]
|
|
445
|
+
|
|
446
|
+
Generated files:
|
|
447
|
+
eval-samples.json Example test cases
|
|
448
|
+
skills/code-review-v1/SKILL.md Baseline skill
|
|
449
|
+
skills/code-review-v2/SKILL.md Treatment skill
|
|
450
|
+
|
|
451
|
+
Next steps:
|
|
452
|
+
1. Edit eval-samples.json with your real test cases
|
|
453
|
+
2. Edit both SKILL.md files with the skill versions to compare
|
|
454
|
+
3. Run omk eval --control code-review-v1 --treatment code-review-v2
|
|
455
|
+
`,
|
|
390
456
|
},
|
|
391
|
-
'cli.help.
|
|
457
|
+
'cli.help.eval': {
|
|
392
458
|
zh: `
|
|
393
|
-
|
|
459
|
+
omk eval — 离线评测 skill 版本,并给出 ship/no-ship verdict
|
|
394
460
|
|
|
395
|
-
|
|
396
|
-
omk
|
|
397
|
-
omk
|
|
398
|
-
omk
|
|
399
|
-
omk bench init [dir] 初始化一个评测项目
|
|
400
|
-
omk bench gen-samples [skill] 从 skill 内容生成 eval-samples
|
|
401
|
-
omk bench diff <id1> <id2> 对比两份评测报告
|
|
402
|
-
omk bench evolve <skill> 通过迭代评测自我改进 skill
|
|
403
|
-
|
|
404
|
-
omk doctor [path] skill 健康检查(评测前置门禁)
|
|
405
|
-
omk analyze <dir> 分析 cc session trace, 生成 skill 健康度日报 (v0.18)
|
|
406
|
-
|
|
407
|
-
bench run 选项:
|
|
408
|
-
|
|
409
|
-
--samples <path> 用例文件 (默认: eval-samples.json)
|
|
410
|
-
--skill-dir <path> skill 定义目录 (默认: skills)
|
|
411
|
-
--control <expr> 对照组 variant 表达式 (实验角色 = control)
|
|
412
|
-
--treatment <v1,v2> 实验组 variant 表达式 (逗号分隔; 角色 = treatment)
|
|
413
|
-
每个 variant 表达式解析为一个 artifact 加上可选运行时上下文:
|
|
414
|
-
"baseline" — 裸模型, 不注入 artifact
|
|
415
|
-
"git:name" — 来自最后一次 commit 的 artifact
|
|
416
|
-
"git:ref:name" — 来自指定 commit 的 artifact
|
|
417
|
-
带 "/" 的路径 — 直接来自文件 (例如 ./v1.md)
|
|
418
|
-
"name@/cwd" — 附加运行时上下文 / cwd
|
|
419
|
-
--control 和 --treatment 至少要给一个。
|
|
420
|
-
--config <path> YAML/JSON 配置文件 (evaluation-as-code)。
|
|
421
|
-
在一个文件里声明 samples + variants + model + executor。
|
|
422
|
-
CLI flag 会覆盖配置文件中的同名字段。
|
|
423
|
-
配置中的相对路径相对于配置文件所在目录解析。
|
|
424
|
-
--model <name> 任务执行模型 (默认: sonnet)
|
|
425
|
-
--output-dir <path> 报告输出目录 (默认: ~/.oh-my-knowledge/reports/)
|
|
426
|
-
--no-judge 跳过 LLM 评委
|
|
427
|
-
--no-cache 禁用结果缓存
|
|
428
|
-
--dry-run 预览任务但不执行
|
|
429
|
-
--blind 双盲 A/B 模式: 报告里隐藏 variant 名称
|
|
430
|
-
--concurrency <n> 并发任务数 (默认: 1)
|
|
431
|
-
--timeout <seconds> 单任务执行超时 (秒, 默认: 120)
|
|
432
|
-
--repeat <n> 跑 N 轮做方差分析 (默认: 1)
|
|
433
|
-
--judge-repeat <n> 每个 (sample × dimension) 调 LLM 评委 N 次评估
|
|
434
|
-
自洽性 (默认: 1)。多轮间高 stddev = 评委在该评分维度
|
|
435
|
-
上不稳定, 分数有噪声。
|
|
436
|
-
--judge-models <list> 评委配置, 逗号分隔的 executor:model, 如
|
|
437
|
-
claude:haiku 或 claude:opus,openai:gpt-4o。
|
|
438
|
-
1 条 = 单评委 (默认 claude:haiku); ≥ 2 条 = ensemble,
|
|
439
|
-
每个评委对所有 (sample × dimension) 打分, 报告
|
|
440
|
-
含每评委分布 + Pearson / MAD 评委间一致性。能反驳
|
|
441
|
-
"Claude 评委评 Claude 同模态偏置" 的质疑。可与
|
|
442
|
-
--judge-repeat 组合。成本 ~ N_judges × N_repeat × N_samples。
|
|
443
|
-
--bootstrap 计算 bootstrap 置信区间 (无分布假设, 对 LLM 序数评分
|
|
444
|
-
比 t 区间更靠谱)。给出每个 variant 均值 CI + treatment
|
|
445
|
-
vs control 差值的 pairwise CI (CI 不跨 0 即显著)。
|
|
446
|
-
同时报告 t 区间和 bootstrap, 旧工具仍可用。
|
|
447
|
-
--bootstrap-samples <n> bootstrap 重采样次数 (默认 1000)。N>10000 触发
|
|
448
|
-
stderr 警告提示耗时。
|
|
449
|
-
--retry <n> 失败任务最多重试 N 次, 指数退避 (默认: 0)
|
|
450
|
-
--resume <report-id> 从历史报告恢复, 跳过已完成任务
|
|
451
|
-
--executor <name> 执行器: claude / claude-sdk / codex / openai / gemini /
|
|
452
|
-
anthropic-api / openai-api, 或任意 shell 命令 (例如 "python my_provider.py")
|
|
453
|
-
--batch 批量评测:每个 skill 独立 vs baseline
|
|
454
|
-
需要每个 skill 有配对的 {name}.eval-samples.json
|
|
455
|
-
--skip-connectivity 跳过 LLM 模型连通性检测 (--resume 时自动跳过)
|
|
456
|
-
--mcp-config <path> 通过 MCP server 抓 URL 用的 MCP 配置文件
|
|
457
|
-
(默认: 当前目录下的 .mcp.json)
|
|
458
|
-
--no-serve 评测后不自动启动报告 server
|
|
459
|
-
--verbose 打印每个用例的详细进度 (执行结果 / 评分阶段)
|
|
460
|
-
--layered-stats 默认在 HTML 报告里展开三层 (fact/behavior/judge) 独立
|
|
461
|
-
显著性细分。不加这个 flag 时, 细分会折叠在每个对比下
|
|
462
|
-
的 click-to-expand summary 里。
|
|
463
|
-
--strict-baseline (默认开启) 对 baseline-kind variant 强制隔离 skill 自动
|
|
464
|
-
发现 + Skill 工具调用, 切断 ~/.claude/skills/ 污染路径,
|
|
465
|
-
保证 skill 评测的 construct validity。eval.yaml 显式
|
|
466
|
-
allowedSkills 优先。
|
|
467
|
-
--no-strict-baseline 显式关闭 strict-baseline (baseline 走默认 SDK skill
|
|
468
|
-
全发现)。少数场景下可能想要这个 (例如评测 skill 文档
|
|
469
|
-
对默认全发现行为的增量影响)。开启时 pre-flight 会
|
|
470
|
-
stderr 提醒, 因为 verdict / Δ 易受污染。
|
|
471
|
-
|
|
472
|
-
bench gate 选项:
|
|
473
|
-
(与 bench run 相同, 额外加:)
|
|
474
|
-
--threshold <number> 三层 gate 阈值 (fact / behavior / LLM judge), 独立应用
|
|
475
|
-
到每一层。任一层低于阈值即失败 — 防止合成均值掩盖单层
|
|
476
|
-
崩塌。默认: 3.5。如果三层全空 (没有 assertion 也没在
|
|
477
|
-
eval-samples 里定义 rubric), gate 失败并提示配置问题,
|
|
478
|
-
不走合成 fallback。
|
|
479
|
-
--trivial-diff <num> 实际可忽略的最小 diff (默认 0.1)。bootstrap diff CI
|
|
480
|
-
显著但 |Δ| 小于此值视为"统计有效但实际无意义",标
|
|
481
|
-
CAUTIOUS 不给 PROGRESS。
|
|
482
|
-
|
|
483
|
-
内部 = bench run + bench verdict, exit code 与 bench verdict 对齐:
|
|
484
|
-
PROGRESS / SOLO-PASS → 0; NOISE / UNDERPOWERED / CAUTIOUS / REGRESS → 1。
|
|
485
|
-
数据 underpowered 时直接 FAIL, 堵住"单轮过 PASS 就 deploy"漏洞。
|
|
486
|
-
|
|
487
|
-
bench report 选项:
|
|
488
|
-
--port <number> server 端口 (默认: 7799)
|
|
489
|
-
--reports-dir <path> 报告目录 (默认: ~/.oh-my-knowledge/reports/)
|
|
490
|
-
--export <id> 把报告导出为独立 HTML 文件
|
|
491
|
-
--dev 开发模式: lib/ 文件改动时自动重启
|
|
492
|
-
|
|
493
|
-
bench gen-samples 选项:
|
|
494
|
-
--batch 为所有还没 eval-samples 的 skill 生成
|
|
495
|
-
--count <n> 每个 skill 生成多少条用例 (默认: 5)
|
|
496
|
-
--model <name> 生成用的模型 (默认: sonnet)
|
|
497
|
-
--skill-dir <path> skill 目录 (默认: skills), 配合 --batch 用
|
|
498
|
-
|
|
499
|
-
analyze 选项:
|
|
500
|
-
<dir> 输入: cc session JSONL 文件 / 目录
|
|
501
|
-
(例如 ~/.claude/projects/<slug>)
|
|
502
|
-
--kb <path> 知识库根路径 (默认: 从 trace cwd 自动推断)
|
|
503
|
-
--last <duration> 时间窗口, 例如 "7d" / "30d" (默认: 全部)
|
|
504
|
-
--from <iso> 窗口起点 (ISO8601), 优先级高于 --last
|
|
505
|
-
--to <iso> 窗口终点 (ISO8601), 优先级高于 --last
|
|
506
|
-
--skills <n1,n2,...> 白名单要分析的 skill (默认: 全部)
|
|
507
|
-
--output-dir <path> 输出目录 (默认: ~/.oh-my-knowledge/analyses/)
|
|
508
|
-
|
|
509
|
-
bench evolve 选项:
|
|
510
|
-
--rounds <n> 最大演化轮数 (默认: 5)
|
|
511
|
-
--target <score> 达到该分数即提前停止
|
|
512
|
-
--samples <path> 用例文件 (默认: eval-samples.json)
|
|
513
|
-
--model <name> 任务执行模型 (默认: sonnet)
|
|
514
|
-
--judge-models <executor:model> 评委 (默认: claude:haiku, evolve 仅支持单评委)
|
|
515
|
-
--improve-model <name> 生成改进版的模型 (默认: sonnet)
|
|
516
|
-
--concurrency <n> 并发评测任务数 (默认: 1)
|
|
517
|
-
--timeout <seconds> 单任务执行超时 (秒, 默认: 120)
|
|
518
|
-
--executor <name> 执行器 (默认: claude)
|
|
519
|
-
|
|
520
|
-
通用选项:
|
|
521
|
-
--lang <zh|en> CLI 输出语言 (默认: zh, 也可设 OMK_LANG 环境变量)
|
|
461
|
+
用法:
|
|
462
|
+
omk eval --control <variant> --treatment <variant> [options]
|
|
463
|
+
omk eval gold <init|validate|compare> ...
|
|
464
|
+
omk eval debias length <report-id> ...
|
|
522
465
|
|
|
523
|
-
|
|
524
|
-
|
|
525
|
-
|
|
526
|
-
|
|
527
|
-
|
|
528
|
-
|
|
529
|
-
|
|
530
|
-
|
|
531
|
-
|
|
532
|
-
|
|
533
|
-
|
|
534
|
-
|
|
535
|
-
|
|
536
|
-
|
|
466
|
+
常用选项:
|
|
467
|
+
--samples <path> 用例文件(默认:eval-samples.json)
|
|
468
|
+
--skill-dir <path> skill 目录(默认:skills)
|
|
469
|
+
--control <expr> 对照组 variant
|
|
470
|
+
--treatment <v1,v2> 实验组 variant,逗号分隔
|
|
471
|
+
--config <path> eval.yaml / JSON 配置
|
|
472
|
+
--executor <name> 执行器:claude / claude-sdk / codex / openai / gemini / custom
|
|
473
|
+
--model <name> 任务执行模型(默认:sonnet)
|
|
474
|
+
--judge-models <list> 评委配置,例如 claude:haiku 或 claude:opus,openai:gpt-4o
|
|
475
|
+
--dry-run 预览任务,不调用模型
|
|
476
|
+
--batch 批量评测:每个 skill 独立 vs baseline
|
|
477
|
+
--bootstrap 显式开启 bootstrap CI;omk eval 默认会自动开启
|
|
478
|
+
--bootstrap-samples <n> bootstrap 重采样次数(默认:1000)
|
|
479
|
+
--threshold <number> 三层 gate 阈值(默认:3.5)
|
|
480
|
+
--trivial-diff <number> 实际可忽略 diff(默认:0.1)
|
|
481
|
+
--report-only / --no-gate 生成报告并打印 verdict,但始终 exit 0
|
|
482
|
+
--no-serve 评测后不自动启动报告 server
|
|
483
|
+
|
|
484
|
+
示例:
|
|
485
|
+
omk eval --control code-review-v1 --treatment code-review-v2
|
|
486
|
+
omk eval --config eval.yaml
|
|
487
|
+
omk eval gold compare v1-vs-v2-20260505-1200 --gold-dir gold-dataset
|
|
537
488
|
`,
|
|
538
489
|
en: `
|
|
539
|
-
|
|
490
|
+
omk eval — run offline skill evaluation and emit a ship/no-ship verdict
|
|
540
491
|
|
|
541
492
|
Usage:
|
|
542
|
-
omk
|
|
543
|
-
omk
|
|
544
|
-
omk
|
|
545
|
-
omk bench init [dir] Scaffold a new eval project
|
|
546
|
-
omk bench gen-samples [skill] Generate eval-samples from skill content
|
|
547
|
-
omk bench diff <id1> <id2> Compare two evaluation reports
|
|
548
|
-
omk bench evolve <skill> Self-improve a skill through iterative evaluation
|
|
549
|
-
|
|
550
|
-
omk doctor [path] skill health check (pre-eval gate)
|
|
551
|
-
omk analyze <dir> Analyze cc session trace(s), produce skill health report (v0.18)
|
|
552
|
-
|
|
553
|
-
Options for "bench run":
|
|
554
|
-
|
|
555
|
-
--samples <path> Sample file (default: eval-samples.json)
|
|
556
|
-
--skill-dir <path> Skill definitions directory (default: skills)
|
|
557
|
-
--control <expr> Control-group variant expression (experiment role = control)
|
|
558
|
-
--treatment <v1,v2> Treatment-group variant expressions (comma-separated; role = treatment)
|
|
559
|
-
Each variant expression resolves to an artifact and optional runtime context:
|
|
560
|
-
"baseline" — bare model, no artifact injected
|
|
561
|
-
"git:name" — artifact from last commit
|
|
562
|
-
"git:ref:name" — artifact from specific commit
|
|
563
|
-
path with "/" — artifact from file directly (e.g. ./v1.md)
|
|
564
|
-
"name@/cwd" — attach runtime context / cwd
|
|
565
|
-
At least one of --control / --treatment must be provided.
|
|
566
|
-
--config <path> YAML/JSON config file (evaluation-as-code).
|
|
567
|
-
Declares samples + variants + model + executor in one file.
|
|
568
|
-
CLI flags override config fields when both are provided.
|
|
569
|
-
Relative paths inside the config are resolved against its directory.
|
|
570
|
-
--model <name> Task execution model (default: sonnet)
|
|
571
|
-
--output-dir <path> Report output directory (default: ~/.oh-my-knowledge/reports/)
|
|
572
|
-
--no-judge Skip LLM judging
|
|
573
|
-
--no-cache Disable result caching
|
|
574
|
-
--dry-run Preview tasks without executing
|
|
575
|
-
--blind Blind A/B mode: hide variant names in report
|
|
576
|
-
--concurrency <n> Number of parallel tasks (default: 1)
|
|
577
|
-
--timeout <seconds> Executor timeout per task in seconds (default: 120)
|
|
578
|
-
--repeat <n> Run evaluation N times for variance analysis (default: 1)
|
|
579
|
-
--judge-repeat <n> Call LLM judge N times per (sample × dimension) for self-
|
|
580
|
-
consistency (default: 1). High stddev across runs = the
|
|
581
|
-
judge is unstable on this rubric and the score is noisy.
|
|
582
|
-
--judge-models <list> Judge configuration. Comma-separated executor:model pairs,
|
|
583
|
-
e.g. claude:haiku or claude:opus,openai:gpt-4o.
|
|
584
|
-
1 entry = single judge (default claude:haiku); ≥ 2 entries
|
|
585
|
-
= ensemble — every judge scores each (sample × dimension);
|
|
586
|
-
the report includes per-judge breakdown + Pearson/MAD
|
|
587
|
-
inter-judge agreement, which refutes "Claude judges Claude
|
|
588
|
-
same-modality bias" critique. Combines with --judge-repeat.
|
|
589
|
-
Cost ~ N_judges × N_repeat × N_samples.
|
|
590
|
-
--bootstrap Compute bootstrap confidence intervals (distribution-free,
|
|
591
|
-
preferred over t-interval for ordinal LLM scores). Adds
|
|
592
|
-
per-variant CI on the mean + pairwise CI on treatment-vs-
|
|
593
|
-
control difference (significant=0 outside CI). Reports both
|
|
594
|
-
t-interval and bootstrap so old tooling still works.
|
|
595
|
-
--bootstrap-samples <n> Number of bootstrap resamples (default 1000). N>10000
|
|
596
|
-
triggers a stderr warning about runtime cost.
|
|
597
|
-
--retry <n> Retry failed tasks up to N times with exponential backoff (default: 0)
|
|
598
|
-
--resume <report-id> Resume from a previous report, skipping completed tasks
|
|
599
|
-
--executor <name> Executor: claude, claude-sdk, codex, openai, gemini,
|
|
600
|
-
anthropic-api, openai-api, or any shell command (e.g. "python my_provider.py")
|
|
601
|
-
--batch Batch evaluation: each skill independently against baseline
|
|
602
|
-
Requires {name}.eval-samples.json paired with each skill
|
|
603
|
-
--skip-connectivity Skip LLM model connectivity check (auto-skipped when --resume)
|
|
604
|
-
--mcp-config <path> MCP config file for URL fetching via MCP servers
|
|
605
|
-
(default: .mcp.json in current directory)
|
|
606
|
-
--no-serve Skip auto-starting report server after evaluation
|
|
607
|
-
--verbose Print detailed progress for each sample (exec result, grading phases)
|
|
608
|
-
--layered-stats Expand the three-layer (fact/behavior/judge) independent
|
|
609
|
-
significance breakdown in the HTML report by default.
|
|
610
|
-
Without this flag, the breakdown is collapsed behind a
|
|
611
|
-
click-to-expand summary under each comparison.
|
|
612
|
-
--strict-baseline (default ON) Isolate skill auto-discovery + Skill tool
|
|
613
|
-
use for baseline-kind variants. Cuts the ~/.claude/skills/
|
|
614
|
-
contamination path so skill evaluations have valid
|
|
615
|
-
construct validity. Explicit eval.yaml allowedSkills
|
|
616
|
-
takes precedence.
|
|
617
|
-
--no-strict-baseline Explicitly turn strict-baseline OFF (baseline sees all
|
|
618
|
-
auto-discovered skills). Use only in narrow scenarios
|
|
619
|
-
(e.g. measuring how much a skill doc adds on top of
|
|
620
|
-
full default discovery). Pre-flight emits a stderr
|
|
621
|
-
warning when this flag is set, because
|
|
622
|
-
verdict / Δ are vulnerable to skill contamination.
|
|
623
|
-
|
|
624
|
-
Options for "bench gate":
|
|
625
|
-
(same as "bench run", plus:)
|
|
626
|
-
--threshold <number> Three-layer gate threshold (fact / behavior / LLM judge),
|
|
627
|
-
applied INDEPENDENTLY to each layer. ANY layer below
|
|
628
|
-
threshold fails the gate — prevents composite averaging
|
|
629
|
-
from masking a single-layer collapse. Default: 3.5.
|
|
630
|
-
If all three layers are absent (no
|
|
631
|
-
assertions and no rubric defined in eval-samples), the
|
|
632
|
-
gate FAILS with a configuration hint — no composite fallback.
|
|
633
|
-
--trivial-diff <num> Smallest diff to treat as practically meaningful
|
|
634
|
-
(default 0.1). Bootstrap diff CI may be statistically
|
|
635
|
-
significant but with |Δ| < this value, treated as
|
|
636
|
-
CAUTIOUS rather than PROGRESS.
|
|
637
|
-
|
|
638
|
-
Internally = bench run + bench verdict. Exit code aligns with bench verdict:
|
|
639
|
-
PROGRESS / SOLO-PASS → 0; NOISE / UNDERPOWERED / CAUTIOUS / REGRESS → 1.
|
|
640
|
-
Underpowered runs fail directly — closes the "single-run PASS = deploy" loophole.
|
|
641
|
-
|
|
642
|
-
Options for "bench report":
|
|
643
|
-
--port <number> Server port (default: 7799)
|
|
644
|
-
--reports-dir <path> Reports directory (default: ~/.oh-my-knowledge/reports/)
|
|
645
|
-
--export <id> Export report as standalone HTML file
|
|
646
|
-
--dev Dev mode: auto-restart on lib/ file changes
|
|
647
|
-
|
|
648
|
-
Options for "bench gen-samples":
|
|
649
|
-
--batch Generate for all skills missing eval-samples
|
|
650
|
-
--count <n> Number of samples to generate per skill (default: 5)
|
|
651
|
-
--model <name> Model for generation (default: sonnet)
|
|
652
|
-
--skill-dir <path> Skill directory (default: skills), used with --batch
|
|
653
|
-
|
|
654
|
-
Options for "analyze":
|
|
655
|
-
<dir> Input: cc session JSONL file / dir (e.g. ~/.claude/projects/<slug>)
|
|
656
|
-
--kb <path> Knowledge base root (default: auto-infer from trace cwd)
|
|
657
|
-
--last <duration> Time window like "7d" / "30d" (default: all)
|
|
658
|
-
--from <iso> Window start (ISO8601), takes precedence over --last
|
|
659
|
-
--to <iso> Window end (ISO8601), takes precedence over --last
|
|
660
|
-
--skills <n1,n2,...> Whitelist skills to analyze (default: all)
|
|
661
|
-
--output-dir <path> Output dir (default: ~/.oh-my-knowledge/analyses/)
|
|
662
|
-
|
|
663
|
-
Options for "bench evolve":
|
|
664
|
-
--rounds <n> Maximum evolution rounds (default: 5)
|
|
665
|
-
--target <score> Stop early when score reaches this threshold
|
|
666
|
-
--samples <path> Sample file (default: eval-samples.json)
|
|
667
|
-
--model <name> Task execution model (default: sonnet)
|
|
668
|
-
--judge-models <executor:model> Judge config (default: claude:haiku; evolve is single-judge only)
|
|
669
|
-
--improve-model <name> Model for generating improvements (default: sonnet)
|
|
670
|
-
--concurrency <n> Parallel eval tasks (default: 1)
|
|
671
|
-
--timeout <seconds> Executor timeout per task in seconds (default: 120)
|
|
672
|
-
--executor <name> Executor to use (default: claude)
|
|
493
|
+
omk eval --control <variant> --treatment <variant> [options]
|
|
494
|
+
omk eval gold <init|validate|compare> ...
|
|
495
|
+
omk eval debias length <report-id> ...
|
|
673
496
|
|
|
674
497
|
Common options:
|
|
675
|
-
--
|
|
498
|
+
--samples <path> Sample file (default: eval-samples.json)
|
|
499
|
+
--skill-dir <path> Skill directory (default: skills)
|
|
500
|
+
--control <expr> Control variant
|
|
501
|
+
--treatment <v1,v2> Treatment variants, comma-separated
|
|
502
|
+
--config <path> eval.yaml / JSON config
|
|
503
|
+
--executor <name> Executor: claude / claude-sdk / codex / openai / gemini / custom
|
|
504
|
+
--model <name> Task execution model (default: sonnet)
|
|
505
|
+
--judge-models <list> Judge config, e.g. claude:haiku or claude:opus,openai:gpt-4o
|
|
506
|
+
--dry-run Preview tasks without model calls
|
|
507
|
+
--batch Batch evaluation: each skill independently against baseline
|
|
508
|
+
--bootstrap Enable bootstrap CI explicitly; omk eval turns it on by default
|
|
509
|
+
--bootstrap-samples <n> Bootstrap resamples (default: 1000)
|
|
510
|
+
--threshold <number> Three-layer gate threshold (default: 3.5)
|
|
511
|
+
--trivial-diff <number> Practically negligible diff (default: 0.1)
|
|
512
|
+
--report-only / --no-gate Produce the report and print verdict, but always exit 0
|
|
513
|
+
--no-serve Do not auto-start report server after evaluation
|
|
676
514
|
|
|
677
515
|
Examples:
|
|
678
|
-
omk
|
|
679
|
-
omk
|
|
680
|
-
omk
|
|
681
|
-
omk bench run --control ./old-skill.md --treatment ./new-skill.md
|
|
682
|
-
omk bench run --control baseline --treatment v1,v2,v3
|
|
683
|
-
omk bench run --config eval.yaml
|
|
684
|
-
omk bench run --config eval.yaml --model sonnet-4.6 # CLI overrides config
|
|
685
|
-
omk bench run --batch
|
|
686
|
-
omk bench run --dry-run
|
|
687
|
-
omk bench report --port 8080
|
|
688
|
-
omk bench report --export v1-vs-v2-20260326-1832
|
|
689
|
-
omk bench init my-eval
|
|
690
|
-
omk bench gen-samples skills/my-skill.md
|
|
516
|
+
omk eval --control code-review-v1 --treatment code-review-v2
|
|
517
|
+
omk eval --config eval.yaml
|
|
518
|
+
omk eval gold compare v1-vs-v2-20260505-1200 --gold-dir gold-dataset
|
|
691
519
|
`,
|
|
692
520
|
},
|
|
693
|
-
'cli.help.
|
|
694
|
-
zh:
|
|
695
|
-
|
|
696
|
-
|
|
697
|
-
|
|
698
|
-
|
|
699
|
-
|
|
700
|
-
|
|
701
|
-
|
|
702
|
-
|
|
703
|
-
|
|
704
|
-
|
|
705
|
-
|
|
706
|
-
|
|
707
|
-
|
|
708
|
-
|
|
709
|
-
|
|
710
|
-
|
|
711
|
-
|
|
712
|
-
|
|
713
|
-
|
|
714
|
-
|
|
715
|
-
|
|
716
|
-
|
|
717
|
-
|
|
718
|
-
|
|
719
|
-
|
|
720
|
-
},
|
|
721
|
-
'cli.help.gold': {
|
|
722
|
-
zh: [
|
|
723
|
-
'',
|
|
724
|
-
'用法: omk bench gold <subcommand>',
|
|
725
|
-
'',
|
|
726
|
-
'子命令:',
|
|
727
|
-
' init [--out <dir>] [--annotator <id>] 生成空白 gold dataset 模板',
|
|
728
|
-
' validate <dir> 校验数据集结构',
|
|
729
|
-
' compare <reportId> --gold-dir <dir> 与已有 report 计算 α / κ / Pearson',
|
|
730
|
-
' [--variant <name>] [--reports-dir <d>]',
|
|
731
|
-
' [--bootstrap-samples N] [--seed N]',
|
|
732
|
-
'',
|
|
733
|
-
].join('\n'),
|
|
734
|
-
en: [
|
|
735
|
-
'',
|
|
736
|
-
'Usage: omk bench gold <subcommand>',
|
|
737
|
-
'',
|
|
738
|
-
'Subcommands:',
|
|
739
|
-
' init [--out <dir>] [--annotator <id>] create a blank gold dataset template',
|
|
740
|
-
' validate <dir> validate dataset structure',
|
|
741
|
-
' compare <reportId> --gold-dir <dir> compute α / κ / Pearson against an existing report',
|
|
742
|
-
' [--variant <name>] [--reports-dir <d>]',
|
|
743
|
-
' [--bootstrap-samples N] [--seed N]',
|
|
744
|
-
'',
|
|
745
|
-
].join('\n'),
|
|
521
|
+
'cli.help.eval_gold': {
|
|
522
|
+
zh: `
|
|
523
|
+
omk eval gold — 管理 human-gold 标注集
|
|
524
|
+
|
|
525
|
+
用法:
|
|
526
|
+
omk eval gold init [--out <dir>] [--annotator <name>]
|
|
527
|
+
omk eval gold validate <dir>
|
|
528
|
+
omk eval gold compare <reportId> --gold-dir <dir>
|
|
529
|
+
|
|
530
|
+
选项:
|
|
531
|
+
--reports-dir <path> 报告目录(compare 使用,默认:~/.oh-my-knowledge/reports)
|
|
532
|
+
--variant <name> 指定 report 中要对比的 variant
|
|
533
|
+
--bootstrap-samples <n> bootstrap 重采样次数(compare 使用)
|
|
534
|
+
`,
|
|
535
|
+
en: `
|
|
536
|
+
omk eval gold — manage human-gold annotation datasets
|
|
537
|
+
|
|
538
|
+
Usage:
|
|
539
|
+
omk eval gold init [--out <dir>] [--annotator <name>]
|
|
540
|
+
omk eval gold validate <dir>
|
|
541
|
+
omk eval gold compare <reportId> --gold-dir <dir>
|
|
542
|
+
|
|
543
|
+
Options:
|
|
544
|
+
--reports-dir <path> Reports directory for compare (default: ~/.oh-my-knowledge/reports)
|
|
545
|
+
--variant <name> Variant in the report to compare
|
|
546
|
+
--bootstrap-samples <n> Bootstrap resamples for compare
|
|
547
|
+
`,
|
|
746
548
|
},
|
|
747
|
-
'cli.help.
|
|
748
|
-
zh:
|
|
749
|
-
|
|
750
|
-
|
|
751
|
-
|
|
752
|
-
|
|
753
|
-
|
|
754
|
-
|
|
755
|
-
|
|
756
|
-
|
|
757
|
-
|
|
758
|
-
|
|
759
|
-
|
|
760
|
-
|
|
761
|
-
|
|
762
|
-
|
|
763
|
-
|
|
764
|
-
|
|
765
|
-
|
|
766
|
-
|
|
767
|
-
|
|
768
|
-
|
|
769
|
-
|
|
770
|
-
|
|
771
|
-
|
|
772
|
-
|
|
773
|
-
|
|
774
|
-
' --reports-dir <dir> report store dir (default: ~/.oh-my-knowledge/reports)',
|
|
775
|
-
' --samples <path> override samples file (default: from report.meta.request)',
|
|
776
|
-
' --variant <name> which variant to validate (default: first)',
|
|
777
|
-
' --judge-models <executor:model> Judge (default: from report.meta.judgeModels[0]; debias-validate is single-judge only)',
|
|
778
|
-
' --bootstrap-samples N bootstrap iterations (default 1000)',
|
|
779
|
-
' --seed N deterministic CI seed',
|
|
780
|
-
'',
|
|
781
|
-
].join('\n'),
|
|
549
|
+
'cli.help.eval_debias': {
|
|
550
|
+
zh: `
|
|
551
|
+
omk eval debias — 验证 length-debias 是否降低评委长度偏差
|
|
552
|
+
|
|
553
|
+
用法:
|
|
554
|
+
omk eval debias length <reportId> [options]
|
|
555
|
+
|
|
556
|
+
选项:
|
|
557
|
+
--samples <path> 用例文件;默认从 report.meta.request 读取
|
|
558
|
+
--reports-dir <path> 报告目录(默认:~/.oh-my-knowledge/reports)
|
|
559
|
+
--variant <name> 只验证指定 variant
|
|
560
|
+
--judge-models <executor:model> 指定单评委
|
|
561
|
+
--bootstrap-samples <n> bootstrap 重采样次数
|
|
562
|
+
`,
|
|
563
|
+
en: `
|
|
564
|
+
omk eval debias — validate whether length-debias reduces judge length bias
|
|
565
|
+
|
|
566
|
+
Usage:
|
|
567
|
+
omk eval debias length <reportId> [options]
|
|
568
|
+
|
|
569
|
+
Options:
|
|
570
|
+
--samples <path> Sample file; defaults to report.meta.request
|
|
571
|
+
--reports-dir <path> Reports directory (default: ~/.oh-my-knowledge/reports)
|
|
572
|
+
--variant <name> Validate only one variant
|
|
573
|
+
--judge-models <executor:model> Single judge to use
|
|
574
|
+
--bootstrap-samples <n> Bootstrap resamples
|
|
575
|
+
`,
|
|
782
576
|
},
|
|
783
|
-
'cli.help.
|
|
784
|
-
zh:
|
|
785
|
-
|
|
786
|
-
|
|
787
|
-
|
|
788
|
-
|
|
789
|
-
|
|
790
|
-
|
|
791
|
-
|
|
792
|
-
|
|
793
|
-
|
|
794
|
-
|
|
795
|
-
|
|
796
|
-
|
|
797
|
-
|
|
798
|
-
|
|
799
|
-
|
|
800
|
-
|
|
801
|
-
|
|
802
|
-
|
|
803
|
-
|
|
804
|
-
|
|
805
|
-
|
|
806
|
-
|
|
807
|
-
|
|
808
|
-
|
|
809
|
-
|
|
810
|
-
|
|
811
|
-
|
|
812
|
-
'',
|
|
813
|
-
'Options:',
|
|
814
|
-
' --reports-dir <dir> report store dir (default: ~/.oh-my-knowledge/reports)',
|
|
815
|
-
' --variant <name> only show one variant (default: all)',
|
|
816
|
-
'',
|
|
817
|
-
].join('\n'),
|
|
577
|
+
'cli.help.observe': {
|
|
578
|
+
zh: `
|
|
579
|
+
omk observe — 分析真实 session trace,生成 skill 健康度日报
|
|
580
|
+
|
|
581
|
+
用法:
|
|
582
|
+
omk observe <sessions-dir> [options]
|
|
583
|
+
|
|
584
|
+
选项:
|
|
585
|
+
--kb <path> 知识库根路径(默认:从 trace cwd 推断)
|
|
586
|
+
--last <duration> 时间窗口,例如 7d / 24h / 30m
|
|
587
|
+
--from <iso> 窗口起点,优先级高于 --last
|
|
588
|
+
--to <iso> 窗口终点,优先级高于 --last
|
|
589
|
+
--skills <n1,n2,...> 只分析指定 skill
|
|
590
|
+
--output-dir <path> 输出目录(默认:~/.oh-my-knowledge/analyses)
|
|
591
|
+
`,
|
|
592
|
+
en: `
|
|
593
|
+
omk observe — analyze production session traces and produce skill health reports
|
|
594
|
+
|
|
595
|
+
Usage:
|
|
596
|
+
omk observe <sessions-dir> [options]
|
|
597
|
+
|
|
598
|
+
Options:
|
|
599
|
+
--kb <path> Knowledge base root (default: infer from trace cwd)
|
|
600
|
+
--last <duration> Time window, e.g. 7d / 24h / 30m
|
|
601
|
+
--from <iso> Window start, overrides --last
|
|
602
|
+
--to <iso> Window end, overrides --last
|
|
603
|
+
--skills <n1,n2,...> Only analyze selected skills
|
|
604
|
+
--output-dir <path> Output directory (default: ~/.oh-my-knowledge/analyses)
|
|
605
|
+
`,
|
|
818
606
|
},
|
|
819
|
-
'cli.help.
|
|
820
|
-
zh:
|
|
821
|
-
|
|
822
|
-
|
|
823
|
-
|
|
824
|
-
|
|
825
|
-
|
|
826
|
-
|
|
827
|
-
|
|
828
|
-
|
|
829
|
-
|
|
830
|
-
|
|
831
|
-
|
|
832
|
-
|
|
833
|
-
|
|
834
|
-
|
|
835
|
-
|
|
836
|
-
|
|
837
|
-
|
|
838
|
-
|
|
839
|
-
|
|
840
|
-
|
|
841
|
-
|
|
842
|
-
|
|
843
|
-
|
|
844
|
-
|
|
845
|
-
|
|
846
|
-
|
|
847
|
-
|
|
848
|
-
|
|
849
|
-
|
|
850
|
-
' REGRESS significant regression — do not ship',
|
|
851
|
-
' NOISE CI crosses 0, no verdict',
|
|
852
|
-
' UNDERPOWERED not enough samples, expand N and re-run',
|
|
853
|
-
' SOLO single-variant report, nothing to compare against',
|
|
854
|
-
'',
|
|
855
|
-
'Options:',
|
|
856
|
-
' --reports-dir <dir> report store dir (default: ~/.oh-my-knowledge/reports)',
|
|
857
|
-
' --threshold <num> 3-layer gate threshold (default 3.5, matches omk bench gate)',
|
|
858
|
-
' --trivial-diff <num> "diff too small" threshold (default 0.1)',
|
|
859
|
-
' --verbose expand per-pair details',
|
|
860
|
-
'',
|
|
861
|
-
].join('\n'),
|
|
607
|
+
'cli.help.improve': {
|
|
608
|
+
zh: `
|
|
609
|
+
omk improve — 从报告或 trace 中得到下一步改进建议
|
|
610
|
+
|
|
611
|
+
用法:
|
|
612
|
+
omk improve <report-id> 输出样本质量诊断和改进计划
|
|
613
|
+
omk improve plan <report-id> 同上,显式 plan 子命令
|
|
614
|
+
omk improve failures <report-id> 聚类失败用例,生成根因和修复方向
|
|
615
|
+
omk improve samples [skill] 为 skill 生成或补齐 eval samples
|
|
616
|
+
omk improve skill <skill> 基于评测循环尝试改进 skill
|
|
617
|
+
|
|
618
|
+
示例:
|
|
619
|
+
omk improve v1-vs-v2-20260505-1200
|
|
620
|
+
omk improve samples skills/code-review/SKILL.md
|
|
621
|
+
omk improve failures v1-vs-v2-20260505-1200
|
|
622
|
+
`,
|
|
623
|
+
en: `
|
|
624
|
+
omk improve — get next-step improvement advice from reports or traces
|
|
625
|
+
|
|
626
|
+
Usage:
|
|
627
|
+
omk improve <report-id> Print sample diagnostics and repair plan
|
|
628
|
+
omk improve plan <report-id> Same as above, explicit plan subcommand
|
|
629
|
+
omk improve failures <report-id> Cluster failed cases into root causes and fixes
|
|
630
|
+
omk improve samples [skill] Generate or fill eval samples for a skill
|
|
631
|
+
omk improve skill <skill> Try to improve a skill through evaluation loops
|
|
632
|
+
|
|
633
|
+
Examples:
|
|
634
|
+
omk improve v1-vs-v2-20260505-1200
|
|
635
|
+
omk improve samples skills/code-review/SKILL.md
|
|
636
|
+
omk improve failures v1-vs-v2-20260505-1200
|
|
637
|
+
`,
|
|
862
638
|
},
|
|
863
|
-
'cli.help.
|
|
639
|
+
'cli.help.improve_plan': {
|
|
864
640
|
zh: [
|
|
865
641
|
'',
|
|
866
|
-
'用法: omk
|
|
642
|
+
'用法: omk improve <reportId> [options]',
|
|
643
|
+
' omk improve plan <reportId> [options]',
|
|
867
644
|
'',
|
|
868
645
|
'诊断用例集本身的质量问题: 区分度低 / 重复 / 歧义 / 成本异常 / 全 fail。',
|
|
869
|
-
'回答 "
|
|
646
|
+
'回答 "评测结论是否被坏用例污染"。',
|
|
870
647
|
'',
|
|
871
648
|
'选项:',
|
|
872
649
|
' --reports-dir <dir> 报告存储目录',
|
|
@@ -881,11 +658,12 @@ Examples:
|
|
|
881
658
|
].join('\n'),
|
|
882
659
|
en: [
|
|
883
660
|
'',
|
|
884
|
-
'Usage: omk
|
|
661
|
+
'Usage: omk improve <reportId> [options]',
|
|
662
|
+
' omk improve plan <reportId> [options]',
|
|
885
663
|
'',
|
|
886
664
|
'Diagnose quality issues in the sample set itself: low discrimination /',
|
|
887
665
|
'duplicates / ambiguity / cost anomalies / all-fail. Answers "is the verdict',
|
|
888
|
-
'tainted by bad samples?"
|
|
666
|
+
'tainted by bad samples?".',
|
|
889
667
|
'',
|
|
890
668
|
'Options:',
|
|
891
669
|
' --reports-dir <dir> report store dir',
|
|
@@ -899,10 +677,10 @@ Examples:
|
|
|
899
677
|
'',
|
|
900
678
|
].join('\n'),
|
|
901
679
|
},
|
|
902
|
-
'cli.help.
|
|
680
|
+
'cli.help.improve_failures': {
|
|
903
681
|
zh: [
|
|
904
682
|
'',
|
|
905
|
-
'用法: omk
|
|
683
|
+
'用法: omk improve failures <reportId> [options]',
|
|
906
684
|
'',
|
|
907
685
|
'把已有 report 的失败用例喂给一次 LLM 调用, 自动聚类并给出修复建议。',
|
|
908
686
|
'失败定义: compositeScore < threshold 或 ok=false。',
|
|
@@ -917,7 +695,7 @@ Examples:
|
|
|
917
695
|
].join('\n'),
|
|
918
696
|
en: [
|
|
919
697
|
'',
|
|
920
|
-
'Usage: omk
|
|
698
|
+
'Usage: omk improve failures <reportId> [options]',
|
|
921
699
|
'',
|
|
922
700
|
'Feed failing samples from an existing report to a single LLM call, auto-cluster',
|
|
923
701
|
'them, and produce per-cluster fix suggestions.',
|
|
@@ -932,6 +710,216 @@ Examples:
|
|
|
932
710
|
'',
|
|
933
711
|
].join('\n'),
|
|
934
712
|
},
|
|
713
|
+
'cli.help.improve_samples': {
|
|
714
|
+
zh: `
|
|
715
|
+
omk improve samples — 生成或补齐 eval-samples 评测用例
|
|
716
|
+
|
|
717
|
+
用法:
|
|
718
|
+
omk improve samples <skill-path> [options]
|
|
719
|
+
omk improve samples --batch [--skill-dir <dir>] [options]
|
|
720
|
+
|
|
721
|
+
选项:
|
|
722
|
+
--count <n> 生成用例数量(默认:5)
|
|
723
|
+
--model <name> 生成模型(默认:sonnet)
|
|
724
|
+
--batch 为 skill 目录下缺少 eval-samples 的 skill 批量生成
|
|
725
|
+
--skill-dir <path> skill 目录(batch 使用,默认:skills)
|
|
726
|
+
`,
|
|
727
|
+
en: `
|
|
728
|
+
omk improve samples — generate or fill eval-samples test cases
|
|
729
|
+
|
|
730
|
+
Usage:
|
|
731
|
+
omk improve samples <skill-path> [options]
|
|
732
|
+
omk improve samples --batch [--skill-dir <dir>] [options]
|
|
733
|
+
|
|
734
|
+
Options:
|
|
735
|
+
--count <n> Number of test cases to generate (default: 5)
|
|
736
|
+
--model <name> Generation model (default: sonnet)
|
|
737
|
+
--batch Generate for skills that are missing eval-samples
|
|
738
|
+
--skill-dir <path> Skill directory for batch mode (default: skills)
|
|
739
|
+
`,
|
|
740
|
+
},
|
|
741
|
+
'cli.help.improve_skill': {
|
|
742
|
+
zh: `
|
|
743
|
+
omk improve skill — 基于评测循环迭代改进 skill
|
|
744
|
+
|
|
745
|
+
用法:
|
|
746
|
+
omk improve skill <skill-path> [options]
|
|
747
|
+
|
|
748
|
+
选项:
|
|
749
|
+
--rounds <n> 迭代轮数(默认:3)
|
|
750
|
+
--target <score> 目标分数
|
|
751
|
+
--model <name> 改进模型
|
|
752
|
+
--judge-models <executor:model> 单评委配置
|
|
753
|
+
`,
|
|
754
|
+
en: `
|
|
755
|
+
omk improve skill — improve a skill through evaluation loops
|
|
756
|
+
|
|
757
|
+
Usage:
|
|
758
|
+
omk improve skill <skill-path> [options]
|
|
759
|
+
|
|
760
|
+
Options:
|
|
761
|
+
--rounds <n> Iteration rounds (default: 3)
|
|
762
|
+
--target <score> Target score
|
|
763
|
+
--model <name> Improvement model
|
|
764
|
+
--judge-models <executor:model> Single judge config
|
|
765
|
+
`,
|
|
766
|
+
},
|
|
767
|
+
'cli.help.export': {
|
|
768
|
+
zh: `
|
|
769
|
+
omk export — 导出可贴到 PR / CI / audit 的证据包
|
|
770
|
+
|
|
771
|
+
用法:
|
|
772
|
+
omk export <report-id> [options]
|
|
773
|
+
omk export diff <report-id> [report-id] [options]
|
|
774
|
+
omk export verdict <report-id> [options]
|
|
775
|
+
omk export saturation <report-id> [options]
|
|
776
|
+
|
|
777
|
+
选项:
|
|
778
|
+
--format <format> html / markdown / github-summary(默认:html)
|
|
779
|
+
--out <path> 输出文件;markdown / github-summary 未指定时输出到 stdout
|
|
780
|
+
--reports-dir <path> 报告目录(默认:~/.oh-my-knowledge/reports)
|
|
781
|
+
|
|
782
|
+
示例:
|
|
783
|
+
omk export v1-vs-v2-20260505-1200 --format github-summary
|
|
784
|
+
omk export v1-vs-v2-20260505-1200 --format markdown --out report.md
|
|
785
|
+
omk export diff v1-vs-v2-20260505-1200 --regressions-only
|
|
786
|
+
omk export verdict v1-vs-v2-20260505-1200
|
|
787
|
+
`,
|
|
788
|
+
en: `
|
|
789
|
+
omk export — export evidence packs for PR / CI / audit
|
|
790
|
+
|
|
791
|
+
Usage:
|
|
792
|
+
omk export <report-id> [options]
|
|
793
|
+
omk export diff <report-id> [report-id] [options]
|
|
794
|
+
omk export verdict <report-id> [options]
|
|
795
|
+
omk export saturation <report-id> [options]
|
|
796
|
+
|
|
797
|
+
Options:
|
|
798
|
+
--format <format> html / markdown / github-summary (default: html)
|
|
799
|
+
--out <path> Output file; markdown / github-summary print to stdout by default
|
|
800
|
+
--reports-dir <path> Reports directory (default: ~/.oh-my-knowledge/reports)
|
|
801
|
+
|
|
802
|
+
Examples:
|
|
803
|
+
omk export v1-vs-v2-20260505-1200 --format github-summary
|
|
804
|
+
omk export v1-vs-v2-20260505-1200 --format markdown --out report.md
|
|
805
|
+
omk export diff v1-vs-v2-20260505-1200 --regressions-only
|
|
806
|
+
omk export verdict v1-vs-v2-20260505-1200
|
|
807
|
+
`,
|
|
808
|
+
},
|
|
809
|
+
'cli.help.export_diff': {
|
|
810
|
+
zh: `
|
|
811
|
+
omk export diff — 导出样本级或报告级差异
|
|
812
|
+
|
|
813
|
+
用法:
|
|
814
|
+
omk export diff <report-id> [--variant <name>] [--regressions-only] [--top <n>]
|
|
815
|
+
omk export diff <report-id-a> <report-id-b>
|
|
816
|
+
|
|
817
|
+
选项:
|
|
818
|
+
--reports-dir <path> 报告目录(默认:~/.oh-my-knowledge/reports)
|
|
819
|
+
--variant <name> 样本级 diff 的实验组 variant
|
|
820
|
+
--regressions-only 只显示回退用例
|
|
821
|
+
--top <n> 最多显示 N 条
|
|
822
|
+
`,
|
|
823
|
+
en: `
|
|
824
|
+
omk export diff — export sample-level or cross-report differences
|
|
825
|
+
|
|
826
|
+
Usage:
|
|
827
|
+
omk export diff <report-id> [--variant <name>] [--regressions-only] [--top <n>]
|
|
828
|
+
omk export diff <report-id-a> <report-id-b>
|
|
829
|
+
|
|
830
|
+
Options:
|
|
831
|
+
--reports-dir <path> Reports directory (default: ~/.oh-my-knowledge/reports)
|
|
832
|
+
--variant <name> Treatment variant for sample-level diff
|
|
833
|
+
--regressions-only Show regressions only
|
|
834
|
+
--top <n> Show at most N rows
|
|
835
|
+
`,
|
|
836
|
+
},
|
|
837
|
+
'cli.help.export_verdict': {
|
|
838
|
+
zh: `
|
|
839
|
+
omk export verdict — 输出已有报告的一行 ship/no-ship verdict
|
|
840
|
+
|
|
841
|
+
用法:
|
|
842
|
+
omk export verdict <report-id> [options]
|
|
843
|
+
|
|
844
|
+
选项:
|
|
845
|
+
--reports-dir <path> 报告目录(默认:~/.oh-my-knowledge/reports)
|
|
846
|
+
--threshold <number> 三层 gate 阈值
|
|
847
|
+
--trivial-diff <number> 实际可忽略 diff
|
|
848
|
+
--verbose 输出完整 verdict 解释
|
|
849
|
+
`,
|
|
850
|
+
en: `
|
|
851
|
+
omk export verdict — print a one-line ship/no-ship verdict for an existing report
|
|
852
|
+
|
|
853
|
+
Usage:
|
|
854
|
+
omk export verdict <report-id> [options]
|
|
855
|
+
|
|
856
|
+
Options:
|
|
857
|
+
--reports-dir <path> Reports directory (default: ~/.oh-my-knowledge/reports)
|
|
858
|
+
--threshold <number> Three-layer gate threshold
|
|
859
|
+
--trivial-diff <number> Practically negligible diff
|
|
860
|
+
--verbose Print the full verdict explanation
|
|
861
|
+
`,
|
|
862
|
+
},
|
|
863
|
+
'cli.help.export_saturation': {
|
|
864
|
+
zh: `
|
|
865
|
+
omk export saturation — 输出重复评测的饱和度证据
|
|
866
|
+
|
|
867
|
+
用法:
|
|
868
|
+
omk export saturation <report-id> [--variant <name>]
|
|
869
|
+
|
|
870
|
+
选项:
|
|
871
|
+
--reports-dir <path> 报告目录(默认:~/.oh-my-knowledge/reports)
|
|
872
|
+
--variant <name> 只输出指定 variant
|
|
873
|
+
`,
|
|
874
|
+
en: `
|
|
875
|
+
omk export saturation — print saturation evidence from repeated evaluations
|
|
876
|
+
|
|
877
|
+
Usage:
|
|
878
|
+
omk export saturation <report-id> [--variant <name>]
|
|
879
|
+
|
|
880
|
+
Options:
|
|
881
|
+
--reports-dir <path> Reports directory (default: ~/.oh-my-knowledge/reports)
|
|
882
|
+
--variant <name> Print only one variant
|
|
883
|
+
`,
|
|
884
|
+
},
|
|
885
|
+
'cli.help.studio': {
|
|
886
|
+
zh: `
|
|
887
|
+
omk studio — 打开本地知识工作台
|
|
888
|
+
|
|
889
|
+
用法:
|
|
890
|
+
omk studio [options]
|
|
891
|
+
|
|
892
|
+
选项:
|
|
893
|
+
--port <n> 本地服务端口(默认:7799)
|
|
894
|
+
--reports-dir <path> 报告目录(默认:~/.oh-my-knowledge/reports)
|
|
895
|
+
--analyses-dir <path> 观测分析目录
|
|
896
|
+
--no-open 只启动服务,不自动打开浏览器
|
|
897
|
+
--dev 开发模式:文件变化时自动重启
|
|
898
|
+
|
|
899
|
+
示例:
|
|
900
|
+
omk studio
|
|
901
|
+
omk studio --port 7798
|
|
902
|
+
omk studio --no-open
|
|
903
|
+
`,
|
|
904
|
+
en: `
|
|
905
|
+
omk studio — open the local knowledge workbench
|
|
906
|
+
|
|
907
|
+
Usage:
|
|
908
|
+
omk studio [options]
|
|
909
|
+
|
|
910
|
+
Options:
|
|
911
|
+
--port <n> Local server port (default: 7799)
|
|
912
|
+
--reports-dir <path> Reports directory (default: ~/.oh-my-knowledge/reports)
|
|
913
|
+
--analyses-dir <path> Observation analyses directory
|
|
914
|
+
--no-open Start the server without opening a browser
|
|
915
|
+
--dev Dev mode: restart on file changes
|
|
916
|
+
|
|
917
|
+
Examples:
|
|
918
|
+
omk studio
|
|
919
|
+
omk studio --port 7798
|
|
920
|
+
omk studio --no-open
|
|
921
|
+
`,
|
|
922
|
+
},
|
|
935
923
|
// sample design coverage block strings
|
|
936
924
|
'cli.diagnose.coverage_header': {
|
|
937
925
|
zh: '用例设计覆盖度 (Sample design coverage):',
|
|
@@ -1098,6 +1086,7 @@ oh-my-knowledge — omk doctor 健康检查
|
|
|
1098
1086
|
--gate 静默模式: 通过 exit 0 / 不通过 exit 1, 仅 stderr 出问题摘要
|
|
1099
1087
|
--executor <name> executor 名(仅向后兼容, doctor 不直接打 LLM)
|
|
1100
1088
|
--model <name> model 名(同上)
|
|
1089
|
+
--samples <path> 显式指定评测用例文件
|
|
1101
1090
|
--timeout <seconds> rule 执行超时(默认 8)
|
|
1102
1091
|
--lang <zh|en> 切换输出语言
|
|
1103
1092
|
|
|
@@ -1113,7 +1102,7 @@ doctor 检查项(纯静态 / 零 LLM 调用):
|
|
|
1113
1102
|
- 用例 ↔ skill 输入约定 (warn 级, 仅传 samples 时跑)
|
|
1114
1103
|
|
|
1115
1104
|
executor / judge 连通性由 evaluation preflight 负责, 不在 doctor 范围内。
|
|
1116
|
-
omk
|
|
1105
|
+
omk eval 内置 doctor 强制门禁, 不可 skip — 静态检查
|
|
1117
1106
|
零成本无理由跳过。LLM 连通性可用 --skip-connectivity 跳过 (--resume 时自动)。
|
|
1118
1107
|
`.trim() + '\n',
|
|
1119
1108
|
en: `
|
|
@@ -1131,6 +1120,7 @@ Options:
|
|
|
1131
1120
|
--gate Silent mode: exit 0 if pass, exit 1 if fail; brief stderr summary only
|
|
1132
1121
|
--executor <name> executor name (kept for compat; doctor does not call LLM)
|
|
1133
1122
|
--model <name> model name (same)
|
|
1123
|
+
--samples <path> Explicit eval samples file
|
|
1134
1124
|
--timeout <seconds> per-rule timeout (default 8)
|
|
1135
1125
|
--lang <zh|en> Output language
|
|
1136
1126
|
|
|
@@ -1146,7 +1136,7 @@ Checks (pure static / zero LLM calls):
|
|
|
1146
1136
|
- samples ↔ skill contract (warn-level, only when samples provided)
|
|
1147
1137
|
|
|
1148
1138
|
executor / judge connectivity is handled by evaluation preflight, not doctor.
|
|
1149
|
-
omk
|
|
1139
|
+
omk eval runs doctor as mandatory; no skip flag — static
|
|
1150
1140
|
checks cost nothing to run. LLM connectivity can be skipped with --skip-connectivity
|
|
1151
1141
|
(auto-skipped on --resume).
|
|
1152
1142
|
`.trim() + '\n',
|
|
@@ -1155,6 +1145,10 @@ checks cost nothing to run. LLM connectivity can be skipped with --skip-connecti
|
|
|
1155
1145
|
zh: '未在 {path} 下发现 skill 文件。\n doctor 期望 .md 文件、目录(包含 .md 或 SKILL.md)或 cwd 下的 skills/ 子目录。',
|
|
1156
1146
|
en: 'No skills found at {path}.\n doctor expects a .md file, a directory (containing .md or SKILL.md), or skills/ under cwd.',
|
|
1157
1147
|
},
|
|
1148
|
+
'cli.doctor.samples_detected': {
|
|
1149
|
+
zh: '✓ 使用评测用例文件:{path}',
|
|
1150
|
+
en: '✓ Using eval samples file: {path}',
|
|
1151
|
+
},
|
|
1158
1152
|
'cli.doctor.gate_blocked': {
|
|
1159
1153
|
zh: 'skill 健康检查未通过, 评测已中止。doctor 是评测必经环节, 无 skip 选项 — 请修复上述问题后重跑。',
|
|
1160
1154
|
en: 'skill health check failed; evaluation aborted. doctor is mandatory and not skippable — fix the issues above and re-run.',
|