oh-my-knowledge 0.26.0 → 0.28.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +105 -346
- package/README.zh.md +142 -375
- package/dist/src/analysis/report-diagnostics.d.ts +2 -2
- package/dist/src/analysis/report-diagnostics.js +2 -2
- package/dist/src/analysis/sample-diagnostics.d.ts +3 -3
- package/dist/src/analysis/sample-diagnostics.js +3 -3
- package/dist/src/authoring/generator.d.ts.map +1 -1
- package/dist/src/authoring/generator.js +9 -8
- package/dist/src/authoring/generator.js.map +1 -1
- package/dist/src/cli/cli-exit.d.ts +15 -0
- package/dist/src/cli/cli-exit.d.ts.map +1 -0
- package/dist/src/cli/cli-exit.js +19 -0
- package/dist/src/cli/cli-exit.js.map +1 -0
- package/dist/src/cli/commands/_shared.d.ts +12 -0
- package/dist/src/cli/commands/_shared.d.ts.map +1 -0
- package/dist/src/cli/commands/_shared.js +26 -0
- package/dist/src/cli/commands/_shared.js.map +1 -0
- package/dist/src/cli/commands/doctor.d.ts +2 -0
- package/dist/src/cli/commands/doctor.d.ts.map +1 -0
- package/dist/src/cli/commands/doctor.js +174 -0
- package/dist/src/cli/commands/doctor.js.map +1 -0
- package/dist/src/cli/commands/eval-gold.d.ts +2 -0
- package/dist/src/cli/commands/eval-gold.d.ts.map +1 -0
- package/dist/src/cli/commands/eval-gold.js +137 -0
- package/dist/src/cli/commands/eval-gold.js.map +1 -0
- package/dist/src/cli/commands/eval-runner.d.ts +2 -0
- package/dist/src/cli/commands/eval-runner.d.ts.map +1 -0
- package/dist/src/cli/commands/eval-runner.js +299 -0
- package/dist/src/cli/commands/eval-runner.js.map +1 -0
- package/dist/src/cli/commands/eval.d.ts +2 -0
- package/dist/src/cli/commands/eval.d.ts.map +1 -0
- package/dist/src/cli/commands/eval.js +11 -0
- package/dist/src/cli/commands/eval.js.map +1 -0
- package/dist/src/cli/commands/evolve.d.ts +2 -0
- package/dist/src/cli/commands/evolve.d.ts.map +1 -0
- package/dist/src/cli/commands/evolve.js +115 -0
- package/dist/src/cli/commands/evolve.js.map +1 -0
- package/dist/src/cli/commands/init.d.ts +2 -0
- package/dist/src/cli/commands/init.d.ts.map +1 -0
- package/dist/src/cli/commands/init.js +110 -0
- package/dist/src/cli/commands/init.js.map +1 -0
- package/dist/src/cli/commands/observe.d.ts +2 -0
- package/dist/src/cli/commands/observe.d.ts.map +1 -0
- package/dist/src/cli/commands/observe.js +77 -0
- package/dist/src/cli/commands/observe.js.map +1 -0
- package/dist/src/cli/commands/registry.d.ts +11 -0
- package/dist/src/cli/commands/registry.d.ts.map +1 -0
- package/dist/src/cli/commands/registry.js +24 -0
- package/dist/src/cli/commands/registry.js.map +1 -0
- package/dist/src/cli/commands/sample.d.ts +2 -0
- package/dist/src/cli/commands/sample.d.ts.map +1 -0
- package/dist/src/cli/commands/sample.js +121 -0
- package/dist/src/cli/commands/sample.js.map +1 -0
- package/dist/src/cli/commands/studio.d.ts +2 -0
- package/dist/src/cli/commands/studio.d.ts.map +1 -0
- package/dist/src/cli/commands/studio.js +76 -0
- package/dist/src/cli/commands/studio.js.map +1 -0
- package/dist/src/cli/i18n-dict.d.ts +2 -4
- package/dist/src/cli/i18n-dict.d.ts.map +1 -1
- package/dist/src/cli/i18n-dict.js +461 -675
- package/dist/src/cli/i18n-dict.js.map +1 -1
- package/dist/src/cli/index.js +43 -1516
- package/dist/src/cli/index.js.map +1 -1
- package/dist/src/cli/parse-run-config.d.ts +3 -4
- package/dist/src/cli/parse-run-config.d.ts.map +1 -1
- package/dist/src/cli/parse-run-config.js +5 -5
- package/dist/src/cli/parse-run-config.js.map +1 -1
- package/dist/src/cli/parse-strict.d.ts +0 -13
- package/dist/src/cli/parse-strict.d.ts.map +1 -1
- package/dist/src/cli/parse-strict.js +2 -1
- package/dist/src/cli/parse-strict.js.map +1 -1
- package/dist/src/doctor/health/builtin-dimensions.d.ts +11 -0
- package/dist/src/doctor/health/builtin-dimensions.d.ts.map +1 -0
- package/dist/src/doctor/health/builtin-dimensions.js +94 -0
- package/dist/src/doctor/health/builtin-dimensions.js.map +1 -0
- package/dist/src/doctor/health/composer.d.ts +19 -0
- package/dist/src/doctor/health/composer.d.ts.map +1 -0
- package/dist/src/doctor/health/composer.js +289 -0
- package/dist/src/doctor/health/composer.js.map +1 -0
- package/dist/src/doctor/health/dimension-registry.d.ts +13 -0
- package/dist/src/doctor/health/dimension-registry.d.ts.map +1 -0
- package/dist/src/doctor/health/dimension-registry.js +28 -0
- package/dist/src/doctor/health/dimension-registry.js.map +1 -0
- package/dist/src/doctor/health/dimension-spec.d.ts +46 -0
- package/dist/src/doctor/health/dimension-spec.d.ts.map +1 -0
- package/dist/src/doctor/health/dimension-spec.js +12 -0
- package/dist/src/doctor/health/dimension-spec.js.map +1 -0
- package/dist/src/doctor/health/parser.d.ts +27 -0
- package/dist/src/doctor/health/parser.d.ts.map +1 -0
- package/dist/src/doctor/health/parser.js +190 -0
- package/dist/src/doctor/health/parser.js.map +1 -0
- package/dist/src/doctor/health/prompt-builder.d.ts +22 -0
- package/dist/src/doctor/health/prompt-builder.d.ts.map +1 -0
- package/dist/src/doctor/health/prompt-builder.js +162 -0
- package/dist/src/doctor/health/prompt-builder.js.map +1 -0
- package/dist/src/doctor/health/register.d.ts +13 -0
- package/dist/src/doctor/health/register.d.ts.map +1 -0
- package/dist/src/doctor/health/register.js +20 -0
- package/dist/src/doctor/health/register.js.map +1 -0
- package/dist/src/doctor/html-renderer.d.ts +20 -0
- package/dist/src/doctor/html-renderer.d.ts.map +1 -0
- package/dist/src/doctor/html-renderer.js +366 -0
- package/dist/src/doctor/html-renderer.js.map +1 -0
- package/dist/src/doctor/index.d.ts +1 -1
- package/dist/src/doctor/index.d.ts.map +1 -1
- package/dist/src/doctor/index.js +57 -8
- package/dist/src/doctor/index.js.map +1 -1
- package/dist/src/doctor/preflight.d.ts +2 -2
- package/dist/src/doctor/preflight.js +2 -2
- package/dist/src/doctor/renderer.d.ts +4 -0
- package/dist/src/doctor/renderer.d.ts.map +1 -1
- package/dist/src/doctor/renderer.js +72 -18
- package/dist/src/doctor/renderer.js.map +1 -1
- package/dist/src/doctor/rules.d.ts +6 -5
- package/dist/src/doctor/rules.d.ts.map +1 -1
- package/dist/src/doctor/rules.js +4 -3
- package/dist/src/doctor/rules.js.map +1 -1
- package/dist/src/eval-core/fact-checker.js +1 -1
- package/dist/src/eval-core/fact-checker.js.map +1 -1
- package/dist/src/eval-core/layer-gates.d.ts +1 -1
- package/dist/src/eval-core/layer-gates.js +1 -1
- package/dist/src/eval-core/verdict.d.ts +4 -4
- package/dist/src/eval-core/verdict.d.ts.map +1 -1
- package/dist/src/eval-core/verdict.js +2 -2
- package/dist/src/eval-workflows/batch-evaluation-workflow.d.ts +15 -2
- package/dist/src/eval-workflows/batch-evaluation-workflow.d.ts.map +1 -1
- package/dist/src/eval-workflows/batch-evaluation-workflow.js +13 -2
- package/dist/src/eval-workflows/batch-evaluation-workflow.js.map +1 -1
- package/dist/src/eval-workflows/evaluation-pipeline.d.ts +6 -3
- package/dist/src/eval-workflows/evaluation-pipeline.d.ts.map +1 -1
- package/dist/src/eval-workflows/evaluation-pipeline.js +10 -9
- package/dist/src/eval-workflows/evaluation-pipeline.js.map +1 -1
- package/dist/src/eval-workflows/run-evaluation.d.ts +1 -1
- package/dist/src/eval-workflows/run-evaluation.d.ts.map +1 -1
- package/dist/src/eval-workflows/run-evaluation.js +13 -6
- package/dist/src/eval-workflows/run-evaluation.js.map +1 -1
- package/dist/src/executors/script.d.ts.map +1 -1
- package/dist/src/executors/script.js +16 -3
- package/dist/src/executors/script.js.map +1 -1
- package/dist/src/grading/debias-validate.d.ts +2 -2
- package/dist/src/grading/debias-validate.js +2 -2
- package/dist/src/grading/gold-cli.d.ts +2 -5
- package/dist/src/grading/gold-cli.d.ts.map +1 -1
- package/dist/src/grading/gold-cli.js +4 -8
- package/dist/src/grading/gold-cli.js.map +1 -1
- package/dist/src/grading/judge.d.ts +1 -1
- package/dist/src/renderer/html-renderer.js +1 -1
- package/dist/src/renderer/layout.js +5 -5
- package/dist/src/renderer/layout.js.map +1 -1
- package/dist/src/renderer/skill-health-renderer.d.ts +1 -1
- package/dist/src/renderer/skill-health-renderer.js +1 -1
- package/dist/src/renderer/summary.js +8 -8
- package/dist/src/server/report-server.d.ts.map +1 -1
- package/dist/src/server/report-server.js +8 -7
- package/dist/src/server/report-server.js.map +1 -1
- package/dist/src/types/doctor.d.ts +44 -6
- package/dist/src/types/doctor.d.ts.map +1 -1
- package/dist/src/types/doctor.js +3 -0
- package/dist/src/types/doctor.js.map +1 -1
- package/dist/src/types/eval.d.ts +1 -1
- package/dist/src/types/report.d.ts +2 -2
- package/dist/src/types/report.d.ts.map +1 -1
- package/package.json +1 -1
- package/dist/src/cli/coverage-renderer.d.ts +0 -15
- package/dist/src/cli/coverage-renderer.d.ts.map +0 -1
- package/dist/src/cli/coverage-renderer.js +0 -74
- package/dist/src/cli/coverage-renderer.js.map +0 -1
|
@@ -16,9 +16,7 @@
|
|
|
16
16
|
* 2. **保留原文的白名单 (产品术语 / 命令 / 文件名)**
|
|
17
17
|
* 以下 token 在两种语言里都保留原文, 不翻译:
|
|
18
18
|
* - 产品名: omk, oh-my-knowledge, Claude, npm
|
|
19
|
-
* -
|
|
20
|
-
* diff, ci, gen-samples, debias-validate, saturation, verdict,
|
|
21
|
-
* diagnose, failures
|
|
19
|
+
* - 命令名: init, doctor, eval, observe, evolve, sample, studio, gold
|
|
22
20
|
* - omk 核心业务术语: skill, variant, sample, judge, executor (出现在产品
|
|
23
21
|
* UI 里时首字母可大写如 "Skill 评测", 描述句中保持小写)
|
|
24
22
|
* - 技术参数: --lang, --control, --treatment, --bootstrap, --judge-repeat,
|
|
@@ -48,21 +46,9 @@
|
|
|
48
46
|
* Record 类型自动强制每 key 加新语言版本。
|
|
49
47
|
*/
|
|
50
48
|
export const CLI_DICT = {
|
|
51
|
-
'cli.common.lang_invalid_silent': {
|
|
52
|
-
zh: '无效的语言代码: {value} (仅支持 zh / en, 已使用默认 zh)',
|
|
53
|
-
en: 'Invalid language code: {value} (supported: zh / en, using default zh)',
|
|
54
|
-
},
|
|
55
|
-
'cli.common.help_hint': {
|
|
56
|
-
zh: "运行 'omk --help' 查看用法",
|
|
57
|
-
en: "Run 'omk --help' to see usage",
|
|
58
|
-
},
|
|
59
49
|
'cli.common.unknown_domain': {
|
|
60
|
-
zh: "
|
|
61
|
-
en: "Unknown
|
|
62
|
-
},
|
|
63
|
-
'cli.common.unknown_bench_command': {
|
|
64
|
-
zh: "未知子命令: bench {command} (运行 'omk --help' 查看可用列表)",
|
|
65
|
-
en: "Unknown bench command: {command} (run 'omk --help' to see all commands)",
|
|
50
|
+
zh: "未知命令:{domain}。运行 'omk --help' 查看可用命令。",
|
|
51
|
+
en: "Unknown command: {domain}. Run 'omk --help' to see available commands.",
|
|
66
52
|
},
|
|
67
53
|
'cli.init.scaffolded': {
|
|
68
54
|
zh: '已初始化测评项目: {dir}',
|
|
@@ -73,7 +59,7 @@ export const CLI_DICT = {
|
|
|
73
59
|
en: 'Next steps:',
|
|
74
60
|
},
|
|
75
61
|
'cli.init.next_step_edit_samples': {
|
|
76
|
-
zh: ' 1. 编辑 eval-samples.json
|
|
62
|
+
zh: ' 1. 编辑 eval-samples.json,加入你要测的评测用例',
|
|
77
63
|
en: ' 1. Edit eval-samples.json to add your test cases',
|
|
78
64
|
},
|
|
79
65
|
'cli.init.next_step_edit_skills': {
|
|
@@ -81,8 +67,8 @@ export const CLI_DICT = {
|
|
|
81
67
|
en: ' 2. Edit skills/code-review-v1/SKILL.md and skills/code-review-v2/SKILL.md with your skill versions',
|
|
82
68
|
},
|
|
83
69
|
'cli.init.next_step_run': {
|
|
84
|
-
zh: ' 3. 运行: omk
|
|
85
|
-
en: ' 3. Run: omk
|
|
70
|
+
zh: ' 3. 运行: omk eval --control code-review-v1 --treatment code-review-v2',
|
|
71
|
+
en: ' 3. Run: omk eval --control code-review-v1 --treatment code-review-v2',
|
|
86
72
|
},
|
|
87
73
|
'cli.init.note_codex_executor': {
|
|
88
74
|
zh: '\n注: omk 评测时把 SKILL.md 整文(含 frontmatter)作为 system prompt 注入 — 跨 executor 一致(claude / codex / openai-api / gemini 都走同一条路径,不依赖任何 executor 的 native skill auto-discovery 或 Skill 工具机制)。frontmatter 在 prompt 头部对 model 行为无显著影响。\n模板带 Claude Code 兼容的 frontmatter(name + description)是为了让同一份 directory-skill 也能 deploy 到 Claude Code:把整个目录复制到 ~/.claude/skills/code-review-v1/(整目录,不是单个 SKILL.md),Claude SDK 才能识别。这是 omk 评测之外的 bonus,一份文件双向 dogfood。',
|
|
@@ -92,6 +78,22 @@ export const CLI_DICT = {
|
|
|
92
78
|
zh: '\n💡 新版本可用: {old} → {new}, 运行 npm update {pkg} -g 升级\n\n',
|
|
93
79
|
en: '\n💡 New version available: {old} → {new}, run npm update {pkg} -g to upgrade\n\n',
|
|
94
80
|
},
|
|
81
|
+
'cli.run.power_warning_tiny_n': {
|
|
82
|
+
zh: '⚠ N={n} < 5:仅适合探索,任何结论都不可靠,CI 会很宽。需要决策时建议 ≥20 条评测用例。',
|
|
83
|
+
en: '⚠ N={n} < 5 (exploration-only): any conclusion is unreliable, CI will be uselessly wide. Decisions need ≥20 cases.',
|
|
84
|
+
},
|
|
85
|
+
'cli.run.power_warning_small_n': {
|
|
86
|
+
zh: '⚠ N={n} < 20:只能识别很大的效果(Cohen\'s d > 0.8),中等效果(d ≈ 0.5)很难检出。要做可靠决策建议 ≥20 条评测用例。',
|
|
87
|
+
en: '⚠ N={n} < 20 (large-effect-only, Cohen\'s d > 0.8): medium effects (d ≈ 0.5) hard to detect. For confident decisions consider ≥20 cases.',
|
|
88
|
+
},
|
|
89
|
+
'cli.run.power_warning_repeat_one': {
|
|
90
|
+
zh: '⚠ --repeat=1:单轮评测无法测稳定性(CV 会标记为未测量)。用 --repeat 3+ 检测同一 variant 内部方差。',
|
|
91
|
+
en: '⚠ --repeat=1: single-run cannot measure stability (CV will be marked "not measured"). Use --repeat 3+ to detect within-variant variance.',
|
|
92
|
+
},
|
|
93
|
+
'cli.run.dry_run_no_scores': {
|
|
94
|
+
zh: 'eval dry-run:仅预览任务,不检查分数',
|
|
95
|
+
en: 'Eval dry-run: no scores to check',
|
|
96
|
+
},
|
|
95
97
|
'cli.progress.preflight_starting': {
|
|
96
98
|
zh: '⏳ 正在预检模型连通性...\n',
|
|
97
99
|
en: '⏳ Preflight: checking model connectivity...\n',
|
|
@@ -168,6 +170,14 @@ export const CLI_DICT = {
|
|
|
168
170
|
zh: '\n✅ 批量评测完成\n',
|
|
169
171
|
en: '\n✅ Batch evaluation done\n',
|
|
170
172
|
},
|
|
173
|
+
'cli.run.batch_verdict_header': {
|
|
174
|
+
zh: '批量评测结论:{status}({passed}/{total} 通过)',
|
|
175
|
+
en: 'Batch verdict: {status} ({passed}/{total} passed)',
|
|
176
|
+
},
|
|
177
|
+
'cli.run.batch_child_report_missing': {
|
|
178
|
+
zh: '⚠ 子报告缺失:{id},将按不可 ship 处理。\n',
|
|
179
|
+
en: '⚠ Child report missing: {id}; treating it as not shippable.\n',
|
|
180
|
+
},
|
|
171
181
|
'cli.run.eval_complete': {
|
|
172
182
|
zh: '\n✅ 评测完成\n',
|
|
173
183
|
en: '\n✅ Evaluation done\n',
|
|
@@ -180,6 +190,10 @@ export const CLI_DICT = {
|
|
|
180
190
|
zh: '📄 报告已保存到: {path}\n',
|
|
181
191
|
en: '📄 Report saved to: {path}\n',
|
|
182
192
|
},
|
|
193
|
+
'cli.run.report_only_gate_skipped': {
|
|
194
|
+
zh: 'ℹ 已启用 report-only 模式:保留 verdict 输出,但本次不使用 verdict 改写 exit code。\n',
|
|
195
|
+
en: 'ℹ Report-only mode enabled: verdict is still printed, but it will not affect the exit code.\n',
|
|
196
|
+
},
|
|
183
197
|
'cli.run.report_server_running': {
|
|
184
198
|
zh: '\n📊 报告服务已启动: {url}\n',
|
|
185
199
|
en: '\n📊 Report server running at {url}\n',
|
|
@@ -197,8 +211,8 @@ export const CLI_DICT = {
|
|
|
197
211
|
en: '\n💡 Non-interactive environment, skipping report server\n',
|
|
198
212
|
},
|
|
199
213
|
'cli.run.no_serve_view_hint': {
|
|
200
|
-
zh: '
|
|
201
|
-
en: ' View report: omk
|
|
214
|
+
zh: ' 查看报告:omk studio --reports-dir {dir}(报告 ID:{id})\n',
|
|
215
|
+
en: ' View report: omk studio --reports-dir {dir} (report id: {id})\n',
|
|
202
216
|
},
|
|
203
217
|
'cli.run.gold_load_failed': {
|
|
204
218
|
zh: '\n⚠ gold dataset 加载失败 ({dir}):\n',
|
|
@@ -216,9 +230,9 @@ export const CLI_DICT = {
|
|
|
216
230
|
zh: '❌ 错误: {message}',
|
|
217
231
|
en: '❌ Error: {message}',
|
|
218
232
|
},
|
|
219
|
-
'cli.
|
|
220
|
-
zh:
|
|
221
|
-
en:
|
|
233
|
+
'cli.observe.view_hint': {
|
|
234
|
+
zh: '分析 JSON 已写入 output-dir;后续可用 omk observe 持续生成日报。',
|
|
235
|
+
en: 'Analysis JSON written to output-dir; use omk observe to keep producing health reports.',
|
|
222
236
|
},
|
|
223
237
|
'cli.common.skill_dir_not_found': {
|
|
224
238
|
zh: '未找到 skill 目录: {path}',
|
|
@@ -237,23 +251,31 @@ export const CLI_DICT = {
|
|
|
237
251
|
en: 'No judge configured. Pass --judge-models <executor:model> or ensure the report has meta.judgeModels.',
|
|
238
252
|
},
|
|
239
253
|
'cli.common.judge_models_single_only': {
|
|
240
|
-
zh: '
|
|
241
|
-
en: '
|
|
242
|
-
},
|
|
243
|
-
'cli.common.usage_gold_validate': {
|
|
244
|
-
zh: '用法: omk bench gold validate <dir>',
|
|
245
|
-
en: 'Usage: omk bench gold validate <dir>',
|
|
254
|
+
zh: '{cmd} 仅支持单评委。--judge-models 只能传一个 executor:model entry。',
|
|
255
|
+
en: '{cmd} only supports a single judge. --judge-models accepts exactly one executor:model entry.',
|
|
246
256
|
},
|
|
247
257
|
'cli.common.warn_load_samples_failed': {
|
|
248
258
|
zh: '⚠ 加载 samples 文件失败 ({path}): {message}\n',
|
|
249
259
|
en: '⚠ Failed to load samples file ({path}): {message}\n',
|
|
250
260
|
},
|
|
261
|
+
'cli.studio.started': {
|
|
262
|
+
zh: 'studio 已启动:{url}',
|
|
263
|
+
en: 'Studio running at {url}',
|
|
264
|
+
},
|
|
265
|
+
'cli.studio.stop_hint': {
|
|
266
|
+
zh: '按 Ctrl+C 停止服务',
|
|
267
|
+
en: 'Press Ctrl+C to stop',
|
|
268
|
+
},
|
|
269
|
+
'cli.studio.open_failed': {
|
|
270
|
+
zh: '⚠ 无法自动打开浏览器({command}):{message}\n',
|
|
271
|
+
en: '⚠ Failed to open browser automatically ({command}): {message}\n',
|
|
272
|
+
},
|
|
251
273
|
'cli.gen.skill_skipped_existing': {
|
|
252
274
|
zh: '⏭️ {name}: eval-samples 已存在, 跳过\n',
|
|
253
275
|
en: '⏭️ {name}: eval-samples already exists, skipping\n',
|
|
254
276
|
},
|
|
255
277
|
'cli.gen.skill_generating': {
|
|
256
|
-
zh: '🔄 {name}: 正在生成 {count}
|
|
278
|
+
zh: '🔄 {name}: 正在生成 {count} 条评测用例...\n',
|
|
257
279
|
en: '🔄 {name}: generating {count} test cases...\n',
|
|
258
280
|
},
|
|
259
281
|
'cli.gen.skill_done': {
|
|
@@ -269,19 +291,19 @@ export const CLI_DICT = {
|
|
|
269
291
|
en: 'No eval-samples need generating (all skills already have paired files)',
|
|
270
292
|
},
|
|
271
293
|
'cli.gen.batch_summary': {
|
|
272
|
-
zh: '\n共生成 {n} 份 eval-samples, 请审查后运行: omk
|
|
273
|
-
en: '\nGenerated {n} eval-samples files. Review them, then run: omk
|
|
294
|
+
zh: '\n共生成 {n} 份 eval-samples, 请审查后运行: omk eval --batch',
|
|
295
|
+
en: '\nGenerated {n} eval-samples files. Review them, then run: omk eval --batch',
|
|
274
296
|
},
|
|
275
297
|
'cli.gen.specify_skill_path': {
|
|
276
|
-
zh: '请指定 skill 文件路径, 例如: omk
|
|
277
|
-
en: 'Please specify a skill file path, e.g.: omk
|
|
298
|
+
zh: '请指定 skill 文件路径, 例如: omk sample skills/my-skill.md',
|
|
299
|
+
en: 'Please specify a skill file path, e.g.: omk sample skills/my-skill.md',
|
|
278
300
|
},
|
|
279
301
|
'cli.gen.samples_already_exists': {
|
|
280
302
|
zh: 'eval-samples.json 已存在。如需覆盖请先删除该文件。',
|
|
281
303
|
en: 'eval-samples.json already exists. Delete it first if you want to overwrite.',
|
|
282
304
|
},
|
|
283
305
|
'cli.gen.single_generating': {
|
|
284
|
-
zh: '🔄 正在生成 {count}
|
|
306
|
+
zh: '🔄 正在生成 {count} 条评测用例...\n',
|
|
285
307
|
en: '🔄 Generating {count} test cases...\n',
|
|
286
308
|
},
|
|
287
309
|
'cli.gen.single_done': {
|
|
@@ -289,20 +311,20 @@ export const CLI_DICT = {
|
|
|
289
311
|
en: '✅ Generated {n} samples → {path}{cost}\n',
|
|
290
312
|
},
|
|
291
313
|
'cli.gen.review_hint': {
|
|
292
|
-
zh: '\n
|
|
293
|
-
en: '\nReview the generated test cases, then run: omk
|
|
314
|
+
zh: '\n请审查生成的评测用例后运行: omk eval',
|
|
315
|
+
en: '\nReview the generated test cases, then run: omk eval',
|
|
294
316
|
},
|
|
295
317
|
'cli.gen.failed': {
|
|
296
318
|
zh: '生成失败: {message}',
|
|
297
319
|
en: 'Generation failed: {message}',
|
|
298
320
|
},
|
|
299
321
|
'cli.evolve.specify_skill_path': {
|
|
300
|
-
zh: '请指定 skill 文件路径, 例如: omk
|
|
301
|
-
en: 'Please specify a skill file path, e.g.: omk
|
|
322
|
+
zh: '请指定 skill 文件路径, 例如: omk evolve skills/my-skill.md',
|
|
323
|
+
en: 'Please specify a skill file path, e.g.: omk evolve skills/my-skill.md',
|
|
302
324
|
},
|
|
303
325
|
'cli.evolve.section_header': {
|
|
304
|
-
zh: '\n===
|
|
305
|
-
en: '\n===
|
|
326
|
+
zh: '\n=== Improve skill: {path} ===\n',
|
|
327
|
+
en: '\n=== Improve skill: {path} ===\n',
|
|
306
328
|
},
|
|
307
329
|
'cli.evolve.round_baseline': {
|
|
308
330
|
zh: '第 0 轮 (基线): score={score} ({cost})\n',
|
|
@@ -329,630 +351,319 @@ export const CLI_DICT = {
|
|
|
329
351
|
en: 'All versions saved at: {dir}/\n',
|
|
330
352
|
},
|
|
331
353
|
'cli.evolve.report_link': {
|
|
332
|
-
zh: '📊
|
|
333
|
-
en: '📊
|
|
334
|
-
},
|
|
335
|
-
'cli.gold.created_files': {
|
|
336
|
-
zh: '已在 {dir} 创建 {n} 个文件:',
|
|
337
|
-
en: 'Created {n} files in {dir}:',
|
|
338
|
-
},
|
|
339
|
-
'cli.gold.next_step_edit_annotations': {
|
|
340
|
-
zh: '\n下一步: 编辑 annotations.yaml 加入真实标注 → 跑 omk bench gold validate',
|
|
341
|
-
en: '\nNext step: edit annotations.yaml with real annotations → run omk bench gold validate',
|
|
342
|
-
},
|
|
343
|
-
'cli.gold.validate_ok': {
|
|
344
|
-
zh: '✓ gold dataset OK — 共 {n} 条标注',
|
|
345
|
-
en: '✓ gold dataset OK — {n} annotations',
|
|
346
|
-
},
|
|
347
|
-
'cli.debias.warn_cost_doubles': {
|
|
348
|
-
zh: '\n⚠ debias-validate 会重判所有 (sample × variant), judge 成本大约翻倍。\n',
|
|
349
|
-
en: '\n⚠ debias-validate will re-judge all (sample × variant) pairs; judge cost will roughly double.\n',
|
|
350
|
-
},
|
|
351
|
-
'cli.saturation.no_data': {
|
|
352
|
-
zh: '该 report 没有 saturation 数据 (需要 --repeat ≥ 2 才会记录)。',
|
|
353
|
-
en: 'This report has no saturation data (requires --repeat ≥ 2 to record).',
|
|
354
|
-
},
|
|
355
|
-
'cli.saturation.verdict_header': {
|
|
356
|
-
zh: '\n Saturation verdict (复述持久化结果)\n',
|
|
357
|
-
en: '\n Saturation verdict (replaying persisted result)\n',
|
|
358
|
-
},
|
|
359
|
-
'cli.saturation.variant_no_trace': {
|
|
360
|
-
zh: ' {variant}: 没有 trace 数据',
|
|
361
|
-
en: ' {variant}: no trace data',
|
|
362
|
-
},
|
|
363
|
-
'cli.saturation.variant_label': {
|
|
364
|
-
zh: ' {variant}:',
|
|
365
|
-
en: ' {variant}:',
|
|
366
|
-
},
|
|
367
|
-
'cli.saturation.checkpoints': {
|
|
368
|
-
zh: ' 检查点: {n} (N={list})',
|
|
369
|
-
en: ' checkpoints: {n} (N={list})',
|
|
354
|
+
zh: '📊 查看报告:omk studio(报告 ID:{id})\n',
|
|
355
|
+
en: '📊 View report: omk studio (report id: {id})\n',
|
|
370
356
|
},
|
|
371
|
-
'cli.
|
|
372
|
-
zh:
|
|
373
|
-
|
|
374
|
-
|
|
375
|
-
|
|
376
|
-
|
|
377
|
-
|
|
378
|
-
|
|
379
|
-
|
|
380
|
-
|
|
381
|
-
|
|
382
|
-
|
|
383
|
-
|
|
384
|
-
|
|
385
|
-
|
|
386
|
-
|
|
387
|
-
|
|
388
|
-
|
|
389
|
-
|
|
357
|
+
'cli.help.product_main': {
|
|
358
|
+
zh: `
|
|
359
|
+
oh-my-knowledge — 知识载体工作台
|
|
360
|
+
|
|
361
|
+
用法:
|
|
362
|
+
omk init [dir] 初始化一个 skill 评测项目
|
|
363
|
+
omk doctor [path] LLM 健康度审计(7 内置维度 + 可扩展);--static-only 切离线静态模式
|
|
364
|
+
omk eval [options] 离线评测:比较版本,输出 verdict + report
|
|
365
|
+
omk observe <sessions-dir> 线上观测:真实 session、gap、失败率、inbox
|
|
366
|
+
omk evolve <skill> 多轮自动迭代改进 skill
|
|
367
|
+
omk sample <skill> 生成或补齐 eval-samples 评测用例(或 --batch 批量模式)
|
|
368
|
+
omk studio 打开本地工作台浏览报告
|
|
369
|
+
|
|
370
|
+
主路径:
|
|
371
|
+
omk doctor
|
|
372
|
+
omk eval --control code-review-v1 --treatment code-review-v2
|
|
373
|
+
omk observe ~/.claude/projects/<project>
|
|
374
|
+
omk evolve skills/code-review-v2/SKILL.md
|
|
375
|
+
omk studio
|
|
376
|
+
|
|
377
|
+
通用选项:
|
|
378
|
+
--lang <zh|en> CLI 输出语言(默认:zh,也可设 OMK_LANG)
|
|
379
|
+
|
|
380
|
+
运行 'omk <command> --help' 查看单个命令的参数。
|
|
381
|
+
`,
|
|
382
|
+
en: `
|
|
383
|
+
oh-my-knowledge — Knowledge Artifact Workbench
|
|
384
|
+
|
|
385
|
+
Usage:
|
|
386
|
+
omk init [dir] Scaffold a skill evaluation project
|
|
387
|
+
omk doctor [path] LLM health audit (7 builtin dimensions, extensible); --static-only for offline static checks
|
|
388
|
+
omk eval [options] Offline evaluation: compare versions, emit verdict + report
|
|
389
|
+
omk observe <sessions-dir> Production observation: sessions, gaps, failure rate, inbox
|
|
390
|
+
omk evolve <skill> Auto-iterate a skill through multi-round eval loops
|
|
391
|
+
omk sample <skill> Generate or fill eval-samples test cases (or --batch for all skills)
|
|
392
|
+
omk studio Open the local workbench to browse reports
|
|
393
|
+
|
|
394
|
+
Main workflow:
|
|
395
|
+
omk doctor
|
|
396
|
+
omk eval --control code-review-v1 --treatment code-review-v2
|
|
397
|
+
omk observe ~/.claude/projects/<project>
|
|
398
|
+
omk evolve skills/code-review-v2/SKILL.md
|
|
399
|
+
omk studio
|
|
400
|
+
|
|
401
|
+
Common options:
|
|
402
|
+
--lang <zh|en> CLI output language (default: zh, or set OMK_LANG)
|
|
403
|
+
|
|
404
|
+
Run 'omk <command> --help' for command-specific options.
|
|
405
|
+
`,
|
|
390
406
|
},
|
|
391
|
-
'cli.help.
|
|
407
|
+
'cli.help.init_usage': {
|
|
392
408
|
zh: `
|
|
393
|
-
|
|
409
|
+
omk init — 初始化 skill 评测项目
|
|
394
410
|
|
|
395
|
-
|
|
396
|
-
omk
|
|
397
|
-
omk bench report [options] 启动报告 server
|
|
398
|
-
omk bench gate [options] 跑评测 + 应用 gate, exit code 0/1 (CI/CD 用)
|
|
399
|
-
omk bench init [dir] 初始化一个评测项目
|
|
400
|
-
omk bench gen-samples [skill] 从 skill 内容生成 eval-samples
|
|
401
|
-
omk bench diff <id1> <id2> 对比两份评测报告
|
|
402
|
-
omk bench evolve <skill> 通过迭代评测自我改进 skill
|
|
411
|
+
用法:
|
|
412
|
+
omk init [dir]
|
|
403
413
|
|
|
404
|
-
|
|
405
|
-
|
|
414
|
+
生成内容:
|
|
415
|
+
eval-samples.json 示例评测用例
|
|
416
|
+
skills/code-review-v1/SKILL.md 基线 skill
|
|
417
|
+
skills/code-review-v2/SKILL.md 实验组 skill
|
|
406
418
|
|
|
407
|
-
|
|
419
|
+
下一步:
|
|
420
|
+
1. 编辑 eval-samples.json,替换成你的真实评测用例
|
|
421
|
+
2. 编辑两个 SKILL.md,填入要对比的 skill 版本
|
|
422
|
+
3. 运行 omk eval --control code-review-v1 --treatment code-review-v2
|
|
423
|
+
`,
|
|
424
|
+
en: `
|
|
425
|
+
omk init — scaffold a skill evaluation project
|
|
408
426
|
|
|
409
|
-
|
|
410
|
-
|
|
411
|
-
--control <expr> 对照组 variant 表达式 (实验角色 = control)
|
|
412
|
-
--treatment <v1,v2> 实验组 variant 表达式 (逗号分隔; 角色 = treatment)
|
|
413
|
-
每个 variant 表达式解析为一个 artifact 加上可选运行时上下文:
|
|
414
|
-
"baseline" — 裸模型, 不注入 artifact
|
|
415
|
-
"git:name" — 来自最后一次 commit 的 artifact
|
|
416
|
-
"git:ref:name" — 来自指定 commit 的 artifact
|
|
417
|
-
带 "/" 的路径 — 直接来自文件 (例如 ./v1.md)
|
|
418
|
-
"name@/cwd" — 附加运行时上下文 / cwd
|
|
419
|
-
--control 和 --treatment 至少要给一个。
|
|
420
|
-
--config <path> YAML/JSON 配置文件 (evaluation-as-code)。
|
|
421
|
-
在一个文件里声明 samples + variants + model + executor。
|
|
422
|
-
CLI flag 会覆盖配置文件中的同名字段。
|
|
423
|
-
配置中的相对路径相对于配置文件所在目录解析。
|
|
424
|
-
--model <name> 任务执行模型 (默认: sonnet)
|
|
425
|
-
--output-dir <path> 报告输出目录 (默认: ~/.oh-my-knowledge/reports/)
|
|
426
|
-
--no-judge 跳过 LLM 评委
|
|
427
|
-
--no-cache 禁用结果缓存
|
|
428
|
-
--dry-run 预览任务但不执行
|
|
429
|
-
--blind 双盲 A/B 模式: 报告里隐藏 variant 名称
|
|
430
|
-
--concurrency <n> 并发任务数 (默认: 1)
|
|
431
|
-
--timeout <seconds> 单任务执行超时 (秒, 默认: 120)
|
|
432
|
-
--repeat <n> 跑 N 轮做方差分析 (默认: 1)
|
|
433
|
-
--judge-repeat <n> 每个 (sample × dimension) 调 LLM 评委 N 次评估
|
|
434
|
-
自洽性 (默认: 1)。多轮间高 stddev = 评委在该评分维度
|
|
435
|
-
上不稳定, 分数有噪声。
|
|
436
|
-
--judge-models <list> 评委配置, 逗号分隔的 executor:model, 如
|
|
437
|
-
claude:haiku 或 claude:opus,openai:gpt-4o。
|
|
438
|
-
1 条 = 单评委 (默认 claude:haiku); ≥ 2 条 = ensemble,
|
|
439
|
-
每个评委对所有 (sample × dimension) 打分, 报告
|
|
440
|
-
含每评委分布 + Pearson / MAD 评委间一致性。能反驳
|
|
441
|
-
"Claude 评委评 Claude 同模态偏置" 的质疑。可与
|
|
442
|
-
--judge-repeat 组合。成本 ~ N_judges × N_repeat × N_samples。
|
|
443
|
-
--bootstrap 计算 bootstrap 置信区间 (无分布假设, 对 LLM 序数评分
|
|
444
|
-
比 t 区间更靠谱)。给出每个 variant 均值 CI + treatment
|
|
445
|
-
vs control 差值的 pairwise CI (CI 不跨 0 即显著)。
|
|
446
|
-
同时报告 t 区间和 bootstrap, 旧工具仍可用。
|
|
447
|
-
--bootstrap-samples <n> bootstrap 重采样次数 (默认 1000)。N>10000 触发
|
|
448
|
-
stderr 警告提示耗时。
|
|
449
|
-
--retry <n> 失败任务最多重试 N 次, 指数退避 (默认: 0)
|
|
450
|
-
--resume <report-id> 从历史报告恢复, 跳过已完成任务
|
|
451
|
-
--executor <name> 执行器: claude / claude-sdk / codex / openai / gemini /
|
|
452
|
-
anthropic-api / openai-api, 或任意 shell 命令 (例如 "python my_provider.py")
|
|
453
|
-
--batch 批量评测:每个 skill 独立 vs baseline
|
|
454
|
-
需要每个 skill 有配对的 {name}.eval-samples.json
|
|
455
|
-
--skip-connectivity 跳过 LLM 模型连通性检测 (--resume 时自动跳过)
|
|
456
|
-
--mcp-config <path> 通过 MCP server 抓 URL 用的 MCP 配置文件
|
|
457
|
-
(默认: 当前目录下的 .mcp.json)
|
|
458
|
-
--no-serve 评测后不自动启动报告 server
|
|
459
|
-
--verbose 打印每个用例的详细进度 (执行结果 / 评分阶段)
|
|
460
|
-
--layered-stats 默认在 HTML 报告里展开三层 (fact/behavior/judge) 独立
|
|
461
|
-
显著性细分。不加这个 flag 时, 细分会折叠在每个对比下
|
|
462
|
-
的 click-to-expand summary 里。
|
|
463
|
-
--strict-baseline (默认开启) 对 baseline-kind variant 强制隔离 skill 自动
|
|
464
|
-
发现 + Skill 工具调用, 切断 ~/.claude/skills/ 污染路径,
|
|
465
|
-
保证 skill 评测的 construct validity。eval.yaml 显式
|
|
466
|
-
allowedSkills 优先。
|
|
467
|
-
--no-strict-baseline 显式关闭 strict-baseline (baseline 走默认 SDK skill
|
|
468
|
-
全发现)。少数场景下可能想要这个 (例如评测 skill 文档
|
|
469
|
-
对默认全发现行为的增量影响)。开启时 pre-flight 会
|
|
470
|
-
stderr 提醒, 因为 verdict / Δ 易受污染。
|
|
427
|
+
Usage:
|
|
428
|
+
omk init [dir]
|
|
471
429
|
|
|
472
|
-
|
|
473
|
-
|
|
474
|
-
|
|
475
|
-
|
|
476
|
-
|
|
477
|
-
|
|
478
|
-
|
|
479
|
-
|
|
480
|
-
|
|
481
|
-
|
|
430
|
+
Generated files:
|
|
431
|
+
eval-samples.json Example test cases
|
|
432
|
+
skills/code-review-v1/SKILL.md Baseline skill
|
|
433
|
+
skills/code-review-v2/SKILL.md Treatment skill
|
|
434
|
+
|
|
435
|
+
Next steps:
|
|
436
|
+
1. Edit eval-samples.json with your real test cases
|
|
437
|
+
2. Edit both SKILL.md files with the skill versions to compare
|
|
438
|
+
3. Run omk eval --control code-review-v1 --treatment code-review-v2
|
|
439
|
+
`,
|
|
440
|
+
},
|
|
441
|
+
'cli.help.eval': {
|
|
442
|
+
zh: `
|
|
443
|
+
omk eval — 离线评测 skill 版本,并给出 ship/no-ship verdict
|
|
444
|
+
|
|
445
|
+
用法:
|
|
446
|
+
omk eval --control <variant> --treatment <variant> [options]
|
|
447
|
+
omk eval gold <init|validate|compare> ...
|
|
448
|
+
|
|
449
|
+
常用选项:
|
|
450
|
+
--samples <path> 用例文件(默认:eval-samples.json)
|
|
451
|
+
--skill-dir <path> skill 目录(默认:skills)
|
|
452
|
+
--control <expr> 对照组 variant
|
|
453
|
+
--treatment <v1,v2> 实验组 variant,逗号分隔
|
|
454
|
+
--config <path> eval.yaml / JSON 配置
|
|
455
|
+
--executor <name> 执行器:claude / claude-sdk / codex / openai / gemini / custom
|
|
456
|
+
--model <name> 任务执行模型(默认:sonnet)
|
|
457
|
+
--judge-models <list> 评委配置,例如 claude:haiku 或 claude:opus,openai:gpt-4o
|
|
458
|
+
--dry-run 预览任务,不调用模型
|
|
459
|
+
--batch 批量评测:每个 skill 独立 vs baseline
|
|
460
|
+
--bootstrap 显式开启 bootstrap CI;omk eval 默认会自动开启
|
|
461
|
+
--bootstrap-samples <n> bootstrap 重采样次数(默认:1000)
|
|
462
|
+
--threshold <number> 三层 gate 阈值(默认:3.5)
|
|
463
|
+
--trivial-diff <number> 实际可忽略 diff(默认:0.1)
|
|
464
|
+
--report-only / --no-gate 生成报告并打印 verdict,但始终 exit 0
|
|
465
|
+
--no-serve 评测后不自动启动报告 server
|
|
466
|
+
|
|
467
|
+
示例:
|
|
468
|
+
omk eval --control baseline --treatment my-skill # 单 skill 必要性测试(baseline 是保留 variant 名,代表「不注入 skill 的裸基线」)
|
|
469
|
+
omk eval --control code-review-v1 --treatment code-review-v2 # 多版本 A/B
|
|
470
|
+
omk eval --config eval.yaml
|
|
471
|
+
omk eval gold compare v1-vs-v2-20260505-1200 --gold-dir gold-dataset
|
|
472
|
+
`,
|
|
473
|
+
en: `
|
|
474
|
+
omk eval — run offline skill evaluation and emit a ship/no-ship verdict
|
|
482
475
|
|
|
483
|
-
|
|
484
|
-
|
|
485
|
-
|
|
476
|
+
Usage:
|
|
477
|
+
omk eval --control <variant> --treatment <variant> [options]
|
|
478
|
+
omk eval gold <init|validate|compare> ...
|
|
486
479
|
|
|
487
|
-
|
|
488
|
-
--
|
|
489
|
-
--
|
|
490
|
-
--
|
|
491
|
-
--
|
|
480
|
+
Common options:
|
|
481
|
+
--samples <path> Sample file (default: eval-samples.json)
|
|
482
|
+
--skill-dir <path> Skill directory (default: skills)
|
|
483
|
+
--control <expr> Control variant
|
|
484
|
+
--treatment <v1,v2> Treatment variants, comma-separated
|
|
485
|
+
--config <path> eval.yaml / JSON config
|
|
486
|
+
--executor <name> Executor: claude / claude-sdk / codex / openai / gemini / custom
|
|
487
|
+
--model <name> Task execution model (default: sonnet)
|
|
488
|
+
--judge-models <list> Judge config, e.g. claude:haiku or claude:opus,openai:gpt-4o
|
|
489
|
+
--dry-run Preview tasks without model calls
|
|
490
|
+
--batch Batch evaluation: each skill independently against baseline
|
|
491
|
+
--bootstrap Enable bootstrap CI explicitly; omk eval turns it on by default
|
|
492
|
+
--bootstrap-samples <n> Bootstrap resamples (default: 1000)
|
|
493
|
+
--threshold <number> Three-layer gate threshold (default: 3.5)
|
|
494
|
+
--trivial-diff <number> Practically negligible diff (default: 0.1)
|
|
495
|
+
--report-only / --no-gate Produce the report and print verdict, but always exit 0
|
|
496
|
+
--no-serve Do not auto-start report server after evaluation
|
|
492
497
|
|
|
493
|
-
|
|
494
|
-
--
|
|
495
|
-
--
|
|
496
|
-
|
|
497
|
-
|
|
498
|
+
Examples:
|
|
499
|
+
omk eval --control baseline --treatment my-skill # Single-skill necessity test (baseline is a reserved variant — "no skill injected")
|
|
500
|
+
omk eval --control code-review-v1 --treatment code-review-v2 # Multi-variant A/B
|
|
501
|
+
omk eval --config eval.yaml
|
|
502
|
+
omk eval gold compare v1-vs-v2-20260505-1200 --gold-dir gold-dataset
|
|
503
|
+
`,
|
|
504
|
+
},
|
|
505
|
+
'cli.help.eval_gold': {
|
|
506
|
+
zh: `
|
|
507
|
+
omk eval gold — 管理 human-gold 标注集
|
|
498
508
|
|
|
499
|
-
|
|
500
|
-
|
|
501
|
-
|
|
502
|
-
|
|
503
|
-
--last <duration> 时间窗口, 例如 "7d" / "30d" (默认: 全部)
|
|
504
|
-
--from <iso> 窗口起点 (ISO8601), 优先级高于 --last
|
|
505
|
-
--to <iso> 窗口终点 (ISO8601), 优先级高于 --last
|
|
506
|
-
--skills <n1,n2,...> 白名单要分析的 skill (默认: 全部)
|
|
507
|
-
--output-dir <path> 输出目录 (默认: ~/.oh-my-knowledge/analyses/)
|
|
509
|
+
用法:
|
|
510
|
+
omk eval gold init [--out <dir>] [--annotator <name>]
|
|
511
|
+
omk eval gold validate <dir>
|
|
512
|
+
omk eval gold compare <reportId> --gold-dir <dir>
|
|
508
513
|
|
|
509
|
-
|
|
510
|
-
--
|
|
511
|
-
--
|
|
512
|
-
--samples <
|
|
513
|
-
|
|
514
|
-
|
|
515
|
-
|
|
516
|
-
--concurrency <n> 并发评测任务数 (默认: 1)
|
|
517
|
-
--timeout <seconds> 单任务执行超时 (秒, 默认: 120)
|
|
518
|
-
--executor <name> 执行器 (默认: claude)
|
|
514
|
+
选项:
|
|
515
|
+
--reports-dir <path> 报告目录(compare 使用,默认:~/.oh-my-knowledge/reports)
|
|
516
|
+
--variant <name> 指定 report 中要对比的 variant
|
|
517
|
+
--bootstrap-samples <n> bootstrap 重采样次数(compare 使用)
|
|
518
|
+
`,
|
|
519
|
+
en: `
|
|
520
|
+
omk eval gold — manage human-gold annotation datasets
|
|
519
521
|
|
|
520
|
-
|
|
521
|
-
|
|
522
|
+
Usage:
|
|
523
|
+
omk eval gold init [--out <dir>] [--annotator <name>]
|
|
524
|
+
omk eval gold validate <dir>
|
|
525
|
+
omk eval gold compare <reportId> --gold-dir <dir>
|
|
522
526
|
|
|
523
|
-
|
|
524
|
-
|
|
525
|
-
|
|
526
|
-
|
|
527
|
-
|
|
528
|
-
|
|
529
|
-
|
|
530
|
-
|
|
531
|
-
|
|
532
|
-
|
|
533
|
-
|
|
534
|
-
omk
|
|
535
|
-
|
|
536
|
-
|
|
527
|
+
Options:
|
|
528
|
+
--reports-dir <path> Reports directory for compare (default: ~/.oh-my-knowledge/reports)
|
|
529
|
+
--variant <name> Variant in the report to compare
|
|
530
|
+
--bootstrap-samples <n> Bootstrap resamples for compare
|
|
531
|
+
`,
|
|
532
|
+
},
|
|
533
|
+
'cli.help.observe': {
|
|
534
|
+
zh: `
|
|
535
|
+
omk observe — 分析真实 session trace,生成 skill 健康度日报
|
|
536
|
+
|
|
537
|
+
用法:
|
|
538
|
+
omk observe <sessions-dir> [options]
|
|
539
|
+
|
|
540
|
+
选项:
|
|
541
|
+
--kb <path> 知识库根路径(默认:从 trace cwd 推断)
|
|
542
|
+
--last <duration> 时间窗口,例如 7d / 24h / 30m
|
|
543
|
+
--from <iso> 窗口起点,优先级高于 --last
|
|
544
|
+
--to <iso> 窗口终点,优先级高于 --last
|
|
545
|
+
--skills <n1,n2,...> 只分析指定 skill
|
|
546
|
+
--output-dir <path> 输出目录(默认:~/.oh-my-knowledge/analyses)
|
|
537
547
|
`,
|
|
538
548
|
en: `
|
|
539
|
-
|
|
549
|
+
omk observe — analyze production session traces and produce skill health reports
|
|
540
550
|
|
|
541
551
|
Usage:
|
|
542
|
-
omk
|
|
543
|
-
omk bench report [options] Start the report server
|
|
544
|
-
omk bench gate [options] Run evaluation + apply gate, exit 0/1 (for CI/CD)
|
|
545
|
-
omk bench init [dir] Scaffold a new eval project
|
|
546
|
-
omk bench gen-samples [skill] Generate eval-samples from skill content
|
|
547
|
-
omk bench diff <id1> <id2> Compare two evaluation reports
|
|
548
|
-
omk bench evolve <skill> Self-improve a skill through iterative evaluation
|
|
552
|
+
omk observe <sessions-dir> [options]
|
|
549
553
|
|
|
550
|
-
|
|
551
|
-
|
|
554
|
+
Options:
|
|
555
|
+
--kb <path> Knowledge base root (default: infer from trace cwd)
|
|
556
|
+
--last <duration> Time window, e.g. 7d / 24h / 30m
|
|
557
|
+
--from <iso> Window start, overrides --last
|
|
558
|
+
--to <iso> Window end, overrides --last
|
|
559
|
+
--skills <n1,n2,...> Only analyze selected skills
|
|
560
|
+
--output-dir <path> Output directory (default: ~/.oh-my-knowledge/analyses)
|
|
561
|
+
`,
|
|
562
|
+
},
|
|
563
|
+
'cli.help.evolve': {
|
|
564
|
+
zh: `
|
|
565
|
+
omk evolve — 多轮自动迭代改进 skill
|
|
566
|
+
|
|
567
|
+
用法:
|
|
568
|
+
omk evolve <skill-path> [options]
|
|
569
|
+
|
|
570
|
+
选项:
|
|
571
|
+
--rounds <n> 迭代轮数(默认:5)
|
|
572
|
+
--target <score> 目标分数
|
|
573
|
+
--model <name> 任务执行模型,每轮跑 eval samples 的被测模型(默认:sonnet)
|
|
574
|
+
--improve-model <name> skill 改写模型,每轮根据反馈改写 skill 的模型(默认:sonnet)
|
|
575
|
+
--judge-models <executor:model> 单评委配置(默认:claude:haiku)
|
|
576
|
+
|
|
577
|
+
示例:
|
|
578
|
+
omk evolve skills/code-review/SKILL.md
|
|
579
|
+
omk evolve skills/code-review/SKILL.md --rounds 10 --target 4.5
|
|
580
|
+
omk evolve skills/code-review/SKILL.md --model sonnet --improve-model opus
|
|
581
|
+
`,
|
|
582
|
+
en: `
|
|
583
|
+
omk evolve — auto-iterate a skill through multi-round evaluation loops
|
|
552
584
|
|
|
553
|
-
|
|
585
|
+
Usage:
|
|
586
|
+
omk evolve <skill-path> [options]
|
|
554
587
|
|
|
555
|
-
|
|
556
|
-
--
|
|
557
|
-
--
|
|
558
|
-
--
|
|
559
|
-
|
|
560
|
-
|
|
561
|
-
"git:name" — artifact from last commit
|
|
562
|
-
"git:ref:name" — artifact from specific commit
|
|
563
|
-
path with "/" — artifact from file directly (e.g. ./v1.md)
|
|
564
|
-
"name@/cwd" — attach runtime context / cwd
|
|
565
|
-
At least one of --control / --treatment must be provided.
|
|
566
|
-
--config <path> YAML/JSON config file (evaluation-as-code).
|
|
567
|
-
Declares samples + variants + model + executor in one file.
|
|
568
|
-
CLI flags override config fields when both are provided.
|
|
569
|
-
Relative paths inside the config are resolved against its directory.
|
|
570
|
-
--model <name> Task execution model (default: sonnet)
|
|
571
|
-
--output-dir <path> Report output directory (default: ~/.oh-my-knowledge/reports/)
|
|
572
|
-
--no-judge Skip LLM judging
|
|
573
|
-
--no-cache Disable result caching
|
|
574
|
-
--dry-run Preview tasks without executing
|
|
575
|
-
--blind Blind A/B mode: hide variant names in report
|
|
576
|
-
--concurrency <n> Number of parallel tasks (default: 1)
|
|
577
|
-
--timeout <seconds> Executor timeout per task in seconds (default: 120)
|
|
578
|
-
--repeat <n> Run evaluation N times for variance analysis (default: 1)
|
|
579
|
-
--judge-repeat <n> Call LLM judge N times per (sample × dimension) for self-
|
|
580
|
-
consistency (default: 1). High stddev across runs = the
|
|
581
|
-
judge is unstable on this rubric and the score is noisy.
|
|
582
|
-
--judge-models <list> Judge configuration. Comma-separated executor:model pairs,
|
|
583
|
-
e.g. claude:haiku or claude:opus,openai:gpt-4o.
|
|
584
|
-
1 entry = single judge (default claude:haiku); ≥ 2 entries
|
|
585
|
-
= ensemble — every judge scores each (sample × dimension);
|
|
586
|
-
the report includes per-judge breakdown + Pearson/MAD
|
|
587
|
-
inter-judge agreement, which refutes "Claude judges Claude
|
|
588
|
-
same-modality bias" critique. Combines with --judge-repeat.
|
|
589
|
-
Cost ~ N_judges × N_repeat × N_samples.
|
|
590
|
-
--bootstrap Compute bootstrap confidence intervals (distribution-free,
|
|
591
|
-
preferred over t-interval for ordinal LLM scores). Adds
|
|
592
|
-
per-variant CI on the mean + pairwise CI on treatment-vs-
|
|
593
|
-
control difference (significant=0 outside CI). Reports both
|
|
594
|
-
t-interval and bootstrap so old tooling still works.
|
|
595
|
-
--bootstrap-samples <n> Number of bootstrap resamples (default 1000). N>10000
|
|
596
|
-
triggers a stderr warning about runtime cost.
|
|
597
|
-
--retry <n> Retry failed tasks up to N times with exponential backoff (default: 0)
|
|
598
|
-
--resume <report-id> Resume from a previous report, skipping completed tasks
|
|
599
|
-
--executor <name> Executor: claude, claude-sdk, codex, openai, gemini,
|
|
600
|
-
anthropic-api, openai-api, or any shell command (e.g. "python my_provider.py")
|
|
601
|
-
--batch Batch evaluation: each skill independently against baseline
|
|
602
|
-
Requires {name}.eval-samples.json paired with each skill
|
|
603
|
-
--skip-connectivity Skip LLM model connectivity check (auto-skipped when --resume)
|
|
604
|
-
--mcp-config <path> MCP config file for URL fetching via MCP servers
|
|
605
|
-
(default: .mcp.json in current directory)
|
|
606
|
-
--no-serve Skip auto-starting report server after evaluation
|
|
607
|
-
--verbose Print detailed progress for each sample (exec result, grading phases)
|
|
608
|
-
--layered-stats Expand the three-layer (fact/behavior/judge) independent
|
|
609
|
-
significance breakdown in the HTML report by default.
|
|
610
|
-
Without this flag, the breakdown is collapsed behind a
|
|
611
|
-
click-to-expand summary under each comparison.
|
|
612
|
-
--strict-baseline (default ON) Isolate skill auto-discovery + Skill tool
|
|
613
|
-
use for baseline-kind variants. Cuts the ~/.claude/skills/
|
|
614
|
-
contamination path so skill evaluations have valid
|
|
615
|
-
construct validity. Explicit eval.yaml allowedSkills
|
|
616
|
-
takes precedence.
|
|
617
|
-
--no-strict-baseline Explicitly turn strict-baseline OFF (baseline sees all
|
|
618
|
-
auto-discovered skills). Use only in narrow scenarios
|
|
619
|
-
(e.g. measuring how much a skill doc adds on top of
|
|
620
|
-
full default discovery). Pre-flight emits a stderr
|
|
621
|
-
warning when this flag is set, because
|
|
622
|
-
verdict / Δ are vulnerable to skill contamination.
|
|
588
|
+
Options:
|
|
589
|
+
--rounds <n> Iteration rounds (default: 5)
|
|
590
|
+
--target <score> Target score
|
|
591
|
+
--model <name> Task executor model — runs eval samples each round (default: sonnet)
|
|
592
|
+
--improve-model <name> Skill rewriter model — rewrites the skill each round (default: sonnet)
|
|
593
|
+
--judge-models <executor:model> Single judge config (default: claude:haiku)
|
|
623
594
|
|
|
624
|
-
|
|
625
|
-
|
|
626
|
-
|
|
627
|
-
|
|
628
|
-
|
|
629
|
-
|
|
630
|
-
|
|
631
|
-
|
|
632
|
-
|
|
633
|
-
--trivial-diff <num> Smallest diff to treat as practically meaningful
|
|
634
|
-
(default 0.1). Bootstrap diff CI may be statistically
|
|
635
|
-
significant but with |Δ| < this value, treated as
|
|
636
|
-
CAUTIOUS rather than PROGRESS.
|
|
595
|
+
Examples:
|
|
596
|
+
omk evolve skills/code-review/SKILL.md
|
|
597
|
+
omk evolve skills/code-review/SKILL.md --rounds 10 --target 4.5
|
|
598
|
+
omk evolve skills/code-review/SKILL.md --model sonnet --improve-model opus
|
|
599
|
+
`,
|
|
600
|
+
},
|
|
601
|
+
'cli.help.sample': {
|
|
602
|
+
zh: `
|
|
603
|
+
omk sample — 生成或补齐 eval-samples 评测用例
|
|
637
604
|
|
|
638
|
-
|
|
639
|
-
|
|
640
|
-
|
|
605
|
+
用法:
|
|
606
|
+
omk sample <skill-path> [options]
|
|
607
|
+
omk sample --batch [--skill-dir <dir>] [options]
|
|
641
608
|
|
|
642
|
-
|
|
643
|
-
--
|
|
644
|
-
--
|
|
645
|
-
--
|
|
646
|
-
--
|
|
609
|
+
选项:
|
|
610
|
+
--count <n> 生成用例数量(默认:5)
|
|
611
|
+
--model <name> 生成模型(默认:sonnet)
|
|
612
|
+
--batch 为 skill 目录下缺少 eval-samples 的 skill 批量生成
|
|
613
|
+
--skill-dir <path> skill 目录(batch 使用,默认:skills)
|
|
614
|
+
`,
|
|
615
|
+
en: `
|
|
616
|
+
omk sample — generate or fill eval-samples test cases
|
|
647
617
|
|
|
648
|
-
|
|
649
|
-
|
|
650
|
-
|
|
651
|
-
--model <name> Model for generation (default: sonnet)
|
|
652
|
-
--skill-dir <path> Skill directory (default: skills), used with --batch
|
|
618
|
+
Usage:
|
|
619
|
+
omk sample <skill-path> [options]
|
|
620
|
+
omk sample --batch [--skill-dir <dir>] [options]
|
|
653
621
|
|
|
654
|
-
Options
|
|
655
|
-
<
|
|
656
|
-
--
|
|
657
|
-
--
|
|
658
|
-
--
|
|
659
|
-
|
|
660
|
-
|
|
661
|
-
|
|
622
|
+
Options:
|
|
623
|
+
--count <n> Number of test cases to generate (default: 5)
|
|
624
|
+
--model <name> Generation model (default: sonnet)
|
|
625
|
+
--batch Generate for skills that are missing eval-samples
|
|
626
|
+
--skill-dir <path> Skill directory for batch mode (default: skills)
|
|
627
|
+
`,
|
|
628
|
+
},
|
|
629
|
+
'cli.help.studio': {
|
|
630
|
+
zh: `
|
|
631
|
+
omk studio — 打开本地知识工作台
|
|
632
|
+
|
|
633
|
+
用法:
|
|
634
|
+
omk studio [options]
|
|
635
|
+
|
|
636
|
+
选项:
|
|
637
|
+
--port <n> 本地服务端口(默认:7799)
|
|
638
|
+
--reports-dir <path> 报告目录(默认:~/.oh-my-knowledge/reports)
|
|
639
|
+
--analyses-dir <path> 观测分析目录
|
|
640
|
+
--no-open 只启动服务,不自动打开浏览器
|
|
641
|
+
--dev 开发模式:文件变化时自动重启
|
|
642
|
+
|
|
643
|
+
示例:
|
|
644
|
+
omk studio
|
|
645
|
+
omk studio --port 7798
|
|
646
|
+
omk studio --no-open
|
|
647
|
+
`,
|
|
648
|
+
en: `
|
|
649
|
+
omk studio — open the local knowledge workbench
|
|
662
650
|
|
|
663
|
-
|
|
664
|
-
|
|
665
|
-
--target <score> Stop early when score reaches this threshold
|
|
666
|
-
--samples <path> Sample file (default: eval-samples.json)
|
|
667
|
-
--model <name> Task execution model (default: sonnet)
|
|
668
|
-
--judge-models <executor:model> Judge config (default: claude:haiku; evolve is single-judge only)
|
|
669
|
-
--improve-model <name> Model for generating improvements (default: sonnet)
|
|
670
|
-
--concurrency <n> Parallel eval tasks (default: 1)
|
|
671
|
-
--timeout <seconds> Executor timeout per task in seconds (default: 120)
|
|
672
|
-
--executor <name> Executor to use (default: claude)
|
|
651
|
+
Usage:
|
|
652
|
+
omk studio [options]
|
|
673
653
|
|
|
674
|
-
|
|
675
|
-
--
|
|
654
|
+
Options:
|
|
655
|
+
--port <n> Local server port (default: 7799)
|
|
656
|
+
--reports-dir <path> Reports directory (default: ~/.oh-my-knowledge/reports)
|
|
657
|
+
--analyses-dir <path> Observation analyses directory
|
|
658
|
+
--no-open Start the server without opening a browser
|
|
659
|
+
--dev Dev mode: restart on file changes
|
|
676
660
|
|
|
677
661
|
Examples:
|
|
678
|
-
omk
|
|
679
|
-
omk
|
|
680
|
-
omk
|
|
681
|
-
omk bench run --control ./old-skill.md --treatment ./new-skill.md
|
|
682
|
-
omk bench run --control baseline --treatment v1,v2,v3
|
|
683
|
-
omk bench run --config eval.yaml
|
|
684
|
-
omk bench run --config eval.yaml --model sonnet-4.6 # CLI overrides config
|
|
685
|
-
omk bench run --batch
|
|
686
|
-
omk bench run --dry-run
|
|
687
|
-
omk bench report --port 8080
|
|
688
|
-
omk bench report --export v1-vs-v2-20260326-1832
|
|
689
|
-
omk bench init my-eval
|
|
690
|
-
omk bench gen-samples skills/my-skill.md
|
|
662
|
+
omk studio
|
|
663
|
+
omk studio --port 7798
|
|
664
|
+
omk studio --no-open
|
|
691
665
|
`,
|
|
692
666
|
},
|
|
693
|
-
'cli.help.diff_usage': {
|
|
694
|
-
zh: [
|
|
695
|
-
'用法:',
|
|
696
|
-
' omk bench diff <reportId> 单 report 内 sample 级 diff',
|
|
697
|
-
' omk bench diff <reportId1> <reportId2> 跨 report variant 级 diff',
|
|
698
|
-
'',
|
|
699
|
-
'选项:',
|
|
700
|
-
' --regressions-only 只列 treatment < control 的用例',
|
|
701
|
-
' --threshold <num> 回退判定阈值 (默认 0, 即任何负 Δ 都算回退)',
|
|
702
|
-
' --variant <name> within-report 模式下指定要钻取的 variant (默认: variants[1])',
|
|
703
|
-
' --top <n> 只列差距最大的前 N 个用例',
|
|
704
|
-
].join('\n'),
|
|
705
|
-
en: [
|
|
706
|
-
'Usage:',
|
|
707
|
-
' omk bench diff <reportId> within-report per-sample diff',
|
|
708
|
-
' omk bench diff <reportId1> <reportId2> cross-report variant-level diff',
|
|
709
|
-
'',
|
|
710
|
-
'Options:',
|
|
711
|
-
' --regressions-only show only samples where treatment < control',
|
|
712
|
-
' --threshold <num> regression threshold (default 0, any negative Δ counts)',
|
|
713
|
-
' --variant <name> within-report mode: which variant to drill (default: variants[1])',
|
|
714
|
-
' --top <n> only show top N samples by absolute diff',
|
|
715
|
-
].join('\n'),
|
|
716
|
-
},
|
|
717
|
-
'cli.help.analyze_usage': {
|
|
718
|
-
zh: '用法: omk analyze <dir> [--kb <path>] [--last 7d] [--from ISO] [--to ISO] [--skills name1,name2]',
|
|
719
|
-
en: 'Usage: omk analyze <dir> [--kb <path>] [--last 7d] [--from ISO] [--to ISO] [--skills name1,name2]',
|
|
720
|
-
},
|
|
721
|
-
'cli.help.gold': {
|
|
722
|
-
zh: [
|
|
723
|
-
'',
|
|
724
|
-
'用法: omk bench gold <subcommand>',
|
|
725
|
-
'',
|
|
726
|
-
'子命令:',
|
|
727
|
-
' init [--out <dir>] [--annotator <id>] 生成空白 gold dataset 模板',
|
|
728
|
-
' validate <dir> 校验数据集结构',
|
|
729
|
-
' compare <reportId> --gold-dir <dir> 与已有 report 计算 α / κ / Pearson',
|
|
730
|
-
' [--variant <name>] [--reports-dir <d>]',
|
|
731
|
-
' [--bootstrap-samples N] [--seed N]',
|
|
732
|
-
'',
|
|
733
|
-
].join('\n'),
|
|
734
|
-
en: [
|
|
735
|
-
'',
|
|
736
|
-
'Usage: omk bench gold <subcommand>',
|
|
737
|
-
'',
|
|
738
|
-
'Subcommands:',
|
|
739
|
-
' init [--out <dir>] [--annotator <id>] create a blank gold dataset template',
|
|
740
|
-
' validate <dir> validate dataset structure',
|
|
741
|
-
' compare <reportId> --gold-dir <dir> compute α / κ / Pearson against an existing report',
|
|
742
|
-
' [--variant <name>] [--reports-dir <d>]',
|
|
743
|
-
' [--bootstrap-samples N] [--seed N]',
|
|
744
|
-
'',
|
|
745
|
-
].join('\n'),
|
|
746
|
-
},
|
|
747
|
-
'cli.help.debias_validate': {
|
|
748
|
-
zh: [
|
|
749
|
-
'',
|
|
750
|
-
'用法: omk bench debias-validate <kind> <reportId> [options]',
|
|
751
|
-
'',
|
|
752
|
-
'类别:',
|
|
753
|
-
' length 用相反的长度去偏设置重新评判, 并对分数差出 bootstrap CI。',
|
|
754
|
-
' judge 成本约为原评判的两倍。',
|
|
755
|
-
'',
|
|
756
|
-
'选项:',
|
|
757
|
-
' --reports-dir <dir> 报告存储目录 (默认: ~/.oh-my-knowledge/reports)',
|
|
758
|
-
' --samples <path> 覆盖用例文件 (默认: 从 report.meta.request 读)',
|
|
759
|
-
' --variant <name> 校验哪个 variant (默认: 第一个)',
|
|
760
|
-
' --judge-models <executor:model> 评委 (默认: 沿用 report.meta.judgeModels[0]; debias-validate 仅支持单评委)',
|
|
761
|
-
' --bootstrap-samples N bootstrap 迭代次数 (默认 1000)',
|
|
762
|
-
' --seed N 固定 CI 随机种子',
|
|
763
|
-
'',
|
|
764
|
-
].join('\n'),
|
|
765
|
-
en: [
|
|
766
|
-
'',
|
|
767
|
-
'Usage: omk bench debias-validate <kind> <reportId> [options]',
|
|
768
|
-
'',
|
|
769
|
-
'Kinds:',
|
|
770
|
-
' length re-judge with the opposite length-debias setting and bootstrap CI',
|
|
771
|
-
' on the score diff. Cost ~doubles vs the original judge pass.',
|
|
772
|
-
'',
|
|
773
|
-
'Options:',
|
|
774
|
-
' --reports-dir <dir> report store dir (default: ~/.oh-my-knowledge/reports)',
|
|
775
|
-
' --samples <path> override samples file (default: from report.meta.request)',
|
|
776
|
-
' --variant <name> which variant to validate (default: first)',
|
|
777
|
-
' --judge-models <executor:model> Judge (default: from report.meta.judgeModels[0]; debias-validate is single-judge only)',
|
|
778
|
-
' --bootstrap-samples N bootstrap iterations (default 1000)',
|
|
779
|
-
' --seed N deterministic CI seed',
|
|
780
|
-
'',
|
|
781
|
-
].join('\n'),
|
|
782
|
-
},
|
|
783
|
-
'cli.help.saturation': {
|
|
784
|
-
zh: [
|
|
785
|
-
'',
|
|
786
|
-
'用法: omk bench saturation <reportId> [options]',
|
|
787
|
-
'',
|
|
788
|
-
'回答 "我跑够用例了吗?"。复述已有 report 中持久化的饱和判定。',
|
|
789
|
-
'',
|
|
790
|
-
'注: 本命令读取 run 时跑出的 verdict (运行时已用 method=bootstrap-ci-width',
|
|
791
|
-
'默认阈值 + 3 窗口持续条件)。如要换 method/threshold 重新计算, 需要重跑',
|
|
792
|
-
'`omk bench run --repeat ≥ 5` (运行时持久化的 trace 不含原始分数, 无法',
|
|
793
|
-
'在事后用其他参数复算)。',
|
|
794
|
-
'',
|
|
795
|
-
'选项:',
|
|
796
|
-
' --reports-dir <dir> 报告存储目录 (默认: ~/.oh-my-knowledge/reports)',
|
|
797
|
-
' --variant <name> 只看一个 variant (默认: 全部)',
|
|
798
|
-
'',
|
|
799
|
-
].join('\n'),
|
|
800
|
-
en: [
|
|
801
|
-
'',
|
|
802
|
-
'Usage: omk bench saturation <reportId> [options]',
|
|
803
|
-
'',
|
|
804
|
-
'Answers "do I have enough samples?". Replays the saturation verdict',
|
|
805
|
-
'persisted in an existing report.',
|
|
806
|
-
'',
|
|
807
|
-
'Note: this command reads the verdict computed at run time (which used',
|
|
808
|
-
'method=bootstrap-ci-width with default threshold + 3-window sustained',
|
|
809
|
-
'condition). To re-compute with a different method/threshold, re-run',
|
|
810
|
-
'`omk bench run --repeat ≥ 5` (the persisted trace does not include raw',
|
|
811
|
-
'scores, so post-hoc parameter sweeps are not possible here).',
|
|
812
|
-
'',
|
|
813
|
-
'Options:',
|
|
814
|
-
' --reports-dir <dir> report store dir (default: ~/.oh-my-knowledge/reports)',
|
|
815
|
-
' --variant <name> only show one variant (default: all)',
|
|
816
|
-
'',
|
|
817
|
-
].join('\n'),
|
|
818
|
-
},
|
|
819
|
-
'cli.help.verdict': {
|
|
820
|
-
zh: [
|
|
821
|
-
'',
|
|
822
|
-
'用法: omk bench verdict <reportId> [options]',
|
|
823
|
-
'',
|
|
824
|
-
'聚合 bootstrap CI / 三层 ci-gate / saturation / human α, 给出一行结论。',
|
|
825
|
-
'',
|
|
826
|
-
'Verdict 等级:',
|
|
827
|
-
' PROGRESS 显著改进 + 三层全过',
|
|
828
|
-
' CAUTIOUS 改进真实但有警告 (gate 破 / 幅度太小 / 控制组本身崩)',
|
|
829
|
-
' REGRESS 显著回退 — 不要 ship',
|
|
830
|
-
' NOISE CI 跨 0, 无法判定',
|
|
831
|
-
' UNDERPOWERED 用例不足, 需要扩 N 重测',
|
|
832
|
-
' SOLO 单 variant 报告, 没有对比对象',
|
|
833
|
-
'',
|
|
834
|
-
'选项:',
|
|
835
|
-
' --reports-dir <dir> 报告存储目录 (默认: ~/.oh-my-knowledge/reports)',
|
|
836
|
-
' --threshold <num> 三层 gate 阈值 (默认 3.5, 与 omk bench gate 对齐)',
|
|
837
|
-
' --trivial-diff <num> "幅度太小" 阈值 (默认 0.1)',
|
|
838
|
-
' --verbose 展开每个 pair 的详情',
|
|
839
|
-
'',
|
|
840
|
-
].join('\n'),
|
|
841
|
-
en: [
|
|
842
|
-
'',
|
|
843
|
-
'Usage: omk bench verdict <reportId> [options]',
|
|
844
|
-
'',
|
|
845
|
-
'Aggregates bootstrap CI / 3-layer ci-gate / saturation / human α into a one-line verdict.',
|
|
846
|
-
'',
|
|
847
|
-
'Verdict levels:',
|
|
848
|
-
' PROGRESS significant improvement + all 3 layers pass',
|
|
849
|
-
' CAUTIOUS real improvement but with warnings (gate fails / diff too small / control collapsed)',
|
|
850
|
-
' REGRESS significant regression — do not ship',
|
|
851
|
-
' NOISE CI crosses 0, no verdict',
|
|
852
|
-
' UNDERPOWERED not enough samples, expand N and re-run',
|
|
853
|
-
' SOLO single-variant report, nothing to compare against',
|
|
854
|
-
'',
|
|
855
|
-
'Options:',
|
|
856
|
-
' --reports-dir <dir> report store dir (default: ~/.oh-my-knowledge/reports)',
|
|
857
|
-
' --threshold <num> 3-layer gate threshold (default 3.5, matches omk bench gate)',
|
|
858
|
-
' --trivial-diff <num> "diff too small" threshold (default 0.1)',
|
|
859
|
-
' --verbose expand per-pair details',
|
|
860
|
-
'',
|
|
861
|
-
].join('\n'),
|
|
862
|
-
},
|
|
863
|
-
'cli.help.diagnose': {
|
|
864
|
-
zh: [
|
|
865
|
-
'',
|
|
866
|
-
'用法: omk bench diagnose <reportId> [options]',
|
|
867
|
-
'',
|
|
868
|
-
'诊断用例集本身的质量问题: 区分度低 / 重复 / 歧义 / 成本异常 / 全 fail。',
|
|
869
|
-
'回答 "测评结论是否被坏用例污染" — 与 omk bench verdict 互补。',
|
|
870
|
-
'',
|
|
871
|
-
'选项:',
|
|
872
|
-
' --reports-dir <dir> 报告存储目录',
|
|
873
|
-
' --samples <path> 用例文件路径 (用于 near-duplicate 检测; 默认从 report.meta.request 读)',
|
|
874
|
-
' --top <n> 每类只显示前 N 个 (默认 10, 0=全部)',
|
|
875
|
-
' --duplicate-rouge <num> near-duplicate ROUGE-1 阈值 (默认 0.7)',
|
|
876
|
-
' --ambiguous-stddev <num> 歧义阈值, judge stddev (默认 1.0, 需要 --judge-repeat ≥ 2 数据)',
|
|
877
|
-
' --cost-k <num> 成本异常倍数 vs 中位数 (默认 3)',
|
|
878
|
-
' --latency-k <num> 耗时异常倍数 vs 中位数 (默认 3)',
|
|
879
|
-
' --flat <num> flat_scores 分差阈值 (默认 0.5)',
|
|
880
|
-
'',
|
|
881
|
-
].join('\n'),
|
|
882
|
-
en: [
|
|
883
|
-
'',
|
|
884
|
-
'Usage: omk bench diagnose <reportId> [options]',
|
|
885
|
-
'',
|
|
886
|
-
'Diagnose quality issues in the sample set itself: low discrimination /',
|
|
887
|
-
'duplicates / ambiguity / cost anomalies / all-fail. Answers "is the verdict',
|
|
888
|
-
'tainted by bad samples?" — complements omk bench verdict.',
|
|
889
|
-
'',
|
|
890
|
-
'Options:',
|
|
891
|
-
' --reports-dir <dir> report store dir',
|
|
892
|
-
' --samples <path> sample file path (for near-duplicate detection; defaults to report.meta.request)',
|
|
893
|
-
' --top <n> top N per category (default 10, 0=all)',
|
|
894
|
-
' --duplicate-rouge <num> near-duplicate ROUGE-1 threshold (default 0.7)',
|
|
895
|
-
' --ambiguous-stddev <num> ambiguity threshold, judge stddev (default 1.0, requires --judge-repeat ≥ 2)',
|
|
896
|
-
' --cost-k <num> cost-outlier multiplier vs median (default 3)',
|
|
897
|
-
' --latency-k <num> latency-outlier multiplier vs median (default 3)',
|
|
898
|
-
' --flat <num> flat_scores spread threshold (default 0.5)',
|
|
899
|
-
'',
|
|
900
|
-
].join('\n'),
|
|
901
|
-
},
|
|
902
|
-
'cli.help.failures': {
|
|
903
|
-
zh: [
|
|
904
|
-
'',
|
|
905
|
-
'用法: omk bench failures <reportId> [options]',
|
|
906
|
-
'',
|
|
907
|
-
'把已有 report 的失败用例喂给一次 LLM 调用, 自动聚类并给出修复建议。',
|
|
908
|
-
'失败定义: compositeScore < threshold 或 ok=false。',
|
|
909
|
-
'',
|
|
910
|
-
'选项:',
|
|
911
|
-
' --reports-dir <dir> 报告存储目录',
|
|
912
|
-
' --judge-models <executor:model> 评委 (默认: 沿用 report.meta.judgeModels[0]; failures 仅支持单评委)',
|
|
913
|
-
' --max-clusters <n> 最多聚成几类 (默认 5)',
|
|
914
|
-
' --threshold <num> compositeScore < threshold 算失败 (默认 3)',
|
|
915
|
-
' --max-feed <n> 最多喂给 LLM 多少条 (默认 50, 超出取最差)',
|
|
916
|
-
'',
|
|
917
|
-
].join('\n'),
|
|
918
|
-
en: [
|
|
919
|
-
'',
|
|
920
|
-
'Usage: omk bench failures <reportId> [options]',
|
|
921
|
-
'',
|
|
922
|
-
'Feed failing samples from an existing report to a single LLM call, auto-cluster',
|
|
923
|
-
'them, and produce per-cluster fix suggestions.',
|
|
924
|
-
'Failure definition: compositeScore < threshold or ok=false.',
|
|
925
|
-
'',
|
|
926
|
-
'Options:',
|
|
927
|
-
' --reports-dir <dir> report store dir',
|
|
928
|
-
' --judge-models <executor:model> Judge (default: from report.meta.judgeModels[0]; failures is single-judge only)',
|
|
929
|
-
' --max-clusters <n> max number of clusters (default 5)',
|
|
930
|
-
' --threshold <num> compositeScore < threshold counts as failure (default 3)',
|
|
931
|
-
' --max-feed <n> max samples to feed the LLM (default 50, takes the worst)',
|
|
932
|
-
'',
|
|
933
|
-
].join('\n'),
|
|
934
|
-
},
|
|
935
|
-
// sample design coverage block strings
|
|
936
|
-
'cli.diagnose.coverage_header': {
|
|
937
|
-
zh: '用例设计覆盖度 (Sample design coverage):',
|
|
938
|
-
en: 'Sample design coverage:',
|
|
939
|
-
},
|
|
940
|
-
'cli.diagnose.coverage_unspecified': {
|
|
941
|
-
zh: '(未声明)',
|
|
942
|
-
en: '(unspecified)',
|
|
943
|
-
},
|
|
944
|
-
'cli.diagnose.coverage_chars': {
|
|
945
|
-
zh: '字符',
|
|
946
|
-
en: 'chars',
|
|
947
|
-
},
|
|
948
|
-
'cli.diagnose.coverage_hint_empty': {
|
|
949
|
-
zh: 'ℹ 该用例集未声明任何 capability / difficulty / construct / provenance 元数据。详见 docs/sample-design-spec.md',
|
|
950
|
-
en: 'ℹ No samples in this set declare capability / difficulty / construct / provenance metadata. See docs/sample-design-spec.md',
|
|
951
|
-
},
|
|
952
|
-
'cli.diagnose.coverage_declared': {
|
|
953
|
-
zh: '声明',
|
|
954
|
-
en: 'declared',
|
|
955
|
-
},
|
|
956
667
|
// ============ omk doctor 健康检查 ============
|
|
957
668
|
'cli.doctor.rule.skill_readable': {
|
|
958
669
|
zh: 'skill 文件可读',
|
|
@@ -970,6 +681,67 @@ Examples:
|
|
|
970
681
|
zh: '用例 ↔ skill 输入约定',
|
|
971
682
|
en: 'samples ↔ skill contract',
|
|
972
683
|
},
|
|
684
|
+
'cli.doctor.rule.skill_health_check': {
|
|
685
|
+
zh: '健康度体检',
|
|
686
|
+
en: 'Health check',
|
|
687
|
+
},
|
|
688
|
+
// ============ skill_health composer (CLI default; --static-only disables it) ============
|
|
689
|
+
'cli.doctor.health.skipped': {
|
|
690
|
+
zh: '健康度体检已跳过(runHealthCheck=false)',
|
|
691
|
+
en: 'health check skipped (runHealthCheck=false)',
|
|
692
|
+
},
|
|
693
|
+
'cli.doctor.health.no_dimensions': {
|
|
694
|
+
zh: '没有注册任何健康度维度,跳过',
|
|
695
|
+
en: 'no health dimensions registered, skipped',
|
|
696
|
+
},
|
|
697
|
+
'cli.doctor.health.fail.executor': {
|
|
698
|
+
zh: 'LLM 调用失败: {error}',
|
|
699
|
+
en: 'LLM call failed: {error}',
|
|
700
|
+
},
|
|
701
|
+
'cli.doctor.health.fail.parse': {
|
|
702
|
+
zh: 'LLM 输出解析失败: {error}',
|
|
703
|
+
en: 'failed to parse LLM output: {error}',
|
|
704
|
+
},
|
|
705
|
+
'cli.doctor.health.fail.empty_output': {
|
|
706
|
+
zh: 'LLM 返回了空输出',
|
|
707
|
+
en: 'LLM returned empty output',
|
|
708
|
+
},
|
|
709
|
+
'cli.doctor.health.hint.executor': {
|
|
710
|
+
zh: '检查 executor 配置(--executor / --model)与网络连通,或调大 --timeout',
|
|
711
|
+
en: 'Verify executor config (--executor / --model) and connectivity, or raise --timeout',
|
|
712
|
+
},
|
|
713
|
+
'cli.doctor.health.hint.parse': {
|
|
714
|
+
zh: 'LLM 没返回合法 JSON;原文存在 detail.rawOutput 截断片段,可重跑或换 model',
|
|
715
|
+
en: 'LLM did not return valid JSON; raw snippet stored in detail.rawOutput. Re-run or switch model',
|
|
716
|
+
},
|
|
717
|
+
'cli.doctor.health.dim.message': {
|
|
718
|
+
zh: '{level}: 错误 {err}/警告 {warn}/建议 {sug}',
|
|
719
|
+
en: '{level}: error {err}/warn {warn}/suggest {sug}',
|
|
720
|
+
},
|
|
721
|
+
'cli.doctor.health.dim.missing': {
|
|
722
|
+
zh: 'LLM 未输出此维度({dim}),已置不适用',
|
|
723
|
+
en: 'LLM omitted dimension ({dim}); treated as N/A',
|
|
724
|
+
},
|
|
725
|
+
'cli.doctor.health.summary.label': {
|
|
726
|
+
zh: '健康度总览',
|
|
727
|
+
en: 'Health summary',
|
|
728
|
+
},
|
|
729
|
+
'cli.doctor.health.summary.message': {
|
|
730
|
+
zh: '{overall} | 维度: 健康 {h}/亚健康 {sh}/不健康 {bad}/不适用 {na} | finding: 错误 {err}/警告 {warn}/建议 {sug}',
|
|
731
|
+
en: '{overall} | dims: healthy {h}/sub {sh}/unhealthy {bad}/n-a {na} | findings: err {err}/warn {warn}/sug {sug}',
|
|
732
|
+
},
|
|
733
|
+
'cli.doctor.health.summary.no_top': {
|
|
734
|
+
zh: '完整详情见 --json 输出或 --html 报告',
|
|
735
|
+
en: 'Full detail in --json output or --html report',
|
|
736
|
+
},
|
|
737
|
+
// 7 内置维度 labelKey (id-based)
|
|
738
|
+
'cli.doctor.health.dim.trigger-boundary': { zh: '触发与边界', en: 'Trigger & boundary' },
|
|
739
|
+
'cli.doctor.health.dim.doc-clarity': { zh: '文档清晰', en: 'Documentation clarity' },
|
|
740
|
+
'cli.doctor.health.dim.instr-precision': { zh: '指令精确性', en: 'Instruction precision' },
|
|
741
|
+
'cli.doctor.health.dim.dependency': { zh: '依赖检查', en: 'Dependency check' },
|
|
742
|
+
'cli.doctor.health.dim.tool-conventions': { zh: '工具规范', en: 'Tool conventions' },
|
|
743
|
+
'cli.doctor.health.dim.security': { zh: '安全与合规', en: 'Security & compliance' },
|
|
744
|
+
'cli.doctor.health.dim.examples': { zh: '示例完备', en: 'Example completeness' },
|
|
973
745
|
// pass
|
|
974
746
|
'cli.doctor.skill_readable.pass': {
|
|
975
747
|
zh: 'skill 内容长度 {length} 字符',
|
|
@@ -1084,10 +856,10 @@ Examples:
|
|
|
1084
856
|
// ============ omk doctor CLI level ============
|
|
1085
857
|
'cli.help.doctor_usage': {
|
|
1086
858
|
zh: `
|
|
1087
|
-
oh-my-knowledge — omk doctor
|
|
859
|
+
oh-my-knowledge — omk doctor 健康度体检 (LLM-judge)
|
|
1088
860
|
|
|
1089
861
|
用法:
|
|
1090
|
-
omk doctor [path] 在 path
|
|
862
|
+
omk doctor [path] 在 path 上跑深度健康度体检
|
|
1091
863
|
omk doctor 在当前目录(或 ./skills)批量跑
|
|
1092
864
|
|
|
1093
865
|
参数:
|
|
@@ -1095,66 +867,80 @@ oh-my-knowledge — omk doctor 健康检查
|
|
|
1095
867
|
|
|
1096
868
|
选项:
|
|
1097
869
|
--json 把 DoctorReport 打到 stdout(CI 消费用)
|
|
1098
|
-
--gate 静默模式:
|
|
1099
|
-
--executor <name> executor
|
|
1100
|
-
--model <name>
|
|
1101
|
-
--
|
|
870
|
+
--gate 静默模式: fatal 问题 exit 1; warnings_only 仍 exit 0, 仅 stderr 出摘要
|
|
871
|
+
--executor <name> LLM executor (默认 claude, 可换 anthropic-api/codex 等)
|
|
872
|
+
--model <name> 模型 (默认 sonnet)
|
|
873
|
+
--samples <path> 显式指定评测用例文件
|
|
874
|
+
--timeout <seconds> 单次 LLM 会话超时 (默认 600)
|
|
875
|
+
--html <path> 产出可视化 HTML 报告到 <path> (可与 --json 同时用)
|
|
876
|
+
--static-only 离线模式: 只跑静态检查 (skill 可读性 / 元数据 / 依赖 / samples 契约), 不调 LLM
|
|
1102
877
|
--lang <zh|en> 切换输出语言
|
|
1103
878
|
|
|
1104
879
|
示例:
|
|
1105
|
-
omk doctor
|
|
1106
|
-
omk doctor examples/code-review/skills --json
|
|
1107
|
-
omk doctor --gate; echo $?
|
|
1108
|
-
|
|
1109
|
-
|
|
1110
|
-
|
|
1111
|
-
-
|
|
1112
|
-
-
|
|
1113
|
-
|
|
1114
|
-
|
|
1115
|
-
|
|
1116
|
-
|
|
1117
|
-
|
|
880
|
+
omk doctor my-skill --html /tmp/report.html # 深度体检 + HTML 报告 (默认)
|
|
881
|
+
omk doctor examples/code-review/skills --json > r.json # JSON 给 CI / 外部工具消费
|
|
882
|
+
omk doctor --gate; echo $? # CI 模式: fatal 问题 exit 1, 警告不阻断
|
|
883
|
+
omk doctor --static-only # 无 LLM 环境 (CI / 断网) 跑纯静态检查
|
|
884
|
+
|
|
885
|
+
doctor = LLM 健康度体检 (单次 LLM 会话):
|
|
886
|
+
- 7 个内置维度: 触发与边界 / 文档清晰 / 指令精确性 / 依赖检查 / 工具规范 / 安全与合规 / 示例完备
|
|
887
|
+
- 用户可扩展: 在自己代码里 registerHealthDimension(spec) 加自定义维度,
|
|
888
|
+
会自动加入同一次 LLM 调用的 prompt + 报告 (顺序 = 注册顺序)
|
|
889
|
+
- 每维度独立给 健康/亚健康/不健康/不适用 + findings + 改进建议
|
|
890
|
+
- HTML 报告: 维度按 fail→warn→pass→skipped 排, 错误 finding 排前面
|
|
891
|
+
|
|
892
|
+
注: omk eval 内部仍跑静态 skill-readability/metadata/dependency
|
|
893
|
+
gate 保护评测质量, 不走 omk doctor 这条 LLM 路径 (角色分离: doctor=审计, eval=评测)。
|
|
894
|
+
LLM 连通性可用 omk eval --skip-connectivity 跳过 (--resume 时自动)。
|
|
1118
895
|
`.trim() + '\n',
|
|
1119
896
|
en: `
|
|
1120
|
-
oh-my-knowledge — omk doctor health
|
|
897
|
+
oh-my-knowledge — omk doctor health audit (LLM-judge)
|
|
1121
898
|
|
|
1122
899
|
Usage:
|
|
1123
|
-
omk doctor [path] Run
|
|
1124
|
-
omk doctor Batch
|
|
900
|
+
omk doctor [path] Run deep LLM-based health audit on path
|
|
901
|
+
omk doctor Batch audit current dir (or ./skills)
|
|
1125
902
|
|
|
1126
903
|
Arguments:
|
|
1127
|
-
path A .md file, directory, or omit (= cwd). Directory
|
|
904
|
+
path A .md file, directory, or omit (= cwd). Directory batches all skills.
|
|
1128
905
|
|
|
1129
906
|
Options:
|
|
1130
907
|
--json Print DoctorReport JSON to stdout (CI-friendly)
|
|
1131
|
-
--gate Silent mode: exit
|
|
1132
|
-
--executor <name> executor
|
|
1133
|
-
--model <name> model name (
|
|
1134
|
-
--
|
|
908
|
+
--gate Silent mode: exit 1 only on fatal failure; warnings_only exits 0
|
|
909
|
+
--executor <name> LLM executor (default 'claude'; switchable to anthropic-api/codex etc)
|
|
910
|
+
--model <name> model name (default 'sonnet')
|
|
911
|
+
--samples <path> Explicit eval samples file
|
|
912
|
+
--timeout <seconds> LLM session timeout (default 600)
|
|
913
|
+
--html <path> Also write a visual HTML report to <path> (combines with --json)
|
|
914
|
+
--static-only Offline mode: run only static checks (readability / metadata / deps / samples contract); no LLM call
|
|
1135
915
|
--lang <zh|en> Output language
|
|
1136
916
|
|
|
1137
917
|
Examples:
|
|
1138
|
-
omk doctor
|
|
1139
|
-
omk doctor examples/code-review/skills --json
|
|
1140
|
-
omk doctor --gate; echo $?
|
|
1141
|
-
|
|
1142
|
-
|
|
1143
|
-
|
|
1144
|
-
-
|
|
1145
|
-
|
|
1146
|
-
-
|
|
1147
|
-
|
|
1148
|
-
|
|
1149
|
-
|
|
1150
|
-
|
|
1151
|
-
|
|
918
|
+
omk doctor my-skill --html /tmp/report.html # deep audit + HTML report (default)
|
|
919
|
+
omk doctor examples/code-review/skills --json > r.json # JSON for CI / external tools
|
|
920
|
+
omk doctor --gate; echo $? # CI mode: fatal failures exit 1; warnings do not block
|
|
921
|
+
omk doctor --static-only # offline (CI / no LLM) static checks only
|
|
922
|
+
|
|
923
|
+
doctor = LLM health audit (single LLM session):
|
|
924
|
+
- 7 builtin dimensions: trigger & boundary / doc clarity / instruction precision /
|
|
925
|
+
dependency / tool conventions / security & compliance / example completeness
|
|
926
|
+
- User-extensible: call registerHealthDimension(spec) in your code to add custom
|
|
927
|
+
dimensions; they join the same LLM call's prompt + report (order = registration order)
|
|
928
|
+
- Each dim graded healthy / sub-healthy / unhealthy / N-A + findings + suggestions
|
|
929
|
+
- HTML report: dims sorted fail→warn→pass→skipped; errors first within each dim
|
|
930
|
+
|
|
931
|
+
Note: omk eval still runs static skill-readability/metadata/dependency gates
|
|
932
|
+
internally (separate from this doctor command). Roles: doctor=audit, eval=evaluate.
|
|
933
|
+
LLM connectivity for omk eval can be skipped with --skip-connectivity (auto on --resume).
|
|
1152
934
|
`.trim() + '\n',
|
|
1153
935
|
},
|
|
1154
936
|
'cli.doctor.no_skill_found': {
|
|
1155
937
|
zh: '未在 {path} 下发现 skill 文件。\n doctor 期望 .md 文件、目录(包含 .md 或 SKILL.md)或 cwd 下的 skills/ 子目录。',
|
|
1156
938
|
en: 'No skills found at {path}.\n doctor expects a .md file, a directory (containing .md or SKILL.md), or skills/ under cwd.',
|
|
1157
939
|
},
|
|
940
|
+
'cli.doctor.samples_detected': {
|
|
941
|
+
zh: '✓ 使用评测用例文件:{path}',
|
|
942
|
+
en: '✓ Using eval samples file: {path}',
|
|
943
|
+
},
|
|
1158
944
|
'cli.doctor.gate_blocked': {
|
|
1159
945
|
zh: 'skill 健康检查未通过, 评测已中止。doctor 是评测必经环节, 无 skip 选项 — 请修复上述问题后重跑。',
|
|
1160
946
|
en: 'skill health check failed; evaluation aborted. doctor is mandatory and not skippable — fix the issues above and re-run.',
|