oh-my-knowledge 0.26.0 → 0.28.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (167) hide show
  1. package/README.md +105 -346
  2. package/README.zh.md +142 -375
  3. package/dist/src/analysis/report-diagnostics.d.ts +2 -2
  4. package/dist/src/analysis/report-diagnostics.js +2 -2
  5. package/dist/src/analysis/sample-diagnostics.d.ts +3 -3
  6. package/dist/src/analysis/sample-diagnostics.js +3 -3
  7. package/dist/src/authoring/generator.d.ts.map +1 -1
  8. package/dist/src/authoring/generator.js +9 -8
  9. package/dist/src/authoring/generator.js.map +1 -1
  10. package/dist/src/cli/cli-exit.d.ts +15 -0
  11. package/dist/src/cli/cli-exit.d.ts.map +1 -0
  12. package/dist/src/cli/cli-exit.js +19 -0
  13. package/dist/src/cli/cli-exit.js.map +1 -0
  14. package/dist/src/cli/commands/_shared.d.ts +12 -0
  15. package/dist/src/cli/commands/_shared.d.ts.map +1 -0
  16. package/dist/src/cli/commands/_shared.js +26 -0
  17. package/dist/src/cli/commands/_shared.js.map +1 -0
  18. package/dist/src/cli/commands/doctor.d.ts +2 -0
  19. package/dist/src/cli/commands/doctor.d.ts.map +1 -0
  20. package/dist/src/cli/commands/doctor.js +174 -0
  21. package/dist/src/cli/commands/doctor.js.map +1 -0
  22. package/dist/src/cli/commands/eval-gold.d.ts +2 -0
  23. package/dist/src/cli/commands/eval-gold.d.ts.map +1 -0
  24. package/dist/src/cli/commands/eval-gold.js +137 -0
  25. package/dist/src/cli/commands/eval-gold.js.map +1 -0
  26. package/dist/src/cli/commands/eval-runner.d.ts +2 -0
  27. package/dist/src/cli/commands/eval-runner.d.ts.map +1 -0
  28. package/dist/src/cli/commands/eval-runner.js +299 -0
  29. package/dist/src/cli/commands/eval-runner.js.map +1 -0
  30. package/dist/src/cli/commands/eval.d.ts +2 -0
  31. package/dist/src/cli/commands/eval.d.ts.map +1 -0
  32. package/dist/src/cli/commands/eval.js +11 -0
  33. package/dist/src/cli/commands/eval.js.map +1 -0
  34. package/dist/src/cli/commands/evolve.d.ts +2 -0
  35. package/dist/src/cli/commands/evolve.d.ts.map +1 -0
  36. package/dist/src/cli/commands/evolve.js +115 -0
  37. package/dist/src/cli/commands/evolve.js.map +1 -0
  38. package/dist/src/cli/commands/init.d.ts +2 -0
  39. package/dist/src/cli/commands/init.d.ts.map +1 -0
  40. package/dist/src/cli/commands/init.js +110 -0
  41. package/dist/src/cli/commands/init.js.map +1 -0
  42. package/dist/src/cli/commands/observe.d.ts +2 -0
  43. package/dist/src/cli/commands/observe.d.ts.map +1 -0
  44. package/dist/src/cli/commands/observe.js +77 -0
  45. package/dist/src/cli/commands/observe.js.map +1 -0
  46. package/dist/src/cli/commands/registry.d.ts +11 -0
  47. package/dist/src/cli/commands/registry.d.ts.map +1 -0
  48. package/dist/src/cli/commands/registry.js +24 -0
  49. package/dist/src/cli/commands/registry.js.map +1 -0
  50. package/dist/src/cli/commands/sample.d.ts +2 -0
  51. package/dist/src/cli/commands/sample.d.ts.map +1 -0
  52. package/dist/src/cli/commands/sample.js +121 -0
  53. package/dist/src/cli/commands/sample.js.map +1 -0
  54. package/dist/src/cli/commands/studio.d.ts +2 -0
  55. package/dist/src/cli/commands/studio.d.ts.map +1 -0
  56. package/dist/src/cli/commands/studio.js +76 -0
  57. package/dist/src/cli/commands/studio.js.map +1 -0
  58. package/dist/src/cli/i18n-dict.d.ts +2 -4
  59. package/dist/src/cli/i18n-dict.d.ts.map +1 -1
  60. package/dist/src/cli/i18n-dict.js +461 -675
  61. package/dist/src/cli/i18n-dict.js.map +1 -1
  62. package/dist/src/cli/index.js +43 -1516
  63. package/dist/src/cli/index.js.map +1 -1
  64. package/dist/src/cli/parse-run-config.d.ts +3 -4
  65. package/dist/src/cli/parse-run-config.d.ts.map +1 -1
  66. package/dist/src/cli/parse-run-config.js +5 -5
  67. package/dist/src/cli/parse-run-config.js.map +1 -1
  68. package/dist/src/cli/parse-strict.d.ts +0 -13
  69. package/dist/src/cli/parse-strict.d.ts.map +1 -1
  70. package/dist/src/cli/parse-strict.js +2 -1
  71. package/dist/src/cli/parse-strict.js.map +1 -1
  72. package/dist/src/doctor/health/builtin-dimensions.d.ts +11 -0
  73. package/dist/src/doctor/health/builtin-dimensions.d.ts.map +1 -0
  74. package/dist/src/doctor/health/builtin-dimensions.js +94 -0
  75. package/dist/src/doctor/health/builtin-dimensions.js.map +1 -0
  76. package/dist/src/doctor/health/composer.d.ts +19 -0
  77. package/dist/src/doctor/health/composer.d.ts.map +1 -0
  78. package/dist/src/doctor/health/composer.js +289 -0
  79. package/dist/src/doctor/health/composer.js.map +1 -0
  80. package/dist/src/doctor/health/dimension-registry.d.ts +13 -0
  81. package/dist/src/doctor/health/dimension-registry.d.ts.map +1 -0
  82. package/dist/src/doctor/health/dimension-registry.js +28 -0
  83. package/dist/src/doctor/health/dimension-registry.js.map +1 -0
  84. package/dist/src/doctor/health/dimension-spec.d.ts +46 -0
  85. package/dist/src/doctor/health/dimension-spec.d.ts.map +1 -0
  86. package/dist/src/doctor/health/dimension-spec.js +12 -0
  87. package/dist/src/doctor/health/dimension-spec.js.map +1 -0
  88. package/dist/src/doctor/health/parser.d.ts +27 -0
  89. package/dist/src/doctor/health/parser.d.ts.map +1 -0
  90. package/dist/src/doctor/health/parser.js +190 -0
  91. package/dist/src/doctor/health/parser.js.map +1 -0
  92. package/dist/src/doctor/health/prompt-builder.d.ts +22 -0
  93. package/dist/src/doctor/health/prompt-builder.d.ts.map +1 -0
  94. package/dist/src/doctor/health/prompt-builder.js +162 -0
  95. package/dist/src/doctor/health/prompt-builder.js.map +1 -0
  96. package/dist/src/doctor/health/register.d.ts +13 -0
  97. package/dist/src/doctor/health/register.d.ts.map +1 -0
  98. package/dist/src/doctor/health/register.js +20 -0
  99. package/dist/src/doctor/health/register.js.map +1 -0
  100. package/dist/src/doctor/html-renderer.d.ts +20 -0
  101. package/dist/src/doctor/html-renderer.d.ts.map +1 -0
  102. package/dist/src/doctor/html-renderer.js +366 -0
  103. package/dist/src/doctor/html-renderer.js.map +1 -0
  104. package/dist/src/doctor/index.d.ts +1 -1
  105. package/dist/src/doctor/index.d.ts.map +1 -1
  106. package/dist/src/doctor/index.js +57 -8
  107. package/dist/src/doctor/index.js.map +1 -1
  108. package/dist/src/doctor/preflight.d.ts +2 -2
  109. package/dist/src/doctor/preflight.js +2 -2
  110. package/dist/src/doctor/renderer.d.ts +4 -0
  111. package/dist/src/doctor/renderer.d.ts.map +1 -1
  112. package/dist/src/doctor/renderer.js +72 -18
  113. package/dist/src/doctor/renderer.js.map +1 -1
  114. package/dist/src/doctor/rules.d.ts +6 -5
  115. package/dist/src/doctor/rules.d.ts.map +1 -1
  116. package/dist/src/doctor/rules.js +4 -3
  117. package/dist/src/doctor/rules.js.map +1 -1
  118. package/dist/src/eval-core/fact-checker.js +1 -1
  119. package/dist/src/eval-core/fact-checker.js.map +1 -1
  120. package/dist/src/eval-core/layer-gates.d.ts +1 -1
  121. package/dist/src/eval-core/layer-gates.js +1 -1
  122. package/dist/src/eval-core/verdict.d.ts +4 -4
  123. package/dist/src/eval-core/verdict.d.ts.map +1 -1
  124. package/dist/src/eval-core/verdict.js +2 -2
  125. package/dist/src/eval-workflows/batch-evaluation-workflow.d.ts +15 -2
  126. package/dist/src/eval-workflows/batch-evaluation-workflow.d.ts.map +1 -1
  127. package/dist/src/eval-workflows/batch-evaluation-workflow.js +13 -2
  128. package/dist/src/eval-workflows/batch-evaluation-workflow.js.map +1 -1
  129. package/dist/src/eval-workflows/evaluation-pipeline.d.ts +6 -3
  130. package/dist/src/eval-workflows/evaluation-pipeline.d.ts.map +1 -1
  131. package/dist/src/eval-workflows/evaluation-pipeline.js +10 -9
  132. package/dist/src/eval-workflows/evaluation-pipeline.js.map +1 -1
  133. package/dist/src/eval-workflows/run-evaluation.d.ts +1 -1
  134. package/dist/src/eval-workflows/run-evaluation.d.ts.map +1 -1
  135. package/dist/src/eval-workflows/run-evaluation.js +13 -6
  136. package/dist/src/eval-workflows/run-evaluation.js.map +1 -1
  137. package/dist/src/executors/script.d.ts.map +1 -1
  138. package/dist/src/executors/script.js +16 -3
  139. package/dist/src/executors/script.js.map +1 -1
  140. package/dist/src/grading/debias-validate.d.ts +2 -2
  141. package/dist/src/grading/debias-validate.js +2 -2
  142. package/dist/src/grading/gold-cli.d.ts +2 -5
  143. package/dist/src/grading/gold-cli.d.ts.map +1 -1
  144. package/dist/src/grading/gold-cli.js +4 -8
  145. package/dist/src/grading/gold-cli.js.map +1 -1
  146. package/dist/src/grading/judge.d.ts +1 -1
  147. package/dist/src/renderer/html-renderer.js +1 -1
  148. package/dist/src/renderer/layout.js +5 -5
  149. package/dist/src/renderer/layout.js.map +1 -1
  150. package/dist/src/renderer/skill-health-renderer.d.ts +1 -1
  151. package/dist/src/renderer/skill-health-renderer.js +1 -1
  152. package/dist/src/renderer/summary.js +8 -8
  153. package/dist/src/server/report-server.d.ts.map +1 -1
  154. package/dist/src/server/report-server.js +8 -7
  155. package/dist/src/server/report-server.js.map +1 -1
  156. package/dist/src/types/doctor.d.ts +44 -6
  157. package/dist/src/types/doctor.d.ts.map +1 -1
  158. package/dist/src/types/doctor.js +3 -0
  159. package/dist/src/types/doctor.js.map +1 -1
  160. package/dist/src/types/eval.d.ts +1 -1
  161. package/dist/src/types/report.d.ts +2 -2
  162. package/dist/src/types/report.d.ts.map +1 -1
  163. package/package.json +1 -1
  164. package/dist/src/cli/coverage-renderer.d.ts +0 -15
  165. package/dist/src/cli/coverage-renderer.d.ts.map +0 -1
  166. package/dist/src/cli/coverage-renderer.js +0 -74
  167. package/dist/src/cli/coverage-renderer.js.map +0 -1
@@ -16,9 +16,7 @@
16
16
  * 2. **保留原文的白名单 (产品术语 / 命令 / 文件名)**
17
17
  * 以下 token 在两种语言里都保留原文, 不翻译:
18
18
  * - 产品名: omk, oh-my-knowledge, Claude, npm
19
- * - 子命令空间和命令名: bench, analyze, run, report, init, evolve, gold,
20
- * diff, ci, gen-samples, debias-validate, saturation, verdict,
21
- * diagnose, failures
19
+ * - 命令名: init, doctor, eval, observe, evolve, sample, studio, gold
22
20
  * - omk 核心业务术语: skill, variant, sample, judge, executor (出现在产品
23
21
  * UI 里时首字母可大写如 "Skill 评测", 描述句中保持小写)
24
22
  * - 技术参数: --lang, --control, --treatment, --bootstrap, --judge-repeat,
@@ -48,21 +46,9 @@
48
46
  * Record 类型自动强制每 key 加新语言版本。
49
47
  */
50
48
  export const CLI_DICT = {
51
- 'cli.common.lang_invalid_silent': {
52
- zh: '无效的语言代码: {value} (仅支持 zh / en, 已使用默认 zh)',
53
- en: 'Invalid language code: {value} (supported: zh / en, using default zh)',
54
- },
55
- 'cli.common.help_hint': {
56
- zh: "运行 'omk --help' 查看用法",
57
- en: "Run 'omk --help' to see usage",
58
- },
59
49
  'cli.common.unknown_domain': {
60
- zh: "未知顶层命令: {domain} (请用 'omk bench <command>'、'omk doctor [path]' 或 'omk analyze <dir>')",
61
- en: "Unknown domain: {domain} (use 'omk bench <command>', 'omk doctor [path]' or 'omk analyze <dir>')",
62
- },
63
- 'cli.common.unknown_bench_command': {
64
- zh: "未知子命令: bench {command} (运行 'omk --help' 查看可用列表)",
65
- en: "Unknown bench command: {command} (run 'omk --help' to see all commands)",
50
+ zh: "未知命令:{domain}。运行 'omk --help' 查看可用命令。",
51
+ en: "Unknown command: {domain}. Run 'omk --help' to see available commands.",
66
52
  },
67
53
  'cli.init.scaffolded': {
68
54
  zh: '已初始化测评项目: {dir}',
@@ -73,7 +59,7 @@ export const CLI_DICT = {
73
59
  en: 'Next steps:',
74
60
  },
75
61
  'cli.init.next_step_edit_samples': {
76
- zh: ' 1. 编辑 eval-samples.json, 加入你要测的测评用例',
62
+ zh: ' 1. 编辑 eval-samples.json,加入你要测的评测用例',
77
63
  en: ' 1. Edit eval-samples.json to add your test cases',
78
64
  },
79
65
  'cli.init.next_step_edit_skills': {
@@ -81,8 +67,8 @@ export const CLI_DICT = {
81
67
  en: ' 2. Edit skills/code-review-v1/SKILL.md and skills/code-review-v2/SKILL.md with your skill versions',
82
68
  },
83
69
  'cli.init.next_step_run': {
84
- zh: ' 3. 运行: omk bench run --control code-review-v1 --treatment code-review-v2',
85
- en: ' 3. Run: omk bench run --control code-review-v1 --treatment code-review-v2',
70
+ zh: ' 3. 运行: omk eval --control code-review-v1 --treatment code-review-v2',
71
+ en: ' 3. Run: omk eval --control code-review-v1 --treatment code-review-v2',
86
72
  },
87
73
  'cli.init.note_codex_executor': {
88
74
  zh: '\n注: omk 评测时把 SKILL.md 整文(含 frontmatter)作为 system prompt 注入 — 跨 executor 一致(claude / codex / openai-api / gemini 都走同一条路径,不依赖任何 executor 的 native skill auto-discovery 或 Skill 工具机制)。frontmatter 在 prompt 头部对 model 行为无显著影响。\n模板带 Claude Code 兼容的 frontmatter(name + description)是为了让同一份 directory-skill 也能 deploy 到 Claude Code:把整个目录复制到 ~/.claude/skills/code-review-v1/(整目录,不是单个 SKILL.md),Claude SDK 才能识别。这是 omk 评测之外的 bonus,一份文件双向 dogfood。',
@@ -92,6 +78,22 @@ export const CLI_DICT = {
92
78
  zh: '\n💡 新版本可用: {old} → {new}, 运行 npm update {pkg} -g 升级\n\n',
93
79
  en: '\n💡 New version available: {old} → {new}, run npm update {pkg} -g to upgrade\n\n',
94
80
  },
81
+ 'cli.run.power_warning_tiny_n': {
82
+ zh: '⚠ N={n} < 5:仅适合探索,任何结论都不可靠,CI 会很宽。需要决策时建议 ≥20 条评测用例。',
83
+ en: '⚠ N={n} < 5 (exploration-only): any conclusion is unreliable, CI will be uselessly wide. Decisions need ≥20 cases.',
84
+ },
85
+ 'cli.run.power_warning_small_n': {
86
+ zh: '⚠ N={n} < 20:只能识别很大的效果(Cohen\'s d > 0.8),中等效果(d ≈ 0.5)很难检出。要做可靠决策建议 ≥20 条评测用例。',
87
+ en: '⚠ N={n} < 20 (large-effect-only, Cohen\'s d > 0.8): medium effects (d ≈ 0.5) hard to detect. For confident decisions consider ≥20 cases.',
88
+ },
89
+ 'cli.run.power_warning_repeat_one': {
90
+ zh: '⚠ --repeat=1:单轮评测无法测稳定性(CV 会标记为未测量)。用 --repeat 3+ 检测同一 variant 内部方差。',
91
+ en: '⚠ --repeat=1: single-run cannot measure stability (CV will be marked "not measured"). Use --repeat 3+ to detect within-variant variance.',
92
+ },
93
+ 'cli.run.dry_run_no_scores': {
94
+ zh: 'eval dry-run:仅预览任务,不检查分数',
95
+ en: 'Eval dry-run: no scores to check',
96
+ },
95
97
  'cli.progress.preflight_starting': {
96
98
  zh: '⏳ 正在预检模型连通性...\n',
97
99
  en: '⏳ Preflight: checking model connectivity...\n',
@@ -168,6 +170,14 @@ export const CLI_DICT = {
168
170
  zh: '\n✅ 批量评测完成\n',
169
171
  en: '\n✅ Batch evaluation done\n',
170
172
  },
173
+ 'cli.run.batch_verdict_header': {
174
+ zh: '批量评测结论:{status}({passed}/{total} 通过)',
175
+ en: 'Batch verdict: {status} ({passed}/{total} passed)',
176
+ },
177
+ 'cli.run.batch_child_report_missing': {
178
+ zh: '⚠ 子报告缺失:{id},将按不可 ship 处理。\n',
179
+ en: '⚠ Child report missing: {id}; treating it as not shippable.\n',
180
+ },
171
181
  'cli.run.eval_complete': {
172
182
  zh: '\n✅ 评测完成\n',
173
183
  en: '\n✅ Evaluation done\n',
@@ -180,6 +190,10 @@ export const CLI_DICT = {
180
190
  zh: '📄 报告已保存到: {path}\n',
181
191
  en: '📄 Report saved to: {path}\n',
182
192
  },
193
+ 'cli.run.report_only_gate_skipped': {
194
+ zh: 'ℹ 已启用 report-only 模式:保留 verdict 输出,但本次不使用 verdict 改写 exit code。\n',
195
+ en: 'ℹ Report-only mode enabled: verdict is still printed, but it will not affect the exit code.\n',
196
+ },
183
197
  'cli.run.report_server_running': {
184
198
  zh: '\n📊 报告服务已启动: {url}\n',
185
199
  en: '\n📊 Report server running at {url}\n',
@@ -197,8 +211,8 @@ export const CLI_DICT = {
197
211
  en: '\n💡 Non-interactive environment, skipping report server\n',
198
212
  },
199
213
  'cli.run.no_serve_view_hint': {
200
- zh: ' 查看报告: omk bench report --reports-dir {dir}\n',
201
- en: ' View report: omk bench report --reports-dir {dir}\n',
214
+ zh: ' 查看报告:omk studio --reports-dir {dir}(报告 ID:{id})\n',
215
+ en: ' View report: omk studio --reports-dir {dir} (report id: {id})\n',
202
216
  },
203
217
  'cli.run.gold_load_failed': {
204
218
  zh: '\n⚠ gold dataset 加载失败 ({dir}):\n',
@@ -216,9 +230,9 @@ export const CLI_DICT = {
216
230
  zh: '❌ 错误: {message}',
217
231
  en: '❌ Error: {message}',
218
232
  },
219
- 'cli.analyze.view_in_browser': {
220
- zh: "在浏览器查看: omk bench report # 打开后点首页的 \"📊 Skill 健康度日报\"",
221
- en: "View in browser: omk bench report # then click \"📊 Skill health report\" on the home page",
233
+ 'cli.observe.view_hint': {
234
+ zh: '分析 JSON 已写入 output-dir;后续可用 omk observe 持续生成日报。',
235
+ en: 'Analysis JSON written to output-dir; use omk observe to keep producing health reports.',
222
236
  },
223
237
  'cli.common.skill_dir_not_found': {
224
238
  zh: '未找到 skill 目录: {path}',
@@ -237,23 +251,31 @@ export const CLI_DICT = {
237
251
  en: 'No judge configured. Pass --judge-models <executor:model> or ensure the report has meta.judgeModels.',
238
252
  },
239
253
  'cli.common.judge_models_single_only': {
240
- zh: 'bench {cmd} 仅支持单评委。--judge-models 只能传一个 executor:model entry。',
241
- en: 'bench {cmd} only supports a single judge. --judge-models accepts exactly one executor:model entry.',
242
- },
243
- 'cli.common.usage_gold_validate': {
244
- zh: '用法: omk bench gold validate <dir>',
245
- en: 'Usage: omk bench gold validate <dir>',
254
+ zh: '{cmd} 仅支持单评委。--judge-models 只能传一个 executor:model entry。',
255
+ en: '{cmd} only supports a single judge. --judge-models accepts exactly one executor:model entry.',
246
256
  },
247
257
  'cli.common.warn_load_samples_failed': {
248
258
  zh: '⚠ 加载 samples 文件失败 ({path}): {message}\n',
249
259
  en: '⚠ Failed to load samples file ({path}): {message}\n',
250
260
  },
261
+ 'cli.studio.started': {
262
+ zh: 'studio 已启动:{url}',
263
+ en: 'Studio running at {url}',
264
+ },
265
+ 'cli.studio.stop_hint': {
266
+ zh: '按 Ctrl+C 停止服务',
267
+ en: 'Press Ctrl+C to stop',
268
+ },
269
+ 'cli.studio.open_failed': {
270
+ zh: '⚠ 无法自动打开浏览器({command}):{message}\n',
271
+ en: '⚠ Failed to open browser automatically ({command}): {message}\n',
272
+ },
251
273
  'cli.gen.skill_skipped_existing': {
252
274
  zh: '⏭️ {name}: eval-samples 已存在, 跳过\n',
253
275
  en: '⏭️ {name}: eval-samples already exists, skipping\n',
254
276
  },
255
277
  'cli.gen.skill_generating': {
256
- zh: '🔄 {name}: 正在生成 {count} 条测评用例...\n',
278
+ zh: '🔄 {name}: 正在生成 {count} 条评测用例...\n',
257
279
  en: '🔄 {name}: generating {count} test cases...\n',
258
280
  },
259
281
  'cli.gen.skill_done': {
@@ -269,19 +291,19 @@ export const CLI_DICT = {
269
291
  en: 'No eval-samples need generating (all skills already have paired files)',
270
292
  },
271
293
  'cli.gen.batch_summary': {
272
- zh: '\n共生成 {n} 份 eval-samples, 请审查后运行: omk bench run --batch',
273
- en: '\nGenerated {n} eval-samples files. Review them, then run: omk bench run --batch',
294
+ zh: '\n共生成 {n} 份 eval-samples, 请审查后运行: omk eval --batch',
295
+ en: '\nGenerated {n} eval-samples files. Review them, then run: omk eval --batch',
274
296
  },
275
297
  'cli.gen.specify_skill_path': {
276
- zh: '请指定 skill 文件路径, 例如: omk bench gen-samples skills/my-skill.md',
277
- en: 'Please specify a skill file path, e.g.: omk bench gen-samples skills/my-skill.md',
298
+ zh: '请指定 skill 文件路径, 例如: omk sample skills/my-skill.md',
299
+ en: 'Please specify a skill file path, e.g.: omk sample skills/my-skill.md',
278
300
  },
279
301
  'cli.gen.samples_already_exists': {
280
302
  zh: 'eval-samples.json 已存在。如需覆盖请先删除该文件。',
281
303
  en: 'eval-samples.json already exists. Delete it first if you want to overwrite.',
282
304
  },
283
305
  'cli.gen.single_generating': {
284
- zh: '🔄 正在生成 {count} 条测评用例...\n',
306
+ zh: '🔄 正在生成 {count} 条评测用例...\n',
285
307
  en: '🔄 Generating {count} test cases...\n',
286
308
  },
287
309
  'cli.gen.single_done': {
@@ -289,20 +311,20 @@ export const CLI_DICT = {
289
311
  en: '✅ Generated {n} samples → {path}{cost}\n',
290
312
  },
291
313
  'cli.gen.review_hint': {
292
- zh: '\n请审查生成的测评用例后运行: omk bench run',
293
- en: '\nReview the generated test cases, then run: omk bench run',
314
+ zh: '\n请审查生成的评测用例后运行: omk eval',
315
+ en: '\nReview the generated test cases, then run: omk eval',
294
316
  },
295
317
  'cli.gen.failed': {
296
318
  zh: '生成失败: {message}',
297
319
  en: 'Generation failed: {message}',
298
320
  },
299
321
  'cli.evolve.specify_skill_path': {
300
- zh: '请指定 skill 文件路径, 例如: omk bench evolve skills/my-skill.md',
301
- en: 'Please specify a skill file path, e.g.: omk bench evolve skills/my-skill.md',
322
+ zh: '请指定 skill 文件路径, 例如: omk evolve skills/my-skill.md',
323
+ en: 'Please specify a skill file path, e.g.: omk evolve skills/my-skill.md',
302
324
  },
303
325
  'cli.evolve.section_header': {
304
- zh: '\n=== Evolution: {path} ===\n',
305
- en: '\n=== Evolution: {path} ===\n',
326
+ zh: '\n=== Improve skill: {path} ===\n',
327
+ en: '\n=== Improve skill: {path} ===\n',
306
328
  },
307
329
  'cli.evolve.round_baseline': {
308
330
  zh: '第 0 轮 (基线): score={score} ({cost})\n',
@@ -329,630 +351,319 @@ export const CLI_DICT = {
329
351
  en: 'All versions saved at: {dir}/\n',
330
352
  },
331
353
  'cli.evolve.report_link': {
332
- zh: '📊 评测报告: omk bench report (ID: {id})\n',
333
- en: '📊 Report: omk bench report (ID: {id})\n',
334
- },
335
- 'cli.gold.created_files': {
336
- zh: '已在 {dir} 创建 {n} 个文件:',
337
- en: 'Created {n} files in {dir}:',
338
- },
339
- 'cli.gold.next_step_edit_annotations': {
340
- zh: '\n下一步: 编辑 annotations.yaml 加入真实标注 → 跑 omk bench gold validate',
341
- en: '\nNext step: edit annotations.yaml with real annotations → run omk bench gold validate',
342
- },
343
- 'cli.gold.validate_ok': {
344
- zh: '✓ gold dataset OK — 共 {n} 条标注',
345
- en: '✓ gold dataset OK — {n} annotations',
346
- },
347
- 'cli.debias.warn_cost_doubles': {
348
- zh: '\n⚠ debias-validate 会重判所有 (sample × variant), judge 成本大约翻倍。\n',
349
- en: '\n⚠ debias-validate will re-judge all (sample × variant) pairs; judge cost will roughly double.\n',
350
- },
351
- 'cli.saturation.no_data': {
352
- zh: '该 report 没有 saturation 数据 (需要 --repeat ≥ 2 才会记录)。',
353
- en: 'This report has no saturation data (requires --repeat ≥ 2 to record).',
354
- },
355
- 'cli.saturation.verdict_header': {
356
- zh: '\n Saturation verdict (复述持久化结果)\n',
357
- en: '\n Saturation verdict (replaying persisted result)\n',
358
- },
359
- 'cli.saturation.variant_no_trace': {
360
- zh: ' {variant}: 没有 trace 数据',
361
- en: ' {variant}: no trace data',
362
- },
363
- 'cli.saturation.variant_label': {
364
- zh: ' {variant}:',
365
- en: ' {variant}:',
366
- },
367
- 'cli.saturation.checkpoints': {
368
- zh: ' 检查点: {n} (N={list})',
369
- en: ' checkpoints: {n} (N={list})',
354
+ zh: '📊 查看报告:omk studio(报告 ID:{id})\n',
355
+ en: '📊 View report: omk studio (report id: {id})\n',
370
356
  },
371
- 'cli.saturation.last_point': {
372
- zh: ' 最后一点 mean={mean}, CI=[{lo}, {hi}]',
373
- en: ' last point mean={mean}, CI=[{lo}, {hi}]',
374
- },
375
- 'cli.saturation.persisted_verdict': {
376
- zh: ' 持久化判定 ({method}): {result} - {reason}',
377
- en: ' persisted verdict ({method}): {result} - {reason}',
378
- },
379
- 'cli.saturation.persisted_verdict_saturated': {
380
- zh: '已饱和@N={n}',
381
- en: 'saturated@N={n}',
382
- },
383
- 'cli.saturation.persisted_verdict_unsaturated': {
384
- zh: '未饱和',
385
- en: 'not saturated',
386
- },
387
- 'cli.saturation.skipped_too_few_points': {
388
- zh: ' 判定: 数据点数 {n} < 5, 跳过 (需要跑 --repeat 5 以上才会输出)',
389
- en: ' verdict: only {n} data points (< 5), skipping (need --repeat 5 or more)',
357
+ 'cli.help.product_main': {
358
+ zh: `
359
+ oh-my-knowledge — 知识载体工作台
360
+
361
+ 用法:
362
+ omk init [dir] 初始化一个 skill 评测项目
363
+ omk doctor [path] LLM 健康度审计(7 内置维度 + 可扩展);--static-only 切离线静态模式
364
+ omk eval [options] 离线评测:比较版本,输出 verdict + report
365
+ omk observe <sessions-dir> 线上观测:真实 session、gap、失败率、inbox
366
+ omk evolve <skill> 多轮自动迭代改进 skill
367
+ omk sample <skill> 生成或补齐 eval-samples 评测用例(或 --batch 批量模式)
368
+ omk studio 打开本地工作台浏览报告
369
+
370
+ 主路径:
371
+ omk doctor
372
+ omk eval --control code-review-v1 --treatment code-review-v2
373
+ omk observe ~/.claude/projects/<project>
374
+ omk evolve skills/code-review-v2/SKILL.md
375
+ omk studio
376
+
377
+ 通用选项:
378
+ --lang <zh|en> CLI 输出语言(默认:zh,也可设 OMK_LANG)
379
+
380
+ 运行 'omk <command> --help' 查看单个命令的参数。
381
+ `,
382
+ en: `
383
+ oh-my-knowledge — Knowledge Artifact Workbench
384
+
385
+ Usage:
386
+ omk init [dir] Scaffold a skill evaluation project
387
+ omk doctor [path] LLM health audit (7 builtin dimensions, extensible); --static-only for offline static checks
388
+ omk eval [options] Offline evaluation: compare versions, emit verdict + report
389
+ omk observe <sessions-dir> Production observation: sessions, gaps, failure rate, inbox
390
+ omk evolve <skill> Auto-iterate a skill through multi-round eval loops
391
+ omk sample <skill> Generate or fill eval-samples test cases (or --batch for all skills)
392
+ omk studio Open the local workbench to browse reports
393
+
394
+ Main workflow:
395
+ omk doctor
396
+ omk eval --control code-review-v1 --treatment code-review-v2
397
+ omk observe ~/.claude/projects/<project>
398
+ omk evolve skills/code-review-v2/SKILL.md
399
+ omk studio
400
+
401
+ Common options:
402
+ --lang <zh|en> CLI output language (default: zh, or set OMK_LANG)
403
+
404
+ Run 'omk <command> --help' for command-specific options.
405
+ `,
390
406
  },
391
- 'cli.help.main': {
407
+ 'cli.help.init_usage': {
392
408
  zh: `
393
- oh-my-knowledge — 知识工件评测工具集
409
+ omk init — 初始化 skill 评测项目
394
410
 
395
- 用法:
396
- omk bench run [options] 跑一轮评测
397
- omk bench report [options] 启动报告 server
398
- omk bench gate [options] 跑评测 + 应用 gate, exit code 0/1 (CI/CD 用)
399
- omk bench init [dir] 初始化一个评测项目
400
- omk bench gen-samples [skill] 从 skill 内容生成 eval-samples
401
- omk bench diff <id1> <id2> 对比两份评测报告
402
- omk bench evolve <skill> 通过迭代评测自我改进 skill
411
+ 用法:
412
+ omk init [dir]
403
413
 
404
- omk doctor [path] skill 健康检查(评测前置门禁)
405
- omk analyze <dir> 分析 cc session trace, 生成 skill 健康度日报 (v0.18)
414
+ 生成内容:
415
+ eval-samples.json 示例评测用例
416
+ skills/code-review-v1/SKILL.md 基线 skill
417
+ skills/code-review-v2/SKILL.md 实验组 skill
406
418
 
407
- bench run 选项:
419
+ 下一步:
420
+ 1. 编辑 eval-samples.json,替换成你的真实评测用例
421
+ 2. 编辑两个 SKILL.md,填入要对比的 skill 版本
422
+ 3. 运行 omk eval --control code-review-v1 --treatment code-review-v2
423
+ `,
424
+ en: `
425
+ omk init — scaffold a skill evaluation project
408
426
 
409
- --samples <path> 用例文件 (默认: eval-samples.json)
410
- --skill-dir <path> skill 定义目录 (默认: skills)
411
- --control <expr> 对照组 variant 表达式 (实验角色 = control)
412
- --treatment <v1,v2> 实验组 variant 表达式 (逗号分隔; 角色 = treatment)
413
- 每个 variant 表达式解析为一个 artifact 加上可选运行时上下文:
414
- "baseline" — 裸模型, 不注入 artifact
415
- "git:name" — 来自最后一次 commit 的 artifact
416
- "git:ref:name" — 来自指定 commit 的 artifact
417
- 带 "/" 的路径 — 直接来自文件 (例如 ./v1.md)
418
- "name@/cwd" — 附加运行时上下文 / cwd
419
- --control 和 --treatment 至少要给一个。
420
- --config <path> YAML/JSON 配置文件 (evaluation-as-code)。
421
- 在一个文件里声明 samples + variants + model + executor。
422
- CLI flag 会覆盖配置文件中的同名字段。
423
- 配置中的相对路径相对于配置文件所在目录解析。
424
- --model <name> 任务执行模型 (默认: sonnet)
425
- --output-dir <path> 报告输出目录 (默认: ~/.oh-my-knowledge/reports/)
426
- --no-judge 跳过 LLM 评委
427
- --no-cache 禁用结果缓存
428
- --dry-run 预览任务但不执行
429
- --blind 双盲 A/B 模式: 报告里隐藏 variant 名称
430
- --concurrency <n> 并发任务数 (默认: 1)
431
- --timeout <seconds> 单任务执行超时 (秒, 默认: 120)
432
- --repeat <n> 跑 N 轮做方差分析 (默认: 1)
433
- --judge-repeat <n> 每个 (sample × dimension) 调 LLM 评委 N 次评估
434
- 自洽性 (默认: 1)。多轮间高 stddev = 评委在该评分维度
435
- 上不稳定, 分数有噪声。
436
- --judge-models <list> 评委配置, 逗号分隔的 executor:model, 如
437
- claude:haiku 或 claude:opus,openai:gpt-4o。
438
- 1 条 = 单评委 (默认 claude:haiku); ≥ 2 条 = ensemble,
439
- 每个评委对所有 (sample × dimension) 打分, 报告
440
- 含每评委分布 + Pearson / MAD 评委间一致性。能反驳
441
- "Claude 评委评 Claude 同模态偏置" 的质疑。可与
442
- --judge-repeat 组合。成本 ~ N_judges × N_repeat × N_samples。
443
- --bootstrap 计算 bootstrap 置信区间 (无分布假设, 对 LLM 序数评分
444
- 比 t 区间更靠谱)。给出每个 variant 均值 CI + treatment
445
- vs control 差值的 pairwise CI (CI 不跨 0 即显著)。
446
- 同时报告 t 区间和 bootstrap, 旧工具仍可用。
447
- --bootstrap-samples <n> bootstrap 重采样次数 (默认 1000)。N>10000 触发
448
- stderr 警告提示耗时。
449
- --retry <n> 失败任务最多重试 N 次, 指数退避 (默认: 0)
450
- --resume <report-id> 从历史报告恢复, 跳过已完成任务
451
- --executor <name> 执行器: claude / claude-sdk / codex / openai / gemini /
452
- anthropic-api / openai-api, 或任意 shell 命令 (例如 "python my_provider.py")
453
- --batch 批量评测:每个 skill 独立 vs baseline
454
- 需要每个 skill 有配对的 {name}.eval-samples.json
455
- --skip-connectivity 跳过 LLM 模型连通性检测 (--resume 时自动跳过)
456
- --mcp-config <path> 通过 MCP server 抓 URL 用的 MCP 配置文件
457
- (默认: 当前目录下的 .mcp.json)
458
- --no-serve 评测后不自动启动报告 server
459
- --verbose 打印每个用例的详细进度 (执行结果 / 评分阶段)
460
- --layered-stats 默认在 HTML 报告里展开三层 (fact/behavior/judge) 独立
461
- 显著性细分。不加这个 flag 时, 细分会折叠在每个对比下
462
- 的 click-to-expand summary 里。
463
- --strict-baseline (默认开启) 对 baseline-kind variant 强制隔离 skill 自动
464
- 发现 + Skill 工具调用, 切断 ~/.claude/skills/ 污染路径,
465
- 保证 skill 评测的 construct validity。eval.yaml 显式
466
- allowedSkills 优先。
467
- --no-strict-baseline 显式关闭 strict-baseline (baseline 走默认 SDK skill
468
- 全发现)。少数场景下可能想要这个 (例如评测 skill 文档
469
- 对默认全发现行为的增量影响)。开启时 pre-flight 会
470
- stderr 提醒, 因为 verdict / Δ 易受污染。
427
+ Usage:
428
+ omk init [dir]
471
429
 
472
- bench gate 选项:
473
- (与 bench run 相同, 额外加:)
474
- --threshold <number> 三层 gate 阈值 (fact / behavior / LLM judge), 独立应用
475
- 到每一层。任一层低于阈值即失败 — 防止合成均值掩盖单层
476
- 崩塌。默认: 3.5。如果三层全空 (没有 assertion 也没在
477
- eval-samples 里定义 rubric), gate 失败并提示配置问题,
478
- 不走合成 fallback。
479
- --trivial-diff <num> 实际可忽略的最小 diff (默认 0.1)。bootstrap diff CI
480
- 显著但 |Δ| 小于此值视为"统计有效但实际无意义",标
481
- CAUTIOUS 不给 PROGRESS。
430
+ Generated files:
431
+ eval-samples.json Example test cases
432
+ skills/code-review-v1/SKILL.md Baseline skill
433
+ skills/code-review-v2/SKILL.md Treatment skill
434
+
435
+ Next steps:
436
+ 1. Edit eval-samples.json with your real test cases
437
+ 2. Edit both SKILL.md files with the skill versions to compare
438
+ 3. Run omk eval --control code-review-v1 --treatment code-review-v2
439
+ `,
440
+ },
441
+ 'cli.help.eval': {
442
+ zh: `
443
+ omk eval — 离线评测 skill 版本,并给出 ship/no-ship verdict
444
+
445
+ 用法:
446
+ omk eval --control <variant> --treatment <variant> [options]
447
+ omk eval gold <init|validate|compare> ...
448
+
449
+ 常用选项:
450
+ --samples <path> 用例文件(默认:eval-samples.json)
451
+ --skill-dir <path> skill 目录(默认:skills)
452
+ --control <expr> 对照组 variant
453
+ --treatment <v1,v2> 实验组 variant,逗号分隔
454
+ --config <path> eval.yaml / JSON 配置
455
+ --executor <name> 执行器:claude / claude-sdk / codex / openai / gemini / custom
456
+ --model <name> 任务执行模型(默认:sonnet)
457
+ --judge-models <list> 评委配置,例如 claude:haiku 或 claude:opus,openai:gpt-4o
458
+ --dry-run 预览任务,不调用模型
459
+ --batch 批量评测:每个 skill 独立 vs baseline
460
+ --bootstrap 显式开启 bootstrap CI;omk eval 默认会自动开启
461
+ --bootstrap-samples <n> bootstrap 重采样次数(默认:1000)
462
+ --threshold <number> 三层 gate 阈值(默认:3.5)
463
+ --trivial-diff <number> 实际可忽略 diff(默认:0.1)
464
+ --report-only / --no-gate 生成报告并打印 verdict,但始终 exit 0
465
+ --no-serve 评测后不自动启动报告 server
466
+
467
+ 示例:
468
+ omk eval --control baseline --treatment my-skill # 单 skill 必要性测试(baseline 是保留 variant 名,代表「不注入 skill 的裸基线」)
469
+ omk eval --control code-review-v1 --treatment code-review-v2 # 多版本 A/B
470
+ omk eval --config eval.yaml
471
+ omk eval gold compare v1-vs-v2-20260505-1200 --gold-dir gold-dataset
472
+ `,
473
+ en: `
474
+ omk eval — run offline skill evaluation and emit a ship/no-ship verdict
482
475
 
483
- 内部 = bench run + bench verdict, exit code 与 bench verdict 对齐:
484
- PROGRESS / SOLO-PASS → 0; NOISE / UNDERPOWERED / CAUTIOUS / REGRESS → 1。
485
- 数据 underpowered 时直接 FAIL, 堵住"单轮过 PASS 就 deploy"漏洞。
476
+ Usage:
477
+ omk eval --control <variant> --treatment <variant> [options]
478
+ omk eval gold <init|validate|compare> ...
486
479
 
487
- bench report 选项:
488
- --port <number> server 端口 (默认: 7799)
489
- --reports-dir <path> 报告目录 (默认: ~/.oh-my-knowledge/reports/)
490
- --export <id> 把报告导出为独立 HTML 文件
491
- --dev 开发模式: lib/ 文件改动时自动重启
480
+ Common options:
481
+ --samples <path> Sample file (default: eval-samples.json)
482
+ --skill-dir <path> Skill directory (default: skills)
483
+ --control <expr> Control variant
484
+ --treatment <v1,v2> Treatment variants, comma-separated
485
+ --config <path> eval.yaml / JSON config
486
+ --executor <name> Executor: claude / claude-sdk / codex / openai / gemini / custom
487
+ --model <name> Task execution model (default: sonnet)
488
+ --judge-models <list> Judge config, e.g. claude:haiku or claude:opus,openai:gpt-4o
489
+ --dry-run Preview tasks without model calls
490
+ --batch Batch evaluation: each skill independently against baseline
491
+ --bootstrap Enable bootstrap CI explicitly; omk eval turns it on by default
492
+ --bootstrap-samples <n> Bootstrap resamples (default: 1000)
493
+ --threshold <number> Three-layer gate threshold (default: 3.5)
494
+ --trivial-diff <number> Practically negligible diff (default: 0.1)
495
+ --report-only / --no-gate Produce the report and print verdict, but always exit 0
496
+ --no-serve Do not auto-start report server after evaluation
492
497
 
493
- bench gen-samples 选项:
494
- --batch 为所有还没 eval-samples 的 skill 生成
495
- --count <n> 每个 skill 生成多少条用例 (默认: 5)
496
- --model <name> 生成用的模型 (默认: sonnet)
497
- --skill-dir <path> skill 目录 (默认: skills), 配合 --batch 用
498
+ Examples:
499
+ omk eval --control baseline --treatment my-skill # Single-skill necessity test (baseline is a reserved variant — "no skill injected")
500
+ omk eval --control code-review-v1 --treatment code-review-v2 # Multi-variant A/B
501
+ omk eval --config eval.yaml
502
+ omk eval gold compare v1-vs-v2-20260505-1200 --gold-dir gold-dataset
503
+ `,
504
+ },
505
+ 'cli.help.eval_gold': {
506
+ zh: `
507
+ omk eval gold — 管理 human-gold 标注集
498
508
 
499
- analyze 选项:
500
- <dir> 输入: cc session JSONL 文件 / 目录
501
- (例如 ~/.claude/projects/<slug>)
502
- --kb <path> 知识库根路径 (默认: 从 trace cwd 自动推断)
503
- --last <duration> 时间窗口, 例如 "7d" / "30d" (默认: 全部)
504
- --from <iso> 窗口起点 (ISO8601), 优先级高于 --last
505
- --to <iso> 窗口终点 (ISO8601), 优先级高于 --last
506
- --skills <n1,n2,...> 白名单要分析的 skill (默认: 全部)
507
- --output-dir <path> 输出目录 (默认: ~/.oh-my-knowledge/analyses/)
509
+ 用法:
510
+ omk eval gold init [--out <dir>] [--annotator <name>]
511
+ omk eval gold validate <dir>
512
+ omk eval gold compare <reportId> --gold-dir <dir>
508
513
 
509
- bench evolve 选项:
510
- --rounds <n> 最大演化轮数 (默认: 5)
511
- --target <score> 达到该分数即提前停止
512
- --samples <path> 用例文件 (默认: eval-samples.json)
513
- --model <name> 任务执行模型 (默认: sonnet)
514
- --judge-models <executor:model> 评委 (默认: claude:haiku, evolve 仅支持单评委)
515
- --improve-model <name> 生成改进版的模型 (默认: sonnet)
516
- --concurrency <n> 并发评测任务数 (默认: 1)
517
- --timeout <seconds> 单任务执行超时 (秒, 默认: 120)
518
- --executor <name> 执行器 (默认: claude)
514
+ 选项:
515
+ --reports-dir <path> 报告目录(compare 使用,默认:~/.oh-my-knowledge/reports)
516
+ --variant <name> 指定 report 中要对比的 variant
517
+ --bootstrap-samples <n> bootstrap 重采样次数(compare 使用)
518
+ `,
519
+ en: `
520
+ omk eval gold — manage human-gold annotation datasets
519
521
 
520
- 通用选项:
521
- --lang <zh|en> CLI 输出语言 (默认: zh, 也可设 OMK_LANG 环境变量)
522
+ Usage:
523
+ omk eval gold init [--out <dir>] [--annotator <name>]
524
+ omk eval gold validate <dir>
525
+ omk eval gold compare <reportId> --gold-dir <dir>
522
526
 
523
- 示例:
524
- omk bench run --control v1 --treatment v2
525
- omk bench run --control baseline --treatment my-skill
526
- omk bench run --control git:my-skill --treatment my-skill
527
- omk bench run --control ./old-skill.md --treatment ./new-skill.md
528
- omk bench run --control baseline --treatment v1,v2,v3
529
- omk bench run --config eval.yaml
530
- omk bench run --config eval.yaml --model sonnet-4.6 # CLI 覆盖配置
531
- omk bench run --batch
532
- omk bench run --dry-run
533
- omk bench report --port 8080
534
- omk bench report --export v1-vs-v2-20260326-1832
535
- omk bench init my-eval
536
- omk bench gen-samples skills/my-skill.md
527
+ Options:
528
+ --reports-dir <path> Reports directory for compare (default: ~/.oh-my-knowledge/reports)
529
+ --variant <name> Variant in the report to compare
530
+ --bootstrap-samples <n> Bootstrap resamples for compare
531
+ `,
532
+ },
533
+ 'cli.help.observe': {
534
+ zh: `
535
+ omk observe — 分析真实 session trace,生成 skill 健康度日报
536
+
537
+ 用法:
538
+ omk observe <sessions-dir> [options]
539
+
540
+ 选项:
541
+ --kb <path> 知识库根路径(默认:从 trace cwd 推断)
542
+ --last <duration> 时间窗口,例如 7d / 24h / 30m
543
+ --from <iso> 窗口起点,优先级高于 --last
544
+ --to <iso> 窗口终点,优先级高于 --last
545
+ --skills <n1,n2,...> 只分析指定 skill
546
+ --output-dir <path> 输出目录(默认:~/.oh-my-knowledge/analyses)
537
547
  `,
538
548
  en: `
539
- oh-my-knowledge — Knowledge artifact evaluation toolkit
549
+ omk observe — analyze production session traces and produce skill health reports
540
550
 
541
551
  Usage:
542
- omk bench run [options] Run an evaluation
543
- omk bench report [options] Start the report server
544
- omk bench gate [options] Run evaluation + apply gate, exit 0/1 (for CI/CD)
545
- omk bench init [dir] Scaffold a new eval project
546
- omk bench gen-samples [skill] Generate eval-samples from skill content
547
- omk bench diff <id1> <id2> Compare two evaluation reports
548
- omk bench evolve <skill> Self-improve a skill through iterative evaluation
552
+ omk observe <sessions-dir> [options]
549
553
 
550
- omk doctor [path] skill health check (pre-eval gate)
551
- omk analyze <dir> Analyze cc session trace(s), produce skill health report (v0.18)
554
+ Options:
555
+ --kb <path> Knowledge base root (default: infer from trace cwd)
556
+ --last <duration> Time window, e.g. 7d / 24h / 30m
557
+ --from <iso> Window start, overrides --last
558
+ --to <iso> Window end, overrides --last
559
+ --skills <n1,n2,...> Only analyze selected skills
560
+ --output-dir <path> Output directory (default: ~/.oh-my-knowledge/analyses)
561
+ `,
562
+ },
563
+ 'cli.help.evolve': {
564
+ zh: `
565
+ omk evolve — 多轮自动迭代改进 skill
566
+
567
+ 用法:
568
+ omk evolve <skill-path> [options]
569
+
570
+ 选项:
571
+ --rounds <n> 迭代轮数(默认:5)
572
+ --target <score> 目标分数
573
+ --model <name> 任务执行模型,每轮跑 eval samples 的被测模型(默认:sonnet)
574
+ --improve-model <name> skill 改写模型,每轮根据反馈改写 skill 的模型(默认:sonnet)
575
+ --judge-models <executor:model> 单评委配置(默认:claude:haiku)
576
+
577
+ 示例:
578
+ omk evolve skills/code-review/SKILL.md
579
+ omk evolve skills/code-review/SKILL.md --rounds 10 --target 4.5
580
+ omk evolve skills/code-review/SKILL.md --model sonnet --improve-model opus
581
+ `,
582
+ en: `
583
+ omk evolve — auto-iterate a skill through multi-round evaluation loops
552
584
 
553
- Options for "bench run":
585
+ Usage:
586
+ omk evolve <skill-path> [options]
554
587
 
555
- --samples <path> Sample file (default: eval-samples.json)
556
- --skill-dir <path> Skill definitions directory (default: skills)
557
- --control <expr> Control-group variant expression (experiment role = control)
558
- --treatment <v1,v2> Treatment-group variant expressions (comma-separated; role = treatment)
559
- Each variant expression resolves to an artifact and optional runtime context:
560
- "baseline" — bare model, no artifact injected
561
- "git:name" — artifact from last commit
562
- "git:ref:name" — artifact from specific commit
563
- path with "/" — artifact from file directly (e.g. ./v1.md)
564
- "name@/cwd" — attach runtime context / cwd
565
- At least one of --control / --treatment must be provided.
566
- --config <path> YAML/JSON config file (evaluation-as-code).
567
- Declares samples + variants + model + executor in one file.
568
- CLI flags override config fields when both are provided.
569
- Relative paths inside the config are resolved against its directory.
570
- --model <name> Task execution model (default: sonnet)
571
- --output-dir <path> Report output directory (default: ~/.oh-my-knowledge/reports/)
572
- --no-judge Skip LLM judging
573
- --no-cache Disable result caching
574
- --dry-run Preview tasks without executing
575
- --blind Blind A/B mode: hide variant names in report
576
- --concurrency <n> Number of parallel tasks (default: 1)
577
- --timeout <seconds> Executor timeout per task in seconds (default: 120)
578
- --repeat <n> Run evaluation N times for variance analysis (default: 1)
579
- --judge-repeat <n> Call LLM judge N times per (sample × dimension) for self-
580
- consistency (default: 1). High stddev across runs = the
581
- judge is unstable on this rubric and the score is noisy.
582
- --judge-models <list> Judge configuration. Comma-separated executor:model pairs,
583
- e.g. claude:haiku or claude:opus,openai:gpt-4o.
584
- 1 entry = single judge (default claude:haiku); ≥ 2 entries
585
- = ensemble — every judge scores each (sample × dimension);
586
- the report includes per-judge breakdown + Pearson/MAD
587
- inter-judge agreement, which refutes "Claude judges Claude
588
- same-modality bias" critique. Combines with --judge-repeat.
589
- Cost ~ N_judges × N_repeat × N_samples.
590
- --bootstrap Compute bootstrap confidence intervals (distribution-free,
591
- preferred over t-interval for ordinal LLM scores). Adds
592
- per-variant CI on the mean + pairwise CI on treatment-vs-
593
- control difference (significant=0 outside CI). Reports both
594
- t-interval and bootstrap so old tooling still works.
595
- --bootstrap-samples <n> Number of bootstrap resamples (default 1000). N>10000
596
- triggers a stderr warning about runtime cost.
597
- --retry <n> Retry failed tasks up to N times with exponential backoff (default: 0)
598
- --resume <report-id> Resume from a previous report, skipping completed tasks
599
- --executor <name> Executor: claude, claude-sdk, codex, openai, gemini,
600
- anthropic-api, openai-api, or any shell command (e.g. "python my_provider.py")
601
- --batch Batch evaluation: each skill independently against baseline
602
- Requires {name}.eval-samples.json paired with each skill
603
- --skip-connectivity Skip LLM model connectivity check (auto-skipped when --resume)
604
- --mcp-config <path> MCP config file for URL fetching via MCP servers
605
- (default: .mcp.json in current directory)
606
- --no-serve Skip auto-starting report server after evaluation
607
- --verbose Print detailed progress for each sample (exec result, grading phases)
608
- --layered-stats Expand the three-layer (fact/behavior/judge) independent
609
- significance breakdown in the HTML report by default.
610
- Without this flag, the breakdown is collapsed behind a
611
- click-to-expand summary under each comparison.
612
- --strict-baseline (default ON) Isolate skill auto-discovery + Skill tool
613
- use for baseline-kind variants. Cuts the ~/.claude/skills/
614
- contamination path so skill evaluations have valid
615
- construct validity. Explicit eval.yaml allowedSkills
616
- takes precedence.
617
- --no-strict-baseline Explicitly turn strict-baseline OFF (baseline sees all
618
- auto-discovered skills). Use only in narrow scenarios
619
- (e.g. measuring how much a skill doc adds on top of
620
- full default discovery). Pre-flight emits a stderr
621
- warning when this flag is set, because
622
- verdict / Δ are vulnerable to skill contamination.
588
+ Options:
589
+ --rounds <n> Iteration rounds (default: 5)
590
+ --target <score> Target score
591
+ --model <name> Task executor model — runs eval samples each round (default: sonnet)
592
+ --improve-model <name> Skill rewriter model — rewrites the skill each round (default: sonnet)
593
+ --judge-models <executor:model> Single judge config (default: claude:haiku)
623
594
 
624
- Options for "bench gate":
625
- (same as "bench run", plus:)
626
- --threshold <number> Three-layer gate threshold (fact / behavior / LLM judge),
627
- applied INDEPENDENTLY to each layer. ANY layer below
628
- threshold fails the gate — prevents composite averaging
629
- from masking a single-layer collapse. Default: 3.5.
630
- If all three layers are absent (no
631
- assertions and no rubric defined in eval-samples), the
632
- gate FAILS with a configuration hint — no composite fallback.
633
- --trivial-diff <num> Smallest diff to treat as practically meaningful
634
- (default 0.1). Bootstrap diff CI may be statistically
635
- significant but with |Δ| < this value, treated as
636
- CAUTIOUS rather than PROGRESS.
595
+ Examples:
596
+ omk evolve skills/code-review/SKILL.md
597
+ omk evolve skills/code-review/SKILL.md --rounds 10 --target 4.5
598
+ omk evolve skills/code-review/SKILL.md --model sonnet --improve-model opus
599
+ `,
600
+ },
601
+ 'cli.help.sample': {
602
+ zh: `
603
+ omk sample — 生成或补齐 eval-samples 评测用例
637
604
 
638
- Internally = bench run + bench verdict. Exit code aligns with bench verdict:
639
- PROGRESS / SOLO-PASS → 0; NOISE / UNDERPOWERED / CAUTIOUS / REGRESS → 1.
640
- Underpowered runs fail directly — closes the "single-run PASS = deploy" loophole.
605
+ 用法:
606
+ omk sample <skill-path> [options]
607
+ omk sample --batch [--skill-dir <dir>] [options]
641
608
 
642
- Options for "bench report":
643
- --port <number> Server port (default: 7799)
644
- --reports-dir <path> Reports directory (default: ~/.oh-my-knowledge/reports/)
645
- --export <id> Export report as standalone HTML file
646
- --dev Dev mode: auto-restart on lib/ file changes
609
+ 选项:
610
+ --count <n> 生成用例数量(默认:5)
611
+ --model <name> 生成模型(默认:sonnet)
612
+ --batch 为 skill 目录下缺少 eval-samples 的 skill 批量生成
613
+ --skill-dir <path> skill 目录(batch 使用,默认:skills)
614
+ `,
615
+ en: `
616
+ omk sample — generate or fill eval-samples test cases
647
617
 
648
- Options for "bench gen-samples":
649
- --batch Generate for all skills missing eval-samples
650
- --count <n> Number of samples to generate per skill (default: 5)
651
- --model <name> Model for generation (default: sonnet)
652
- --skill-dir <path> Skill directory (default: skills), used with --batch
618
+ Usage:
619
+ omk sample <skill-path> [options]
620
+ omk sample --batch [--skill-dir <dir>] [options]
653
621
 
654
- Options for "analyze":
655
- <dir> Input: cc session JSONL file / dir (e.g. ~/.claude/projects/<slug>)
656
- --kb <path> Knowledge base root (default: auto-infer from trace cwd)
657
- --last <duration> Time window like "7d" / "30d" (default: all)
658
- --from <iso> Window start (ISO8601), takes precedence over --last
659
- --to <iso> Window end (ISO8601), takes precedence over --last
660
- --skills <n1,n2,...> Whitelist skills to analyze (default: all)
661
- --output-dir <path> Output dir (default: ~/.oh-my-knowledge/analyses/)
622
+ Options:
623
+ --count <n> Number of test cases to generate (default: 5)
624
+ --model <name> Generation model (default: sonnet)
625
+ --batch Generate for skills that are missing eval-samples
626
+ --skill-dir <path> Skill directory for batch mode (default: skills)
627
+ `,
628
+ },
629
+ 'cli.help.studio': {
630
+ zh: `
631
+ omk studio — 打开本地知识工作台
632
+
633
+ 用法:
634
+ omk studio [options]
635
+
636
+ 选项:
637
+ --port <n> 本地服务端口(默认:7799)
638
+ --reports-dir <path> 报告目录(默认:~/.oh-my-knowledge/reports)
639
+ --analyses-dir <path> 观测分析目录
640
+ --no-open 只启动服务,不自动打开浏览器
641
+ --dev 开发模式:文件变化时自动重启
642
+
643
+ 示例:
644
+ omk studio
645
+ omk studio --port 7798
646
+ omk studio --no-open
647
+ `,
648
+ en: `
649
+ omk studio — open the local knowledge workbench
662
650
 
663
- Options for "bench evolve":
664
- --rounds <n> Maximum evolution rounds (default: 5)
665
- --target <score> Stop early when score reaches this threshold
666
- --samples <path> Sample file (default: eval-samples.json)
667
- --model <name> Task execution model (default: sonnet)
668
- --judge-models <executor:model> Judge config (default: claude:haiku; evolve is single-judge only)
669
- --improve-model <name> Model for generating improvements (default: sonnet)
670
- --concurrency <n> Parallel eval tasks (default: 1)
671
- --timeout <seconds> Executor timeout per task in seconds (default: 120)
672
- --executor <name> Executor to use (default: claude)
651
+ Usage:
652
+ omk studio [options]
673
653
 
674
- Common options:
675
- --lang <zh|en> CLI output language (default: zh, also via OMK_LANG env)
654
+ Options:
655
+ --port <n> Local server port (default: 7799)
656
+ --reports-dir <path> Reports directory (default: ~/.oh-my-knowledge/reports)
657
+ --analyses-dir <path> Observation analyses directory
658
+ --no-open Start the server without opening a browser
659
+ --dev Dev mode: restart on file changes
676
660
 
677
661
  Examples:
678
- omk bench run --control v1 --treatment v2
679
- omk bench run --control baseline --treatment my-skill
680
- omk bench run --control git:my-skill --treatment my-skill
681
- omk bench run --control ./old-skill.md --treatment ./new-skill.md
682
- omk bench run --control baseline --treatment v1,v2,v3
683
- omk bench run --config eval.yaml
684
- omk bench run --config eval.yaml --model sonnet-4.6 # CLI overrides config
685
- omk bench run --batch
686
- omk bench run --dry-run
687
- omk bench report --port 8080
688
- omk bench report --export v1-vs-v2-20260326-1832
689
- omk bench init my-eval
690
- omk bench gen-samples skills/my-skill.md
662
+ omk studio
663
+ omk studio --port 7798
664
+ omk studio --no-open
691
665
  `,
692
666
  },
693
- 'cli.help.diff_usage': {
694
- zh: [
695
- '用法:',
696
- ' omk bench diff <reportId> 单 report 内 sample 级 diff',
697
- ' omk bench diff <reportId1> <reportId2> 跨 report variant 级 diff',
698
- '',
699
- '选项:',
700
- ' --regressions-only 只列 treatment < control 的用例',
701
- ' --threshold <num> 回退判定阈值 (默认 0, 即任何负 Δ 都算回退)',
702
- ' --variant <name> within-report 模式下指定要钻取的 variant (默认: variants[1])',
703
- ' --top <n> 只列差距最大的前 N 个用例',
704
- ].join('\n'),
705
- en: [
706
- 'Usage:',
707
- ' omk bench diff <reportId> within-report per-sample diff',
708
- ' omk bench diff <reportId1> <reportId2> cross-report variant-level diff',
709
- '',
710
- 'Options:',
711
- ' --regressions-only show only samples where treatment < control',
712
- ' --threshold <num> regression threshold (default 0, any negative Δ counts)',
713
- ' --variant <name> within-report mode: which variant to drill (default: variants[1])',
714
- ' --top <n> only show top N samples by absolute diff',
715
- ].join('\n'),
716
- },
717
- 'cli.help.analyze_usage': {
718
- zh: '用法: omk analyze <dir> [--kb <path>] [--last 7d] [--from ISO] [--to ISO] [--skills name1,name2]',
719
- en: 'Usage: omk analyze <dir> [--kb <path>] [--last 7d] [--from ISO] [--to ISO] [--skills name1,name2]',
720
- },
721
- 'cli.help.gold': {
722
- zh: [
723
- '',
724
- '用法: omk bench gold <subcommand>',
725
- '',
726
- '子命令:',
727
- ' init [--out <dir>] [--annotator <id>] 生成空白 gold dataset 模板',
728
- ' validate <dir> 校验数据集结构',
729
- ' compare <reportId> --gold-dir <dir> 与已有 report 计算 α / κ / Pearson',
730
- ' [--variant <name>] [--reports-dir <d>]',
731
- ' [--bootstrap-samples N] [--seed N]',
732
- '',
733
- ].join('\n'),
734
- en: [
735
- '',
736
- 'Usage: omk bench gold <subcommand>',
737
- '',
738
- 'Subcommands:',
739
- ' init [--out <dir>] [--annotator <id>] create a blank gold dataset template',
740
- ' validate <dir> validate dataset structure',
741
- ' compare <reportId> --gold-dir <dir> compute α / κ / Pearson against an existing report',
742
- ' [--variant <name>] [--reports-dir <d>]',
743
- ' [--bootstrap-samples N] [--seed N]',
744
- '',
745
- ].join('\n'),
746
- },
747
- 'cli.help.debias_validate': {
748
- zh: [
749
- '',
750
- '用法: omk bench debias-validate <kind> <reportId> [options]',
751
- '',
752
- '类别:',
753
- ' length 用相反的长度去偏设置重新评判, 并对分数差出 bootstrap CI。',
754
- ' judge 成本约为原评判的两倍。',
755
- '',
756
- '选项:',
757
- ' --reports-dir <dir> 报告存储目录 (默认: ~/.oh-my-knowledge/reports)',
758
- ' --samples <path> 覆盖用例文件 (默认: 从 report.meta.request 读)',
759
- ' --variant <name> 校验哪个 variant (默认: 第一个)',
760
- ' --judge-models <executor:model> 评委 (默认: 沿用 report.meta.judgeModels[0]; debias-validate 仅支持单评委)',
761
- ' --bootstrap-samples N bootstrap 迭代次数 (默认 1000)',
762
- ' --seed N 固定 CI 随机种子',
763
- '',
764
- ].join('\n'),
765
- en: [
766
- '',
767
- 'Usage: omk bench debias-validate <kind> <reportId> [options]',
768
- '',
769
- 'Kinds:',
770
- ' length re-judge with the opposite length-debias setting and bootstrap CI',
771
- ' on the score diff. Cost ~doubles vs the original judge pass.',
772
- '',
773
- 'Options:',
774
- ' --reports-dir <dir> report store dir (default: ~/.oh-my-knowledge/reports)',
775
- ' --samples <path> override samples file (default: from report.meta.request)',
776
- ' --variant <name> which variant to validate (default: first)',
777
- ' --judge-models <executor:model> Judge (default: from report.meta.judgeModels[0]; debias-validate is single-judge only)',
778
- ' --bootstrap-samples N bootstrap iterations (default 1000)',
779
- ' --seed N deterministic CI seed',
780
- '',
781
- ].join('\n'),
782
- },
783
- 'cli.help.saturation': {
784
- zh: [
785
- '',
786
- '用法: omk bench saturation <reportId> [options]',
787
- '',
788
- '回答 "我跑够用例了吗?"。复述已有 report 中持久化的饱和判定。',
789
- '',
790
- '注: 本命令读取 run 时跑出的 verdict (运行时已用 method=bootstrap-ci-width',
791
- '默认阈值 + 3 窗口持续条件)。如要换 method/threshold 重新计算, 需要重跑',
792
- '`omk bench run --repeat ≥ 5` (运行时持久化的 trace 不含原始分数, 无法',
793
- '在事后用其他参数复算)。',
794
- '',
795
- '选项:',
796
- ' --reports-dir <dir> 报告存储目录 (默认: ~/.oh-my-knowledge/reports)',
797
- ' --variant <name> 只看一个 variant (默认: 全部)',
798
- '',
799
- ].join('\n'),
800
- en: [
801
- '',
802
- 'Usage: omk bench saturation <reportId> [options]',
803
- '',
804
- 'Answers "do I have enough samples?". Replays the saturation verdict',
805
- 'persisted in an existing report.',
806
- '',
807
- 'Note: this command reads the verdict computed at run time (which used',
808
- 'method=bootstrap-ci-width with default threshold + 3-window sustained',
809
- 'condition). To re-compute with a different method/threshold, re-run',
810
- '`omk bench run --repeat ≥ 5` (the persisted trace does not include raw',
811
- 'scores, so post-hoc parameter sweeps are not possible here).',
812
- '',
813
- 'Options:',
814
- ' --reports-dir <dir> report store dir (default: ~/.oh-my-knowledge/reports)',
815
- ' --variant <name> only show one variant (default: all)',
816
- '',
817
- ].join('\n'),
818
- },
819
- 'cli.help.verdict': {
820
- zh: [
821
- '',
822
- '用法: omk bench verdict <reportId> [options]',
823
- '',
824
- '聚合 bootstrap CI / 三层 ci-gate / saturation / human α, 给出一行结论。',
825
- '',
826
- 'Verdict 等级:',
827
- ' PROGRESS 显著改进 + 三层全过',
828
- ' CAUTIOUS 改进真实但有警告 (gate 破 / 幅度太小 / 控制组本身崩)',
829
- ' REGRESS 显著回退 — 不要 ship',
830
- ' NOISE CI 跨 0, 无法判定',
831
- ' UNDERPOWERED 用例不足, 需要扩 N 重测',
832
- ' SOLO 单 variant 报告, 没有对比对象',
833
- '',
834
- '选项:',
835
- ' --reports-dir <dir> 报告存储目录 (默认: ~/.oh-my-knowledge/reports)',
836
- ' --threshold <num> 三层 gate 阈值 (默认 3.5, 与 omk bench gate 对齐)',
837
- ' --trivial-diff <num> "幅度太小" 阈值 (默认 0.1)',
838
- ' --verbose 展开每个 pair 的详情',
839
- '',
840
- ].join('\n'),
841
- en: [
842
- '',
843
- 'Usage: omk bench verdict <reportId> [options]',
844
- '',
845
- 'Aggregates bootstrap CI / 3-layer ci-gate / saturation / human α into a one-line verdict.',
846
- '',
847
- 'Verdict levels:',
848
- ' PROGRESS significant improvement + all 3 layers pass',
849
- ' CAUTIOUS real improvement but with warnings (gate fails / diff too small / control collapsed)',
850
- ' REGRESS significant regression — do not ship',
851
- ' NOISE CI crosses 0, no verdict',
852
- ' UNDERPOWERED not enough samples, expand N and re-run',
853
- ' SOLO single-variant report, nothing to compare against',
854
- '',
855
- 'Options:',
856
- ' --reports-dir <dir> report store dir (default: ~/.oh-my-knowledge/reports)',
857
- ' --threshold <num> 3-layer gate threshold (default 3.5, matches omk bench gate)',
858
- ' --trivial-diff <num> "diff too small" threshold (default 0.1)',
859
- ' --verbose expand per-pair details',
860
- '',
861
- ].join('\n'),
862
- },
863
- 'cli.help.diagnose': {
864
- zh: [
865
- '',
866
- '用法: omk bench diagnose <reportId> [options]',
867
- '',
868
- '诊断用例集本身的质量问题: 区分度低 / 重复 / 歧义 / 成本异常 / 全 fail。',
869
- '回答 "测评结论是否被坏用例污染" — 与 omk bench verdict 互补。',
870
- '',
871
- '选项:',
872
- ' --reports-dir <dir> 报告存储目录',
873
- ' --samples <path> 用例文件路径 (用于 near-duplicate 检测; 默认从 report.meta.request 读)',
874
- ' --top <n> 每类只显示前 N 个 (默认 10, 0=全部)',
875
- ' --duplicate-rouge <num> near-duplicate ROUGE-1 阈值 (默认 0.7)',
876
- ' --ambiguous-stddev <num> 歧义阈值, judge stddev (默认 1.0, 需要 --judge-repeat ≥ 2 数据)',
877
- ' --cost-k <num> 成本异常倍数 vs 中位数 (默认 3)',
878
- ' --latency-k <num> 耗时异常倍数 vs 中位数 (默认 3)',
879
- ' --flat <num> flat_scores 分差阈值 (默认 0.5)',
880
- '',
881
- ].join('\n'),
882
- en: [
883
- '',
884
- 'Usage: omk bench diagnose <reportId> [options]',
885
- '',
886
- 'Diagnose quality issues in the sample set itself: low discrimination /',
887
- 'duplicates / ambiguity / cost anomalies / all-fail. Answers "is the verdict',
888
- 'tainted by bad samples?" — complements omk bench verdict.',
889
- '',
890
- 'Options:',
891
- ' --reports-dir <dir> report store dir',
892
- ' --samples <path> sample file path (for near-duplicate detection; defaults to report.meta.request)',
893
- ' --top <n> top N per category (default 10, 0=all)',
894
- ' --duplicate-rouge <num> near-duplicate ROUGE-1 threshold (default 0.7)',
895
- ' --ambiguous-stddev <num> ambiguity threshold, judge stddev (default 1.0, requires --judge-repeat ≥ 2)',
896
- ' --cost-k <num> cost-outlier multiplier vs median (default 3)',
897
- ' --latency-k <num> latency-outlier multiplier vs median (default 3)',
898
- ' --flat <num> flat_scores spread threshold (default 0.5)',
899
- '',
900
- ].join('\n'),
901
- },
902
- 'cli.help.failures': {
903
- zh: [
904
- '',
905
- '用法: omk bench failures <reportId> [options]',
906
- '',
907
- '把已有 report 的失败用例喂给一次 LLM 调用, 自动聚类并给出修复建议。',
908
- '失败定义: compositeScore < threshold 或 ok=false。',
909
- '',
910
- '选项:',
911
- ' --reports-dir <dir> 报告存储目录',
912
- ' --judge-models <executor:model> 评委 (默认: 沿用 report.meta.judgeModels[0]; failures 仅支持单评委)',
913
- ' --max-clusters <n> 最多聚成几类 (默认 5)',
914
- ' --threshold <num> compositeScore < threshold 算失败 (默认 3)',
915
- ' --max-feed <n> 最多喂给 LLM 多少条 (默认 50, 超出取最差)',
916
- '',
917
- ].join('\n'),
918
- en: [
919
- '',
920
- 'Usage: omk bench failures <reportId> [options]',
921
- '',
922
- 'Feed failing samples from an existing report to a single LLM call, auto-cluster',
923
- 'them, and produce per-cluster fix suggestions.',
924
- 'Failure definition: compositeScore < threshold or ok=false.',
925
- '',
926
- 'Options:',
927
- ' --reports-dir <dir> report store dir',
928
- ' --judge-models <executor:model> Judge (default: from report.meta.judgeModels[0]; failures is single-judge only)',
929
- ' --max-clusters <n> max number of clusters (default 5)',
930
- ' --threshold <num> compositeScore < threshold counts as failure (default 3)',
931
- ' --max-feed <n> max samples to feed the LLM (default 50, takes the worst)',
932
- '',
933
- ].join('\n'),
934
- },
935
- // sample design coverage block strings
936
- 'cli.diagnose.coverage_header': {
937
- zh: '用例设计覆盖度 (Sample design coverage):',
938
- en: 'Sample design coverage:',
939
- },
940
- 'cli.diagnose.coverage_unspecified': {
941
- zh: '(未声明)',
942
- en: '(unspecified)',
943
- },
944
- 'cli.diagnose.coverage_chars': {
945
- zh: '字符',
946
- en: 'chars',
947
- },
948
- 'cli.diagnose.coverage_hint_empty': {
949
- zh: 'ℹ 该用例集未声明任何 capability / difficulty / construct / provenance 元数据。详见 docs/sample-design-spec.md',
950
- en: 'ℹ No samples in this set declare capability / difficulty / construct / provenance metadata. See docs/sample-design-spec.md',
951
- },
952
- 'cli.diagnose.coverage_declared': {
953
- zh: '声明',
954
- en: 'declared',
955
- },
956
667
  // ============ omk doctor 健康检查 ============
957
668
  'cli.doctor.rule.skill_readable': {
958
669
  zh: 'skill 文件可读',
@@ -970,6 +681,67 @@ Examples:
970
681
  zh: '用例 ↔ skill 输入约定',
971
682
  en: 'samples ↔ skill contract',
972
683
  },
684
+ 'cli.doctor.rule.skill_health_check': {
685
+ zh: '健康度体检',
686
+ en: 'Health check',
687
+ },
688
+ // ============ skill_health composer (CLI default; --static-only disables it) ============
689
+ 'cli.doctor.health.skipped': {
690
+ zh: '健康度体检已跳过(runHealthCheck=false)',
691
+ en: 'health check skipped (runHealthCheck=false)',
692
+ },
693
+ 'cli.doctor.health.no_dimensions': {
694
+ zh: '没有注册任何健康度维度,跳过',
695
+ en: 'no health dimensions registered, skipped',
696
+ },
697
+ 'cli.doctor.health.fail.executor': {
698
+ zh: 'LLM 调用失败: {error}',
699
+ en: 'LLM call failed: {error}',
700
+ },
701
+ 'cli.doctor.health.fail.parse': {
702
+ zh: 'LLM 输出解析失败: {error}',
703
+ en: 'failed to parse LLM output: {error}',
704
+ },
705
+ 'cli.doctor.health.fail.empty_output': {
706
+ zh: 'LLM 返回了空输出',
707
+ en: 'LLM returned empty output',
708
+ },
709
+ 'cli.doctor.health.hint.executor': {
710
+ zh: '检查 executor 配置(--executor / --model)与网络连通,或调大 --timeout',
711
+ en: 'Verify executor config (--executor / --model) and connectivity, or raise --timeout',
712
+ },
713
+ 'cli.doctor.health.hint.parse': {
714
+ zh: 'LLM 没返回合法 JSON;原文存在 detail.rawOutput 截断片段,可重跑或换 model',
715
+ en: 'LLM did not return valid JSON; raw snippet stored in detail.rawOutput. Re-run or switch model',
716
+ },
717
+ 'cli.doctor.health.dim.message': {
718
+ zh: '{level}: 错误 {err}/警告 {warn}/建议 {sug}',
719
+ en: '{level}: error {err}/warn {warn}/suggest {sug}',
720
+ },
721
+ 'cli.doctor.health.dim.missing': {
722
+ zh: 'LLM 未输出此维度({dim}),已置不适用',
723
+ en: 'LLM omitted dimension ({dim}); treated as N/A',
724
+ },
725
+ 'cli.doctor.health.summary.label': {
726
+ zh: '健康度总览',
727
+ en: 'Health summary',
728
+ },
729
+ 'cli.doctor.health.summary.message': {
730
+ zh: '{overall} | 维度: 健康 {h}/亚健康 {sh}/不健康 {bad}/不适用 {na} | finding: 错误 {err}/警告 {warn}/建议 {sug}',
731
+ en: '{overall} | dims: healthy {h}/sub {sh}/unhealthy {bad}/n-a {na} | findings: err {err}/warn {warn}/sug {sug}',
732
+ },
733
+ 'cli.doctor.health.summary.no_top': {
734
+ zh: '完整详情见 --json 输出或 --html 报告',
735
+ en: 'Full detail in --json output or --html report',
736
+ },
737
+ // 7 内置维度 labelKey (id-based)
738
+ 'cli.doctor.health.dim.trigger-boundary': { zh: '触发与边界', en: 'Trigger & boundary' },
739
+ 'cli.doctor.health.dim.doc-clarity': { zh: '文档清晰', en: 'Documentation clarity' },
740
+ 'cli.doctor.health.dim.instr-precision': { zh: '指令精确性', en: 'Instruction precision' },
741
+ 'cli.doctor.health.dim.dependency': { zh: '依赖检查', en: 'Dependency check' },
742
+ 'cli.doctor.health.dim.tool-conventions': { zh: '工具规范', en: 'Tool conventions' },
743
+ 'cli.doctor.health.dim.security': { zh: '安全与合规', en: 'Security & compliance' },
744
+ 'cli.doctor.health.dim.examples': { zh: '示例完备', en: 'Example completeness' },
973
745
  // pass
974
746
  'cli.doctor.skill_readable.pass': {
975
747
  zh: 'skill 内容长度 {length} 字符',
@@ -1084,10 +856,10 @@ Examples:
1084
856
  // ============ omk doctor CLI level ============
1085
857
  'cli.help.doctor_usage': {
1086
858
  zh: `
1087
- oh-my-knowledge — omk doctor 健康检查
859
+ oh-my-knowledge — omk doctor 健康度体检 (LLM-judge)
1088
860
 
1089
861
  用法:
1090
- omk doctor [path] 在 path 上跑评测前置健康检查
862
+ omk doctor [path] 在 path 上跑深度健康度体检
1091
863
  omk doctor 在当前目录(或 ./skills)批量跑
1092
864
 
1093
865
  参数:
@@ -1095,66 +867,80 @@ oh-my-knowledge — omk doctor 健康检查
1095
867
 
1096
868
  选项:
1097
869
  --json 把 DoctorReport 打到 stdout(CI 消费用)
1098
- --gate 静默模式: 通过 exit 0 / 不通过 exit 1, 仅 stderr 出问题摘要
1099
- --executor <name> executor 名(仅向后兼容, doctor 不直接打 LLM)
1100
- --model <name> model 名(同上)
1101
- --timeout <seconds> rule 执行超时(默认 8)
870
+ --gate 静默模式: fatal 问题 exit 1; warnings_only 仍 exit 0, 仅 stderr 出摘要
871
+ --executor <name> LLM executor (默认 claude, 可换 anthropic-api/codex 等)
872
+ --model <name> 模型 (默认 sonnet)
873
+ --samples <path> 显式指定评测用例文件
874
+ --timeout <seconds> 单次 LLM 会话超时 (默认 600)
875
+ --html <path> 产出可视化 HTML 报告到 <path> (可与 --json 同时用)
876
+ --static-only 离线模式: 只跑静态检查 (skill 可读性 / 元数据 / 依赖 / samples 契约), 不调 LLM
1102
877
  --lang <zh|en> 切换输出语言
1103
878
 
1104
879
  示例:
1105
- omk doctor examples/code-review/skills/v1.md
1106
- omk doctor examples/code-review/skills --json | jq .outcome # passed | warnings_only | failed
1107
- omk doctor --gate; echo $?
1108
-
1109
- doctor 检查项(纯静态 / 零 LLM 调用):
1110
- - skill 文件可读 + 内容有最小长度
1111
- - skill 元数据合法 (front-matter 若有)
1112
- - 前置依赖完整 (引用的 CLI 工具 / 文件 / 环境变量 / preflight 命令)
1113
- - 用例 ↔ skill 输入约定 (warn 级, 仅传 samples 时跑)
1114
-
1115
- executor / judge 连通性由 evaluation preflight 负责, 不在 doctor 范围内。
1116
- omk bench run / omk bench gate 内置 doctor 强制门禁, 不可 skip — 静态检查
1117
- 零成本无理由跳过。LLM 连通性可用 --skip-connectivity 跳过 (--resume 时自动)。
880
+ omk doctor my-skill --html /tmp/report.html # 深度体检 + HTML 报告 (默认)
881
+ omk doctor examples/code-review/skills --json > r.json # JSON 给 CI / 外部工具消费
882
+ omk doctor --gate; echo $? # CI 模式: fatal 问题 exit 1, 警告不阻断
883
+ omk doctor --static-only # 无 LLM 环境 (CI / 断网) 跑纯静态检查
884
+
885
+ doctor = LLM 健康度体检 (单次 LLM 会话):
886
+ - 7 个内置维度: 触发与边界 / 文档清晰 / 指令精确性 / 依赖检查 / 工具规范 / 安全与合规 / 示例完备
887
+ - 用户可扩展: 在自己代码里 registerHealthDimension(spec) 加自定义维度,
888
+ 会自动加入同一次 LLM 调用的 prompt + 报告 (顺序 = 注册顺序)
889
+ - 每维度独立给 健康/亚健康/不健康/不适用 + findings + 改进建议
890
+ - HTML 报告: 维度按 fail→warn→pass→skipped 排, 错误 finding 排前面
891
+
892
+ 注: omk eval 内部仍跑静态 skill-readability/metadata/dependency
893
+ gate 保护评测质量, 不走 omk doctor 这条 LLM 路径 (角色分离: doctor=审计, eval=评测)。
894
+ LLM 连通性可用 omk eval --skip-connectivity 跳过 (--resume 时自动)。
1118
895
  `.trim() + '\n',
1119
896
  en: `
1120
- oh-my-knowledge — omk doctor health check
897
+ oh-my-knowledge — omk doctor health audit (LLM-judge)
1121
898
 
1122
899
  Usage:
1123
- omk doctor [path] Run pre-evaluation health check on path
1124
- omk doctor Batch check current dir (or ./skills)
900
+ omk doctor [path] Run deep LLM-based health audit on path
901
+ omk doctor Batch audit current dir (or ./skills)
1125
902
 
1126
903
  Arguments:
1127
- path A .md file, directory, or omit (= cwd). Directory mode batches all skills.
904
+ path A .md file, directory, or omit (= cwd). Directory batches all skills.
1128
905
 
1129
906
  Options:
1130
907
  --json Print DoctorReport JSON to stdout (CI-friendly)
1131
- --gate Silent mode: exit 0 if pass, exit 1 if fail; brief stderr summary only
1132
- --executor <name> executor name (kept for compat; doctor does not call LLM)
1133
- --model <name> model name (same)
1134
- --timeout <seconds> per-rule timeout (default 8)
908
+ --gate Silent mode: exit 1 only on fatal failure; warnings_only exits 0
909
+ --executor <name> LLM executor (default 'claude'; switchable to anthropic-api/codex etc)
910
+ --model <name> model name (default 'sonnet')
911
+ --samples <path> Explicit eval samples file
912
+ --timeout <seconds> LLM session timeout (default 600)
913
+ --html <path> Also write a visual HTML report to <path> (combines with --json)
914
+ --static-only Offline mode: run only static checks (readability / metadata / deps / samples contract); no LLM call
1135
915
  --lang <zh|en> Output language
1136
916
 
1137
917
  Examples:
1138
- omk doctor examples/code-review/skills/v1.md
1139
- omk doctor examples/code-review/skills --json | jq .outcome # passed | warnings_only | failed
1140
- omk doctor --gate; echo $?
1141
-
1142
- Checks (pure static / zero LLM calls):
1143
- - skill file readable + minimum content length
1144
- - skill metadata valid (front-matter if present)
1145
- - dependencies present (referenced CLI tools / files / env vars / preflight commands)
1146
- - samples ↔ skill contract (warn-level, only when samples provided)
1147
-
1148
- executor / judge connectivity is handled by evaluation preflight, not doctor.
1149
- omk bench run / omk bench gate run doctor as mandatory; no skip flag — static
1150
- checks cost nothing to run. LLM connectivity can be skipped with --skip-connectivity
1151
- (auto-skipped on --resume).
918
+ omk doctor my-skill --html /tmp/report.html # deep audit + HTML report (default)
919
+ omk doctor examples/code-review/skills --json > r.json # JSON for CI / external tools
920
+ omk doctor --gate; echo $? # CI mode: fatal failures exit 1; warnings do not block
921
+ omk doctor --static-only # offline (CI / no LLM) static checks only
922
+
923
+ doctor = LLM health audit (single LLM session):
924
+ - 7 builtin dimensions: trigger & boundary / doc clarity / instruction precision /
925
+ dependency / tool conventions / security & compliance / example completeness
926
+ - User-extensible: call registerHealthDimension(spec) in your code to add custom
927
+ dimensions; they join the same LLM call's prompt + report (order = registration order)
928
+ - Each dim graded healthy / sub-healthy / unhealthy / N-A + findings + suggestions
929
+ - HTML report: dims sorted fail→warn→pass→skipped; errors first within each dim
930
+
931
+ Note: omk eval still runs static skill-readability/metadata/dependency gates
932
+ internally (separate from this doctor command). Roles: doctor=audit, eval=evaluate.
933
+ LLM connectivity for omk eval can be skipped with --skip-connectivity (auto on --resume).
1152
934
  `.trim() + '\n',
1153
935
  },
1154
936
  'cli.doctor.no_skill_found': {
1155
937
  zh: '未在 {path} 下发现 skill 文件。\n doctor 期望 .md 文件、目录(包含 .md 或 SKILL.md)或 cwd 下的 skills/ 子目录。',
1156
938
  en: 'No skills found at {path}.\n doctor expects a .md file, a directory (containing .md or SKILL.md), or skills/ under cwd.',
1157
939
  },
940
+ 'cli.doctor.samples_detected': {
941
+ zh: '✓ 使用评测用例文件:{path}',
942
+ en: '✓ Using eval samples file: {path}',
943
+ },
1158
944
  'cli.doctor.gate_blocked': {
1159
945
  zh: 'skill 健康检查未通过, 评测已中止。doctor 是评测必经环节, 无 skip 选项 — 请修复上述问题后重跑。',
1160
946
  en: 'skill health check failed; evaluation aborted. doctor is mandatory and not skippable — fix the issues above and re-run.',