oh-my-knowledge 0.20.1 → 0.22.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (200) hide show
  1. package/README.md +10 -4
  2. package/README.zh.md +9 -4
  3. package/dist/src/analysis/coverage-analyzer.d.ts +1 -1
  4. package/dist/src/analysis/coverage-analyzer.d.ts.map +1 -1
  5. package/dist/src/analysis/failure-clusterer.d.ts +1 -1
  6. package/dist/src/analysis/failure-clusterer.d.ts.map +1 -1
  7. package/dist/src/analysis/gap-analyzer.d.ts +2 -2
  8. package/dist/src/analysis/gap-analyzer.d.ts.map +1 -1
  9. package/dist/src/analysis/gap-analyzer.js +4 -4
  10. package/dist/src/analysis/hedging-classifier.d.ts +1 -1
  11. package/dist/src/analysis/hedging-classifier.d.ts.map +1 -1
  12. package/dist/src/analysis/report-diagnostics.d.ts +1 -1
  13. package/dist/src/analysis/report-diagnostics.d.ts.map +1 -1
  14. package/dist/src/analysis/report-diagnostics.js +7 -7
  15. package/dist/src/analysis/sample-diagnostics.d.ts +1 -1
  16. package/dist/src/analysis/sample-diagnostics.d.ts.map +1 -1
  17. package/dist/src/analysis/sample-diagnostics.js +15 -15
  18. package/dist/src/authoring/evolver.d.ts +1 -1
  19. package/dist/src/authoring/evolver.d.ts.map +1 -1
  20. package/dist/src/authoring/evolver.js +5 -5
  21. package/dist/src/authoring/evolver.js.map +1 -1
  22. package/dist/src/authoring/generator.d.ts +1 -1
  23. package/dist/src/authoring/generator.d.ts.map +1 -1
  24. package/dist/src/authoring/generator.js +8 -8
  25. package/dist/src/authoring/generator.js.map +1 -1
  26. package/dist/src/cli/i18n-dict.d.ts +56 -0
  27. package/dist/src/cli/i18n-dict.d.ts.map +1 -0
  28. package/dist/src/cli/i18n-dict.js +934 -0
  29. package/dist/src/cli/i18n-dict.js.map +1 -0
  30. package/dist/src/cli/i18n.d.ts +24 -0
  31. package/dist/src/cli/i18n.d.ts.map +1 -0
  32. package/dist/src/cli/i18n.js +53 -0
  33. package/dist/src/cli/i18n.js.map +1 -0
  34. package/dist/src/cli.js +320 -413
  35. package/dist/src/cli.js.map +1 -1
  36. package/dist/src/eval-core/cache.d.ts +7 -3
  37. package/dist/src/eval-core/cache.d.ts.map +1 -1
  38. package/dist/src/eval-core/cache.js +14 -4
  39. package/dist/src/eval-core/cache.js.map +1 -1
  40. package/dist/src/eval-core/dependency-checker.d.ts +16 -1
  41. package/dist/src/eval-core/dependency-checker.d.ts.map +1 -1
  42. package/dist/src/eval-core/dependency-checker.js +79 -4
  43. package/dist/src/eval-core/dependency-checker.js.map +1 -1
  44. package/dist/src/eval-core/evaluation-execution.d.ts +3 -3
  45. package/dist/src/eval-core/evaluation-execution.d.ts.map +1 -1
  46. package/dist/src/eval-core/evaluation-execution.js +4 -2
  47. package/dist/src/eval-core/evaluation-execution.js.map +1 -1
  48. package/dist/src/eval-core/evaluation-job.d.ts +2 -2
  49. package/dist/src/eval-core/evaluation-job.d.ts.map +1 -1
  50. package/dist/src/eval-core/evaluation-reporting.d.ts +1 -1
  51. package/dist/src/eval-core/evaluation-reporting.d.ts.map +1 -1
  52. package/dist/src/eval-core/evaluation-reporting.js +4 -0
  53. package/dist/src/eval-core/evaluation-reporting.js.map +1 -1
  54. package/dist/src/eval-core/execution-strategy.d.ts +1 -1
  55. package/dist/src/eval-core/execution-strategy.d.ts.map +1 -1
  56. package/dist/src/eval-core/execution-strategy.js +35 -2
  57. package/dist/src/eval-core/execution-strategy.js.map +1 -1
  58. package/dist/src/eval-core/layer-gates.d.ts +17 -0
  59. package/dist/src/eval-core/layer-gates.d.ts.map +1 -0
  60. package/dist/src/eval-core/{ci-gates.js → layer-gates.js} +5 -5
  61. package/dist/src/eval-core/layer-gates.js.map +1 -0
  62. package/dist/src/eval-core/schema.d.ts +1 -1
  63. package/dist/src/eval-core/schema.d.ts.map +1 -1
  64. package/dist/src/eval-core/schema.js +17 -3
  65. package/dist/src/eval-core/schema.js.map +1 -1
  66. package/dist/src/eval-core/task-planner.d.ts +1 -1
  67. package/dist/src/eval-core/task-planner.d.ts.map +1 -1
  68. package/dist/src/eval-core/task-planner.js +1 -1
  69. package/dist/src/eval-core/task-planner.js.map +1 -1
  70. package/dist/src/eval-core/verdict.d.ts +5 -2
  71. package/dist/src/eval-core/verdict.d.ts.map +1 -1
  72. package/dist/src/eval-core/verdict.js +47 -8
  73. package/dist/src/eval-core/verdict.js.map +1 -1
  74. package/dist/src/eval-workflows/each-evaluation-workflow.d.ts +15 -8
  75. package/dist/src/eval-workflows/each-evaluation-workflow.d.ts.map +1 -1
  76. package/dist/src/eval-workflows/each-evaluation-workflow.js +3 -2
  77. package/dist/src/eval-workflows/each-evaluation-workflow.js.map +1 -1
  78. package/dist/src/eval-workflows/evaluation-pipeline.d.ts +33 -4
  79. package/dist/src/eval-workflows/evaluation-pipeline.d.ts.map +1 -1
  80. package/dist/src/eval-workflows/evaluation-pipeline.js +81 -2
  81. package/dist/src/eval-workflows/evaluation-pipeline.js.map +1 -1
  82. package/dist/src/eval-workflows/evaluation-preparation.d.ts +10 -6
  83. package/dist/src/eval-workflows/evaluation-preparation.d.ts.map +1 -1
  84. package/dist/src/eval-workflows/evaluation-preparation.js +2 -2
  85. package/dist/src/eval-workflows/evaluation-preparation.js.map +1 -1
  86. package/dist/src/eval-workflows/run-evaluation.d.ts +16 -5
  87. package/dist/src/eval-workflows/run-evaluation.d.ts.map +1 -1
  88. package/dist/src/eval-workflows/run-evaluation.js +34 -5
  89. package/dist/src/eval-workflows/run-evaluation.js.map +1 -1
  90. package/dist/src/executors/anthropic-api.d.ts +1 -1
  91. package/dist/src/executors/anthropic-api.d.ts.map +1 -1
  92. package/dist/src/executors/anthropic-api.js +1 -1
  93. package/dist/src/executors/anthropic-api.js.map +1 -1
  94. package/dist/src/executors/claude-cli.d.ts +2 -2
  95. package/dist/src/executors/claude-cli.d.ts.map +1 -1
  96. package/dist/src/executors/claude-cli.js +21 -1
  97. package/dist/src/executors/claude-cli.js.map +1 -1
  98. package/dist/src/executors/claude-sdk-trace.d.ts +1 -1
  99. package/dist/src/executors/claude-sdk-trace.d.ts.map +1 -1
  100. package/dist/src/executors/claude-sdk.d.ts +15 -2
  101. package/dist/src/executors/claude-sdk.d.ts.map +1 -1
  102. package/dist/src/executors/claude-sdk.js +19 -1
  103. package/dist/src/executors/claude-sdk.js.map +1 -1
  104. package/dist/src/executors/gemini.d.ts +1 -1
  105. package/dist/src/executors/gemini.d.ts.map +1 -1
  106. package/dist/src/executors/index.d.ts +1 -1
  107. package/dist/src/executors/index.d.ts.map +1 -1
  108. package/dist/src/executors/openai-api.d.ts +1 -1
  109. package/dist/src/executors/openai-api.d.ts.map +1 -1
  110. package/dist/src/executors/openai-api.js +1 -1
  111. package/dist/src/executors/openai-api.js.map +1 -1
  112. package/dist/src/executors/openai-cli.d.ts +1 -1
  113. package/dist/src/executors/openai-cli.d.ts.map +1 -1
  114. package/dist/src/executors/script.d.ts +1 -1
  115. package/dist/src/executors/script.d.ts.map +1 -1
  116. package/dist/src/executors/script.js +10 -1
  117. package/dist/src/executors/script.js.map +1 -1
  118. package/dist/src/executors/shared.d.ts +1 -1
  119. package/dist/src/executors/shared.d.ts.map +1 -1
  120. package/dist/src/grading/assertions.d.ts +1 -1
  121. package/dist/src/grading/assertions.d.ts.map +1 -1
  122. package/dist/src/grading/debias-validate.d.ts +1 -1
  123. package/dist/src/grading/debias-validate.d.ts.map +1 -1
  124. package/dist/src/grading/debias-validate.js +3 -3
  125. package/dist/src/grading/gold-cli.d.ts +1 -1
  126. package/dist/src/grading/gold-cli.d.ts.map +1 -1
  127. package/dist/src/grading/gold-cli.js +3 -3
  128. package/dist/src/grading/gold-cli.js.map +1 -1
  129. package/dist/src/grading/index.d.ts +1 -1
  130. package/dist/src/grading/index.d.ts.map +1 -1
  131. package/dist/src/grading/judge.d.ts +1 -1
  132. package/dist/src/grading/judge.d.ts.map +1 -1
  133. package/dist/src/grading/layered-scores.d.ts +1 -1
  134. package/dist/src/grading/layered-scores.d.ts.map +1 -1
  135. package/dist/src/inputs/eval-config.d.ts +1 -1
  136. package/dist/src/inputs/eval-config.d.ts.map +1 -1
  137. package/dist/src/inputs/eval-config.js +37 -18
  138. package/dist/src/inputs/eval-config.js.map +1 -1
  139. package/dist/src/inputs/load-samples.d.ts +1 -1
  140. package/dist/src/inputs/load-samples.d.ts.map +1 -1
  141. package/dist/src/inputs/load-samples.js +4 -4
  142. package/dist/src/inputs/load-samples.js.map +1 -1
  143. package/dist/src/inputs/mcp-resolver.d.ts +1 -1
  144. package/dist/src/inputs/mcp-resolver.d.ts.map +1 -1
  145. package/dist/src/inputs/mcp-resolver.js +4 -4
  146. package/dist/src/inputs/mcp-resolver.js.map +1 -1
  147. package/dist/src/inputs/skill-loader.d.ts +11 -2
  148. package/dist/src/inputs/skill-loader.d.ts.map +1 -1
  149. package/dist/src/inputs/skill-loader.js +30 -7
  150. package/dist/src/inputs/skill-loader.js.map +1 -1
  151. package/dist/src/inputs/url-fetcher.d.ts +1 -1
  152. package/dist/src/inputs/url-fetcher.d.ts.map +1 -1
  153. package/dist/src/inputs/url-fetcher.js +2 -2
  154. package/dist/src/inputs/url-fetcher.js.map +1 -1
  155. package/dist/src/observability/skill-health-analyzer.d.ts +1 -1
  156. package/dist/src/observability/skill-health-analyzer.d.ts.map +1 -1
  157. package/dist/src/observability/trace-adapter.d.ts +1 -1
  158. package/dist/src/observability/trace-adapter.d.ts.map +1 -1
  159. package/dist/src/renderer/html-renderer.d.ts +1 -1
  160. package/dist/src/renderer/html-renderer.d.ts.map +1 -1
  161. package/dist/src/renderer/html-renderer.js +3 -3
  162. package/dist/src/renderer/html-renderer.js.map +1 -1
  163. package/dist/src/renderer/layout.d.ts +1 -1
  164. package/dist/src/renderer/layout.d.ts.map +1 -1
  165. package/dist/src/renderer/layout.js +1 -1
  166. package/dist/src/renderer/layout.js.map +1 -1
  167. package/dist/src/renderer/skill-health-renderer.d.ts +1 -1
  168. package/dist/src/renderer/skill-health-renderer.d.ts.map +1 -1
  169. package/dist/src/renderer/summary.d.ts +1 -1
  170. package/dist/src/renderer/summary.d.ts.map +1 -1
  171. package/dist/src/renderer/summary.js +22 -20
  172. package/dist/src/renderer/summary.js.map +1 -1
  173. package/dist/src/renderer/table.d.ts +1 -1
  174. package/dist/src/renderer/table.d.ts.map +1 -1
  175. package/dist/src/renderer/trends.d.ts +1 -1
  176. package/dist/src/renderer/trends.d.ts.map +1 -1
  177. package/dist/src/renderer/trends.js +1 -1
  178. package/dist/src/renderer/trends.js.map +1 -1
  179. package/dist/src/server/job-store.d.ts +1 -1
  180. package/dist/src/server/job-store.d.ts.map +1 -1
  181. package/dist/src/server/report-server.d.ts +1 -1
  182. package/dist/src/server/report-server.d.ts.map +1 -1
  183. package/dist/src/server/report-server.js +20 -20
  184. package/dist/src/server/report-server.js.map +1 -1
  185. package/dist/src/server/report-store.d.ts +1 -1
  186. package/dist/src/server/report-store.d.ts.map +1 -1
  187. package/dist/src/types/eval.d.ts +9 -0
  188. package/dist/src/types/eval.d.ts.map +1 -1
  189. package/dist/src/types/executor.d.ts +1 -0
  190. package/dist/src/types/executor.d.ts.map +1 -1
  191. package/dist/src/types/report.d.ts +10 -0
  192. package/dist/src/types/report.d.ts.map +1 -1
  193. package/package.json +2 -2
  194. package/dist/src/eval-core/ci-gates.d.ts +0 -17
  195. package/dist/src/eval-core/ci-gates.d.ts.map +0 -1
  196. package/dist/src/eval-core/ci-gates.js.map +0 -1
  197. package/dist/src/types.d.ts +0 -2
  198. package/dist/src/types.d.ts.map +0 -1
  199. package/dist/src/types.js +0 -6
  200. package/dist/src/types.js.map +0 -1
@@ -0,0 +1,934 @@
1
+ /**
2
+ * CLI 文案字典。
3
+ *
4
+ * 命名约定: `cli.<command>.<event>` 或 `cli.common.<event>`。
5
+ * 占位符用 `{name}` 形式,在 tCli(params) 处替换。
6
+ *
7
+ * ============================================================================
8
+ * 翻译守则 (受 cc-viewer i18n 方案启发)
9
+ * ============================================================================
10
+ *
11
+ * 1. **彻底本地化, 不接受中英混搭**
12
+ * "中文用户读到的中文"和"英文用户读到的英文"必须是各自语言里自然的表达,
13
+ * 不能机械翻译, 不能在中文里塞英文短语解释术语。如果某个英文短语没有
14
+ * 自然的中文译法, 重新组织句子结构, 而不是混着写。
15
+ *
16
+ * 2. **保留原文的白名单 (产品术语 / 命令 / 文件名)**
17
+ * 以下 token 在两种语言里都保留原文, 不翻译:
18
+ * - 产品名: omk, oh-my-knowledge, Claude, npm
19
+ * - 子命令空间和命令名: bench, analyze, run, report, init, evolve, gold,
20
+ * diff, ci, gen-samples, debias-validate, saturation, verdict,
21
+ * diagnose, failures
22
+ * - omk 核心业务术语: skill, variant, sample, judge, executor (出现在产品
23
+ * UI 里时首字母可大写如 "Skill 评测", 描述句中保持小写)
24
+ * - 技术参数: --lang, --control, --treatment, --bootstrap, --judge-repeat,
25
+ * OMK_LANG, JUDGE_PROMPT_VERSION_*
26
+ * - 文件名 / 路径: eval-samples.json, skills/v1.md, ~/.oh-my-knowledge/...
27
+ * - 数学概念缩写: CI, α, RAG (其译法可在配套描述里说明, 但术语本身留原文)
28
+ *
29
+ * 3. **必须翻译的内容**
30
+ * 动作 (run / edit / scaffold / generate), 状态 (success / failed /
31
+ * invalid), 引导文案 (next steps / try this / see also), 解释性描述。
32
+ *
33
+ * 4. **不要机械直译**
34
+ * "Next steps:" 译 "下一步:" 而不是 "下一步骤:"。
35
+ * "Run: ..." 译 "运行: ..." 而不是 "跑: ..."。
36
+ * 选用 omk 项目长期使用的中文措辞 (LLM judge 译"评委" 不译"判官", 见
37
+ * feedback_ui_translation.md)。
38
+ *
39
+ * 5. **新增 key 流程**
40
+ * a. 加到 CliMessageKey union 类型里
41
+ * b. 在 CLI_DICT 里同时给出 zh / en (Record 类型强制 zh/en 双写, 漏写
42
+ * tsc 直接报错)
43
+ * c. 自查: 中文里有没有非白名单的英文? 英文里有没有中文?
44
+ * d. 自查: 措辞自然度 — 把中文版念出来, 像不像中文项目的命令行输出?
45
+ * e. test/cli-i18n.test.ts 会跑 runtime parity 检查
46
+ *
47
+ * 未来扩 Lang (zh-TW / ja / ko ...): 改 src/types/shared.ts 的 Lang union,
48
+ * Record 类型自动强制每 key 加新语言版本。
49
+ */
50
+ export const CLI_DICT = {
51
+ 'cli.common.lang_invalid_silent': {
52
+ zh: '无效的语言代码: {value} (仅支持 zh / en, 已使用默认 zh)',
53
+ en: 'Invalid language code: {value} (supported: zh / en, using default zh)',
54
+ },
55
+ 'cli.common.help_hint': {
56
+ zh: "运行 'omk --help' 查看用法",
57
+ en: "Run 'omk --help' to see usage",
58
+ },
59
+ 'cli.common.unknown_domain': {
60
+ zh: "未知顶层命令: {domain} (请用 'omk bench <command>' 或 'omk analyze <dir>')",
61
+ en: "Unknown domain: {domain} (use 'omk bench <command>' or 'omk analyze <dir>')",
62
+ },
63
+ 'cli.common.unknown_bench_command': {
64
+ zh: "未知子命令: bench {command} (运行 'omk --help' 查看可用列表)",
65
+ en: "Unknown bench command: {command} (run 'omk --help' to see all commands)",
66
+ },
67
+ 'cli.init.scaffolded': {
68
+ zh: '已初始化测评项目: {dir}',
69
+ en: 'Eval project scaffolded at: {dir}',
70
+ },
71
+ 'cli.init.next_steps_title': {
72
+ zh: '下一步:',
73
+ en: 'Next steps:',
74
+ },
75
+ 'cli.init.next_step_edit_samples': {
76
+ zh: ' 1. 编辑 eval-samples.json, 加入你要测的测评用例',
77
+ en: ' 1. Edit eval-samples.json to add your test cases',
78
+ },
79
+ 'cli.init.next_step_edit_skills': {
80
+ zh: ' 2. 编辑 skills/v1.md 和 skills/v2.md, 为两个 skill 版本填入实际内容',
81
+ en: ' 2. Edit skills/v1.md and skills/v2.md with your skill versions',
82
+ },
83
+ 'cli.init.next_step_run': {
84
+ zh: ' 3. 运行: omk bench run --control v1 --treatment v2',
85
+ en: ' 3. Run: omk bench run --control v1 --treatment v2',
86
+ },
87
+ 'cli.update.new_version_available': {
88
+ zh: '\n💡 新版本可用: {old} → {new}, 运行 npm update {pkg} -g 升级\n\n',
89
+ en: '\n💡 New version available: {old} → {new}, run npm update {pkg} -g to upgrade\n\n',
90
+ },
91
+ 'cli.progress.preflight_starting': {
92
+ zh: '⏳ 正在预检模型连通性...\n',
93
+ en: '⏳ Preflight: checking model connectivity...\n',
94
+ },
95
+ 'cli.progress.sample_retry': {
96
+ zh: '[{i}/{n}] {sample}/{variant} 🔄 重试 {attempt}/{max}...\n',
97
+ en: '[{i}/{n}] {sample}/{variant} 🔄 retry {attempt}/{max}...\n',
98
+ },
99
+ 'cli.progress.sample_error': {
100
+ zh: '[{i}/{n}] {sample}/{variant} ❌ {error}\n',
101
+ en: '[{i}/{n}] {sample}/{variant} ❌ {error}\n',
102
+ },
103
+ 'cli.progress.sample_executing': {
104
+ zh: '[{i}/{n}] {sample}/{variant} ⏳ 执行中...\n',
105
+ en: '[{i}/{n}] {sample}/{variant} ⏳ running...\n',
106
+ },
107
+ 'cli.progress.sample_exec_done': {
108
+ zh: '[{i}/{n}] {sample}/{variant} 执行完成 {ms}ms {input}+{output} tokens{cost}\n',
109
+ en: '[{i}/{n}] {sample}/{variant} done {ms}ms {input}+{output} tokens{cost}\n',
110
+ },
111
+ 'cli.progress.output_preview': {
112
+ zh: ' 输出预览: {preview}\n',
113
+ en: ' output preview: {preview}\n',
114
+ },
115
+ 'cli.progress.judging': {
116
+ zh: '[{i}/{n}] {sample}/{variant} 评委评审中{dim}...\n',
117
+ en: '[{i}/{n}] {sample}/{variant} judging{dim}...\n',
118
+ },
119
+ 'cli.progress.judged': {
120
+ zh: '[{i}/{n}] {sample}/{variant} 评委评审完成{dim} score={score}\n',
121
+ en: '[{i}/{n}] {sample}/{variant} judged{dim} score={score}\n',
122
+ },
123
+ 'cli.progress.skipped': {
124
+ zh: '[{i}/{n}] {sample}/{variant} ⏭ 已跳过 (已有结果)\n',
125
+ en: '[{i}/{n}] {sample}/{variant} ⏭ skipped (cached)\n',
126
+ },
127
+ 'cli.progress.sample_done': {
128
+ zh: '[{i}/{n}] {sample}/{variant} ✓ {ms}ms {input}+{output} tokens{cost}{score}\n',
129
+ en: '[{i}/{n}] {sample}/{variant} ✓ {ms}ms {input}+{output} tokens{cost}{score}\n',
130
+ },
131
+ 'cli.run.invalid_repeat': {
132
+ zh: '⚠ --repeat "{value}" 无效 (期望 ≥ 1 的整数), 已按 1 次评测执行\n',
133
+ en: '⚠ --repeat "{value}" is invalid (expected an integer ≥ 1), falling back to 1 run\n',
134
+ },
135
+ 'cli.run.invalid_judge_repeat': {
136
+ zh: '⚠ --judge-repeat "{value}" 无效 (期望 ≥ 1 的整数), 已按 1 次 judge 执行\n',
137
+ en: '⚠ --judge-repeat "{value}" is invalid (expected an integer ≥ 1), falling back to 1 judge call\n',
138
+ },
139
+ 'cli.run.invalid_judge_models_format': {
140
+ zh: '--judge-models 格式错误: "{part}", 应为 "executor:model" (例如 claude:opus)',
141
+ en: '--judge-models format error: "{part}", expected "executor:model" (e.g. claude:opus)',
142
+ },
143
+ 'cli.run.judge_models_single_warning': {
144
+ zh: 'ℹ --judge-models 只指定了 1 个 judge ({executor}:{model}), 不会进入 ensemble 模式。如需 ensemble, 至少配 2 个。\n',
145
+ en: 'ℹ --judge-models specified only 1 judge ({executor}:{model}); ensemble not triggered. Configure at least 2 for ensemble mode.\n',
146
+ },
147
+ 'cli.run.no_debias_length_active': {
148
+ zh: 'ℹ --no-debias-length 已生效: judge prompt 退回 v2-cot, 与 < v0.21 报告 hash 一致。\n',
149
+ en: 'ℹ --no-debias-length is active: judge prompt reverts to v2-cot, matching < v0.21 report hashes.\n',
150
+ },
151
+ 'cli.run.invalid_bootstrap_samples': {
152
+ zh: '⚠ --bootstrap-samples "{value}" 无效 (期望 ≥ 100 的整数), 已按 1000 执行\n',
153
+ en: '⚠ --bootstrap-samples "{value}" is invalid (expected an integer ≥ 100), falling back to 1000\n',
154
+ },
155
+ 'cli.run.bootstrap_samples_too_large': {
156
+ zh: '⚠ --bootstrap-samples {n} 较大, 可能耗时数秒。1000 是业内标准, 通常已够用。\n',
157
+ en: '⚠ --bootstrap-samples {n} is large and may take several seconds. 1000 is the industry standard and usually sufficient.\n',
158
+ },
159
+ 'cli.run.skill_section': {
160
+ zh: '\n=== [{i}/{n}] Skill: {skill} ===\n',
161
+ en: '\n=== [{i}/{n}] Skill: {skill} ===\n',
162
+ },
163
+ 'cli.run.run_section': {
164
+ zh: '\n=== 第 {i}/{n} 轮 ===\n',
165
+ en: '\n=== Run {i}/{n} ===\n',
166
+ },
167
+ 'cli.run.batch_complete': {
168
+ zh: '\n✅ 批量评测完成\n',
169
+ en: '\n✅ Batch evaluation done\n',
170
+ },
171
+ 'cli.run.eval_complete': {
172
+ zh: '\n✅ 评测完成\n',
173
+ en: '\n✅ Evaluation done\n',
174
+ },
175
+ 'cli.run.report_saved': {
176
+ zh: '📄 报告已保存到: {path}\n',
177
+ en: '📄 Report saved to: {path}\n',
178
+ },
179
+ 'cli.run.report_server_running': {
180
+ zh: '\n📊 报告服务已启动: {url}\n',
181
+ en: '\n📊 Report server running at {url}\n',
182
+ },
183
+ 'cli.run.report_server_view': {
184
+ zh: '👉 查看报告: {url}\n',
185
+ en: '👉 View report: {url}\n',
186
+ },
187
+ 'cli.run.report_server_stop': {
188
+ zh: '\n按 Ctrl+C 停止服务\n',
189
+ en: '\nPress Ctrl+C to stop the server\n',
190
+ },
191
+ 'cli.run.no_serve_in_non_tty': {
192
+ zh: '\n💡 非交互环境, 已跳过 report server\n',
193
+ en: '\n💡 Non-interactive environment, skipping report server\n',
194
+ },
195
+ 'cli.run.no_serve_view_hint': {
196
+ zh: ' 查看报告: omk bench report --reports-dir {dir}\n',
197
+ en: ' View report: omk bench report --reports-dir {dir}\n',
198
+ },
199
+ 'cli.run.gold_load_failed': {
200
+ zh: '\n⚠ gold dataset 加载失败 ({dir}):\n',
201
+ en: '\n⚠ Failed to load gold dataset ({dir}):\n',
202
+ },
203
+ 'cli.run.gold_load_issue': {
204
+ zh: ' - {message}\n',
205
+ en: ' - {message}\n',
206
+ },
207
+ 'cli.run.contamination_warning': {
208
+ zh: '\n⚠ {warning}\n',
209
+ en: '\n⚠ {warning}\n',
210
+ },
211
+ 'cli.common.error_prefix': {
212
+ zh: '错误: {message}',
213
+ en: 'Error: {message}',
214
+ },
215
+ 'cli.analyze.view_in_browser': {
216
+ zh: "在浏览器查看: omk bench report # 打开后点首页的 \"📊 Skill 健康度日报\"",
217
+ en: "View in browser: omk bench report # then click \"📊 Skill health report\" on the home page",
218
+ },
219
+ 'cli.common.skill_dir_not_found': {
220
+ zh: '未找到 skill 目录: {path}',
221
+ en: 'Skill directory not found: {path}',
222
+ },
223
+ 'cli.common.skill_file_not_found': {
224
+ zh: '未找到 skill 文件: {path}',
225
+ en: 'Skill file not found: {path}',
226
+ },
227
+ 'cli.common.report_not_found': {
228
+ zh: '未找到 report: {id}',
229
+ en: 'Report not found: {id}',
230
+ },
231
+ 'cli.common.no_judge_model': {
232
+ zh: '未指定评委模型。请加 --judge-model <id>, 或确保 report.meta.judgeModel 已写。',
233
+ en: 'No judge model. Pass --judge-model <id> or ensure report has meta.judgeModel.',
234
+ },
235
+ 'cli.common.usage_gold_validate': {
236
+ zh: '用法: omk bench gold validate <dir>',
237
+ en: 'Usage: omk bench gold validate <dir>',
238
+ },
239
+ 'cli.common.warn_load_samples_failed': {
240
+ zh: '⚠ 加载 samples 文件失败 ({path}): {message}\n',
241
+ en: '⚠ Failed to load samples file ({path}): {message}\n',
242
+ },
243
+ 'cli.gen.skill_skipped_existing': {
244
+ zh: '⏭️ {name}: eval-samples 已存在, 跳过\n',
245
+ en: '⏭️ {name}: eval-samples already exists, skipping\n',
246
+ },
247
+ 'cli.gen.skill_generating': {
248
+ zh: '🔄 {name}: 正在生成 {count} 条测评用例...\n',
249
+ en: '🔄 {name}: generating {count} test cases...\n',
250
+ },
251
+ 'cli.gen.skill_done': {
252
+ zh: '✅ {name}: 已生成 {n} 条用例 → {path}{cost}\n',
253
+ en: '✅ {name}: generated {n} samples → {path}{cost}\n',
254
+ },
255
+ 'cli.gen.skill_failed': {
256
+ zh: '❌ {name}: {message}\n',
257
+ en: '❌ {name}: {message}\n',
258
+ },
259
+ 'cli.gen.batch_none_needed': {
260
+ zh: '没有需要生成的 eval-samples (所有 skill 都已有配对文件)',
261
+ en: 'No eval-samples need generating (all skills already have paired files)',
262
+ },
263
+ 'cli.gen.batch_summary': {
264
+ zh: '\n共生成 {n} 份 eval-samples, 请审查后运行: omk bench run --each',
265
+ en: '\nGenerated {n} eval-samples files. Review them, then run: omk bench run --each',
266
+ },
267
+ 'cli.gen.specify_skill_path': {
268
+ zh: '请指定 skill 文件路径, 例如: omk bench gen-samples skills/my-skill.md',
269
+ en: 'Please specify a skill file path, e.g.: omk bench gen-samples skills/my-skill.md',
270
+ },
271
+ 'cli.gen.samples_already_exists': {
272
+ zh: 'eval-samples.json 已存在。如需覆盖请先删除该文件。',
273
+ en: 'eval-samples.json already exists. Delete it first if you want to overwrite.',
274
+ },
275
+ 'cli.gen.single_generating': {
276
+ zh: '🔄 正在生成 {count} 条测评用例...\n',
277
+ en: '🔄 Generating {count} test cases...\n',
278
+ },
279
+ 'cli.gen.single_done': {
280
+ zh: '✅ 已生成 {n} 条用例 → {path}{cost}\n',
281
+ en: '✅ Generated {n} samples → {path}{cost}\n',
282
+ },
283
+ 'cli.gen.review_hint': {
284
+ zh: '\n请审查生成的测评用例后运行: omk bench run',
285
+ en: '\nReview the generated test cases, then run: omk bench run',
286
+ },
287
+ 'cli.gen.failed': {
288
+ zh: '生成失败: {message}',
289
+ en: 'Generation failed: {message}',
290
+ },
291
+ 'cli.evolve.specify_skill_path': {
292
+ zh: '请指定 skill 文件路径, 例如: omk bench evolve skills/my-skill.md',
293
+ en: 'Please specify a skill file path, e.g.: omk bench evolve skills/my-skill.md',
294
+ },
295
+ 'cli.evolve.section_header': {
296
+ zh: '\n=== Evolution: {path} ===\n',
297
+ en: '\n=== Evolution: {path} ===\n',
298
+ },
299
+ 'cli.evolve.round_baseline': {
300
+ zh: '第 0 轮 (基线): score={score} (${cost})\n',
301
+ en: 'Round 0 (baseline): score={score} (${cost})\n',
302
+ },
303
+ 'cli.evolve.round_error': {
304
+ zh: '第 {round} 轮: ✗ 改进生成失败: {error}\n',
305
+ en: 'Round {round}: ✗ improvement generation failed: {error}\n',
306
+ },
307
+ 'cli.evolve.round_done': {
308
+ zh: '第 {round} 轮: score={score} ({delta}) {status} (${cost})\n',
309
+ en: 'Round {round}: score={score} ({delta}) {status} (${cost})\n',
310
+ },
311
+ 'cli.evolve.summary': {
312
+ zh: '\n✅ {start} → {final} (+{percent}%) | 共 {rounds} 轮 | ${cost}\n',
313
+ en: '\n✅ {start} → {final} (+{percent}%) | {rounds} rounds | ${cost}\n',
314
+ },
315
+ 'cli.evolve.best_path': {
316
+ zh: '最优版本: {best} → {target}\n',
317
+ en: 'Best: {best} → {target}\n',
318
+ },
319
+ 'cli.evolve.versions_saved': {
320
+ zh: '所有版本已保存在: {dir}/\n',
321
+ en: 'All versions saved at: {dir}/\n',
322
+ },
323
+ 'cli.evolve.report_link': {
324
+ zh: '📊 评测报告: omk bench report (ID: {id})\n',
325
+ en: '📊 Report: omk bench report (ID: {id})\n',
326
+ },
327
+ 'cli.gold.created_files': {
328
+ zh: '已在 {dir} 创建 {n} 个文件:',
329
+ en: 'Created {n} files in {dir}:',
330
+ },
331
+ 'cli.gold.next_step_edit_annotations': {
332
+ zh: '\n下一步: 编辑 annotations.yaml 加入真实标注 → 跑 omk bench gold validate',
333
+ en: '\nNext step: edit annotations.yaml with real annotations → run omk bench gold validate',
334
+ },
335
+ 'cli.gold.validate_ok': {
336
+ zh: '✓ gold dataset OK — 共 {n} 条标注',
337
+ en: '✓ gold dataset OK — {n} annotations',
338
+ },
339
+ 'cli.debias.warn_cost_doubles': {
340
+ zh: '\n⚠ debias-validate 会重判所有 (sample × variant), judge 成本大约翻倍。\n',
341
+ en: '\n⚠ debias-validate will re-judge all (sample × variant) pairs; judge cost will roughly double.\n',
342
+ },
343
+ 'cli.saturation.no_data': {
344
+ zh: '该 report 没有 saturation 数据 (需要 --repeat ≥ 2 才会记录)。',
345
+ en: 'This report has no saturation data (requires --repeat ≥ 2 to record).',
346
+ },
347
+ 'cli.saturation.verdict_header': {
348
+ zh: '\n Saturation verdict (复述持久化结果)\n',
349
+ en: '\n Saturation verdict (replaying persisted result)\n',
350
+ },
351
+ 'cli.saturation.variant_no_trace': {
352
+ zh: ' {variant}: 没有 trace 数据',
353
+ en: ' {variant}: no trace data',
354
+ },
355
+ 'cli.saturation.variant_label': {
356
+ zh: ' {variant}:',
357
+ en: ' {variant}:',
358
+ },
359
+ 'cli.saturation.checkpoints': {
360
+ zh: ' 检查点: {n} (N={list})',
361
+ en: ' checkpoints: {n} (N={list})',
362
+ },
363
+ 'cli.saturation.last_point': {
364
+ zh: ' 最后一点 mean={mean}, CI=[{lo}, {hi}]',
365
+ en: ' last point mean={mean}, CI=[{lo}, {hi}]',
366
+ },
367
+ 'cli.saturation.persisted_verdict': {
368
+ zh: ' 持久化判定 ({method}): {result} - {reason}',
369
+ en: ' persisted verdict ({method}): {result} - {reason}',
370
+ },
371
+ 'cli.saturation.persisted_verdict_saturated': {
372
+ zh: '已饱和@N={n}',
373
+ en: 'saturated@N={n}',
374
+ },
375
+ 'cli.saturation.persisted_verdict_unsaturated': {
376
+ zh: '未饱和',
377
+ en: 'not saturated',
378
+ },
379
+ 'cli.saturation.skipped_too_few_points': {
380
+ zh: ' 判定: 数据点数 {n} < 5, 跳过 (需要跑 --repeat 5 以上才会输出)',
381
+ en: ' verdict: only {n} data points (< 5), skipping (need --repeat 5 or more)',
382
+ },
383
+ 'cli.help.main': {
384
+ zh: `
385
+ oh-my-knowledge — 知识工件评测工具集
386
+
387
+ 用法:
388
+ omk bench run [options] 跑一轮评测
389
+ omk bench report [options] 启动报告 server
390
+ omk bench gate [options] 跑评测 + 应用 gate, exit code 0/1 (CI/CD 用)
391
+ omk bench init [dir] 初始化一个评测项目
392
+ omk bench gen-samples [skill] 从 skill 内容生成 eval-samples
393
+ omk bench diff <id1> <id2> 对比两份评测报告
394
+ omk bench evolve <skill> 通过迭代评测自我改进 skill
395
+
396
+ omk analyze <dir> 分析 cc session trace, 生成 skill 健康度日报 (v0.18)
397
+
398
+ bench run 选项:
399
+
400
+ --samples <path> 用例文件 (默认: eval-samples.json)
401
+ --skill-dir <path> skill 定义目录 (默认: skills)
402
+ --control <expr> 对照组 variant 表达式 (实验角色 = control)
403
+ --treatment <v1,v2> 实验组 variant 表达式 (逗号分隔; 角色 = treatment)
404
+ 每个 variant 表达式解析为一个 artifact 加上可选运行时上下文:
405
+ "baseline" — 裸模型, 不注入 artifact
406
+ "git:name" — 来自最后一次 commit 的 artifact
407
+ "git:ref:name" — 来自指定 commit 的 artifact
408
+ 带 "/" 的路径 — 直接来自文件 (例如 ./v1.md)
409
+ "name@/cwd" — 附加运行时上下文 / cwd
410
+ --control 和 --treatment 至少要给一个。
411
+ --config <path> YAML/JSON 配置文件 (evaluation-as-code)。
412
+ 在一个文件里声明 samples + variants + model + executor。
413
+ CLI flag 会覆盖配置文件中的同名字段。
414
+ 配置中的相对路径相对于配置文件所在目录解析。
415
+ --model <name> 被测模型 (默认: sonnet)
416
+ --judge-model <name> 评委模型 (默认: haiku)
417
+ --output-dir <path> 报告输出目录 (默认: ~/.oh-my-knowledge/reports/)
418
+ --no-judge 跳过 LLM 评委
419
+ --no-cache 禁用结果缓存
420
+ --dry-run 预览任务但不执行
421
+ --blind 双盲 A/B 模式: 报告里隐藏 variant 名称
422
+ --concurrency <n> 并发任务数 (默认: 1)
423
+ --timeout <seconds> 单任务执行超时 (秒, 默认: 120)
424
+ --repeat <n> 跑 N 轮做方差分析 (默认: 1)
425
+ --judge-repeat <n> 每个 (sample × dimension) 调 LLM 评委 N 次评估
426
+ 自洽性 (默认: 1)。多轮间高 stddev = 评委在该评分维度
427
+ 上不稳定, 分数有噪声。
428
+ --judge-models <list> 多评委 ensemble。逗号分隔的 executor:model, 如
429
+ claude:opus,openai:gpt-4o,gemini:pro。每个评委对所有
430
+ (sample × dimension) 打分; 报告含每评委分布 + Pearson
431
+ / MAD 评委间一致性。能反驳 "Claude 评委评 Claude 同
432
+ 模态偏置" 的质疑。可与 --judge-repeat 组合。
433
+ 成本 ~ N_judges × N_repeat × N_samples。
434
+ --bootstrap 计算 bootstrap 置信区间 (无分布假设, 对 LLM 序数评分
435
+ 比 t 区间更靠谱)。给出每个 variant 均值 CI + treatment
436
+ vs control 差值的 pairwise CI (CI 不跨 0 即显著)。
437
+ 同时报告 t 区间和 bootstrap, 旧工具仍可用。
438
+ --bootstrap-samples <n> bootstrap 重采样次数 (默认 1000)。N>10000 触发
439
+ stderr 警告提示耗时。
440
+ --retry <n> 失败任务最多重试 N 次, 指数退避 (默认: 0)
441
+ --resume <report-id> 从历史报告恢复, 跳过已完成任务
442
+ --executor <name> 执行器: claude / openai / gemini / anthropic-api /
443
+ openai-api, 或任意 shell 命令 (例如 "python my_provider.py")
444
+ --judge-executor <name> 评委执行器 (默认: 同 --executor)
445
+ --each 对每个 skill 独立 vs baseline 评测
446
+ 需要每个 skill 有配对的 {name}.eval-samples.json
447
+ --skip-preflight 评测前跳过模型连通性预检
448
+ --mcp-config <path> 通过 MCP server 抓 URL 用的 MCP 配置文件
449
+ (默认: 当前目录下的 .mcp.json)
450
+ --no-serve 评测后不自动启动报告 server
451
+ --verbose 打印每个用例的详细进度 (执行结果 / 评分阶段)
452
+ --layered-stats 默认在 HTML 报告里展开三层 (fact/behavior/judge) 独立
453
+ 显著性细分。不加这个 flag 时, 细分会折叠在每个对比下
454
+ 的 click-to-expand summary 里。
455
+ --strict-baseline (默认开启) 对 baseline-kind variant 强制隔离 skill 自动
456
+ 发现 + Skill 工具调用, 切断 ~/.claude/skills/ 污染路径,
457
+ 保证 skill 评测的 construct validity。eval.yaml 显式
458
+ allowedSkills 优先。注: v0.22 默认行为变更, 旧报告与新
459
+ 报告 verdict / Δ 不可跨版本对比 (CHANGELOG 标注)。
460
+ --no-strict-baseline 显式关闭 strict-baseline (退回 v0.21 行为, baseline 走
461
+ 默认 SDK skill 全发现)。少数场景下可能想要这个 (例如
462
+ 评测 skill 文档对默认全发现行为的增量影响)。开启时
463
+ pre-flight 会 stderr 提醒, 因为 verdict / Δ 易受污染。
464
+
465
+ bench gate 选项:
466
+ (与 bench run 相同, 额外加:)
467
+ --threshold <number> 三层 gate 阈值 (fact / behavior / LLM judge), 独立应用
468
+ 到每一层。任一层低于阈值即失败 — 防止合成均值掩盖单层
469
+ 崩塌。默认: 3.5。如果三层全空 (没有 assertion 也没在
470
+ eval-samples 里定义 rubric), gate 失败并提示配置问题,
471
+ 不走合成 fallback。
472
+ --trivial-diff <num> 实际可忽略的最小 diff (默认 0.1)。bootstrap diff CI
473
+ 显著但 |Δ| 小于此值视为"统计有效但实际无意义",标
474
+ CAUTIOUS 不给 PROGRESS。
475
+
476
+ 内部 = bench run + bench verdict, exit code 与 bench verdict 对齐:
477
+ PROGRESS / SOLO-PASS → 0; NOISE / UNDERPOWERED / CAUTIOUS / REGRESS → 1。
478
+ 数据 underpowered 时直接 FAIL, 堵住"单轮过 PASS 就 deploy"漏洞。
479
+
480
+ bench report 选项:
481
+ --port <number> server 端口 (默认: 7799)
482
+ --reports-dir <path> 报告目录 (默认: ~/.oh-my-knowledge/reports/)
483
+ --export <id> 把报告导出为独立 HTML 文件
484
+ --dev 开发模式: lib/ 文件改动时自动重启
485
+
486
+ bench gen-samples 选项:
487
+ --each 为所有还没 eval-samples 的 skill 生成
488
+ --count <n> 每个 skill 生成多少条用例 (默认: 5)
489
+ --model <name> 生成用的模型 (默认: sonnet)
490
+ --skill-dir <path> skill 目录 (默认: skills), 配合 --each 用
491
+
492
+ analyze 选项:
493
+ <dir> 输入: cc session JSONL 文件 / 目录
494
+ (例如 ~/.claude/projects/<slug>)
495
+ --kb <path> 知识库根路径 (默认: 从 trace cwd 自动推断)
496
+ --last <duration> 时间窗口, 例如 "7d" / "30d" (默认: 全部)
497
+ --from <iso> 窗口起点 (ISO8601), 优先级高于 --last
498
+ --to <iso> 窗口终点 (ISO8601), 优先级高于 --last
499
+ --skills <n1,n2,...> 白名单要分析的 skill (默认: 全部)
500
+ --output-dir <path> 输出目录 (默认: ~/.oh-my-knowledge/analyses/)
501
+
502
+ bench evolve 选项:
503
+ --rounds <n> 最大演化轮数 (默认: 5)
504
+ --target <score> 达到该分数即提前停止
505
+ --samples <path> 用例文件 (默认: eval-samples.json)
506
+ --model <name> 被测模型 (默认: sonnet)
507
+ --judge-model <name> 评委模型 (默认: haiku)
508
+ --improve-model <name> 生成改进版的模型 (默认: sonnet)
509
+ --concurrency <n> 并发评测任务数 (默认: 1)
510
+ --timeout <seconds> 单任务执行超时 (秒, 默认: 120)
511
+ --executor <name> 执行器 (默认: claude)
512
+
513
+ 通用选项:
514
+ --lang <zh|en> CLI 输出语言 (默认: zh, 也可设 OMK_LANG 环境变量)
515
+
516
+ 示例:
517
+ omk bench run --control v1 --treatment v2
518
+ omk bench run --control baseline --treatment my-skill
519
+ omk bench run --control git:my-skill --treatment my-skill
520
+ omk bench run --control ./old-skill.md --treatment ./new-skill.md
521
+ omk bench run --control baseline --treatment v1,v2,v3
522
+ omk bench run --config eval.yaml
523
+ omk bench run --config eval.yaml --model sonnet-4.6 # CLI 覆盖配置
524
+ omk bench run --each
525
+ omk bench run --dry-run
526
+ omk bench report --port 8080
527
+ omk bench report --export v1-vs-v2-20260326-1832
528
+ omk bench init my-eval
529
+ omk bench gen-samples skills/my-skill.md
530
+ `,
531
+ en: `
532
+ oh-my-knowledge — Knowledge artifact evaluation toolkit
533
+
534
+ Usage:
535
+ omk bench run [options] Run an evaluation
536
+ omk bench report [options] Start the report server
537
+ omk bench gate [options] Run evaluation + apply gate, exit 0/1 (for CI/CD)
538
+ omk bench init [dir] Scaffold a new eval project
539
+ omk bench gen-samples [skill] Generate eval-samples from skill content
540
+ omk bench diff <id1> <id2> Compare two evaluation reports
541
+ omk bench evolve <skill> Self-improve a skill through iterative evaluation
542
+
543
+ omk analyze <dir> Analyze cc session trace(s), produce skill health report (v0.18)
544
+
545
+ Options for "bench run":
546
+
547
+ --samples <path> Sample file (default: eval-samples.json)
548
+ --skill-dir <path> Skill definitions directory (default: skills)
549
+ --control <expr> Control-group variant expression (experiment role = control)
550
+ --treatment <v1,v2> Treatment-group variant expressions (comma-separated; role = treatment)
551
+ Each variant expression resolves to an artifact and optional runtime context:
552
+ "baseline" — bare model, no artifact injected
553
+ "git:name" — artifact from last commit
554
+ "git:ref:name" — artifact from specific commit
555
+ path with "/" — artifact from file directly (e.g. ./v1.md)
556
+ "name@/cwd" — attach runtime context / cwd
557
+ At least one of --control / --treatment must be provided.
558
+ --config <path> YAML/JSON config file (evaluation-as-code).
559
+ Declares samples + variants + model + executor in one file.
560
+ CLI flags override config fields when both are provided.
561
+ Relative paths inside the config are resolved against its directory.
562
+ --model <name> Model under test (default: sonnet)
563
+ --judge-model <name> Judge model (default: haiku)
564
+ --output-dir <path> Report output directory (default: ~/.oh-my-knowledge/reports/)
565
+ --no-judge Skip LLM judging
566
+ --no-cache Disable result caching
567
+ --dry-run Preview tasks without executing
568
+ --blind Blind A/B mode: hide variant names in report
569
+ --concurrency <n> Number of parallel tasks (default: 1)
570
+ --timeout <seconds> Executor timeout per task in seconds (default: 120)
571
+ --repeat <n> Run evaluation N times for variance analysis (default: 1)
572
+ --judge-repeat <n> Call LLM judge N times per (sample × dimension) for self-
573
+ consistency (default: 1). High stddev across runs = the
574
+ judge is unstable on this rubric and the score is noisy.
575
+ --judge-models <list> Multi-judge ensemble. Comma-separated executor:model pairs,
576
+ e.g. claude:opus,openai:gpt-4o,gemini:pro. Each judge scores
577
+ every (sample × dimension); report includes per-judge break-
578
+ down + Pearson/MAD inter-judge agreement. Refutes "Claude
579
+ judge Claude same-modality bias" critique. Combines with
580
+ --judge-repeat. Cost ~ N_judges × N_repeat × N_samples.
581
+ --bootstrap Compute bootstrap confidence intervals (distribution-free,
582
+ preferred over t-interval for ordinal LLM scores). Adds
583
+ per-variant CI on the mean + pairwise CI on treatment-vs-
584
+ control difference (significant=0 outside CI). Reports both
585
+ t-interval and bootstrap so old tooling still works.
586
+ --bootstrap-samples <n> Number of bootstrap resamples (default 1000). N>10000
587
+ triggers a stderr warning about runtime cost.
588
+ --retry <n> Retry failed tasks up to N times with exponential backoff (default: 0)
589
+ --resume <report-id> Resume from a previous report, skipping completed tasks
590
+ --executor <name> Executor: claude, openai, gemini, anthropic-api, openai-api,
591
+ or any shell command (e.g. "python my_provider.py")
592
+ --judge-executor <name> Executor for LLM judge (default: same as --executor)
593
+ --each Evaluate each skill independently against baseline
594
+ Requires {name}.eval-samples.json paired with each skill
595
+ --skip-preflight Skip model connectivity check before evaluation
596
+ --mcp-config <path> MCP config file for URL fetching via MCP servers
597
+ (default: .mcp.json in current directory)
598
+ --no-serve Skip auto-starting report server after evaluation
599
+ --verbose Print detailed progress for each sample (exec result, grading phases)
600
+ --layered-stats Expand the three-layer (fact/behavior/judge) independent
601
+ significance breakdown in the HTML report by default.
602
+ Without this flag, the breakdown is collapsed behind a
603
+ click-to-expand summary under each comparison.
604
+ --strict-baseline (default ON) Isolate skill auto-discovery + Skill tool
605
+ use for baseline-kind variants. Cuts the ~/.claude/skills/
606
+ contamination path so skill evaluations have valid
607
+ construct validity. Explicit eval.yaml allowedSkills
608
+ takes precedence. Note: v0.22 default-behavior change —
609
+ pre-v0.22 reports cannot be compared head-to-head on
610
+ verdict / Δ (see CHANGELOG).
611
+ --no-strict-baseline Explicitly turn strict-baseline OFF (reverts to pre-v0.22
612
+ behavior: baseline sees all auto-discovered skills). Use
613
+ only in narrow scenarios (e.g. measuring how much a skill
614
+ doc adds on top of full default discovery). Pre-flight
615
+ emits a stderr warning when this flag is set, because
616
+ verdict / Δ are vulnerable to skill contamination.
617
+
618
+ Options for "bench gate":
619
+ (same as "bench run", plus:)
620
+ --threshold <number> Three-layer gate threshold (fact / behavior / LLM judge),
621
+ applied INDEPENDENTLY to each layer. ANY layer below
622
+ threshold fails the gate — prevents composite averaging
623
+ from masking a single-layer collapse. Default: 3.5.
624
+ If all three layers are absent (no
625
+ assertions and no rubric defined in eval-samples), the
626
+ gate FAILS with a configuration hint — no composite fallback.
627
+ --trivial-diff <num> Smallest diff to treat as practically meaningful
628
+ (default 0.1). Bootstrap diff CI may be statistically
629
+ significant but with |Δ| < this value, treated as
630
+ CAUTIOUS rather than PROGRESS.
631
+
632
+ Internally = bench run + bench verdict. Exit code aligns with bench verdict:
633
+ PROGRESS / SOLO-PASS → 0; NOISE / UNDERPOWERED / CAUTIOUS / REGRESS → 1.
634
+ Underpowered runs fail directly — closes the "single-run PASS = deploy" loophole.
635
+
636
+ Options for "bench report":
637
+ --port <number> Server port (default: 7799)
638
+ --reports-dir <path> Reports directory (default: ~/.oh-my-knowledge/reports/)
639
+ --export <id> Export report as standalone HTML file
640
+ --dev Dev mode: auto-restart on lib/ file changes
641
+
642
+ Options for "bench gen-samples":
643
+ --each Generate for all skills missing eval-samples
644
+ --count <n> Number of samples to generate per skill (default: 5)
645
+ --model <name> Model for generation (default: sonnet)
646
+ --skill-dir <path> Skill directory (default: skills), used with --each
647
+
648
+ Options for "analyze":
649
+ <dir> Input: cc session JSONL file / dir (e.g. ~/.claude/projects/<slug>)
650
+ --kb <path> Knowledge base root (default: auto-infer from trace cwd)
651
+ --last <duration> Time window like "7d" / "30d" (default: all)
652
+ --from <iso> Window start (ISO8601), takes precedence over --last
653
+ --to <iso> Window end (ISO8601), takes precedence over --last
654
+ --skills <n1,n2,...> Whitelist skills to analyze (default: all)
655
+ --output-dir <path> Output dir (default: ~/.oh-my-knowledge/analyses/)
656
+
657
+ Options for "bench evolve":
658
+ --rounds <n> Maximum evolution rounds (default: 5)
659
+ --target <score> Stop early when score reaches this threshold
660
+ --samples <path> Sample file (default: eval-samples.json)
661
+ --model <name> Model under test (default: sonnet)
662
+ --judge-model <name> Judge model (default: haiku)
663
+ --improve-model <name> Model for generating improvements (default: sonnet)
664
+ --concurrency <n> Parallel eval tasks (default: 1)
665
+ --timeout <seconds> Executor timeout per task in seconds (default: 120)
666
+ --executor <name> Executor to use (default: claude)
667
+
668
+ Common options:
669
+ --lang <zh|en> CLI output language (default: zh, also via OMK_LANG env)
670
+
671
+ Examples:
672
+ omk bench run --control v1 --treatment v2
673
+ omk bench run --control baseline --treatment my-skill
674
+ omk bench run --control git:my-skill --treatment my-skill
675
+ omk bench run --control ./old-skill.md --treatment ./new-skill.md
676
+ omk bench run --control baseline --treatment v1,v2,v3
677
+ omk bench run --config eval.yaml
678
+ omk bench run --config eval.yaml --model sonnet-4.6 # CLI overrides config
679
+ omk bench run --each
680
+ omk bench run --dry-run
681
+ omk bench report --port 8080
682
+ omk bench report --export v1-vs-v2-20260326-1832
683
+ omk bench init my-eval
684
+ omk bench gen-samples skills/my-skill.md
685
+ `,
686
+ },
687
+ 'cli.help.diff_usage': {
688
+ zh: [
689
+ '用法:',
690
+ ' omk bench diff <reportId> 单 report 内 sample 级 diff (v0.22)',
691
+ ' omk bench diff <reportId1> <reportId2> 跨 report variant 级 diff',
692
+ '',
693
+ '选项:',
694
+ ' --regressions-only 只列 treatment < control 的用例',
695
+ ' --threshold <num> 回退判定阈值 (默认 0, 即任何负 Δ 都算回退)',
696
+ ' --variant <name> within-report 模式下指定要钻取的 variant (默认: variants[1])',
697
+ ' --top <n> 只列差距最大的前 N 个用例',
698
+ ].join('\n'),
699
+ en: [
700
+ 'Usage:',
701
+ ' omk bench diff <reportId> within-report per-sample diff (v0.22)',
702
+ ' omk bench diff <reportId1> <reportId2> cross-report variant-level diff',
703
+ '',
704
+ 'Options:',
705
+ ' --regressions-only show only samples where treatment < control',
706
+ ' --threshold <num> regression threshold (default 0, any negative Δ counts)',
707
+ ' --variant <name> within-report mode: which variant to drill (default: variants[1])',
708
+ ' --top <n> only show top N samples by absolute diff',
709
+ ].join('\n'),
710
+ },
711
+ 'cli.help.analyze_usage': {
712
+ zh: '用法: omk analyze <dir> [--kb <path>] [--last 7d] [--from ISO] [--to ISO] [--skills name1,name2]',
713
+ en: 'Usage: omk analyze <dir> [--kb <path>] [--last 7d] [--from ISO] [--to ISO] [--skills name1,name2]',
714
+ },
715
+ 'cli.help.gold': {
716
+ zh: [
717
+ '',
718
+ '用法: omk bench gold <subcommand>',
719
+ '',
720
+ '子命令:',
721
+ ' init [--out <dir>] [--annotator <id>] 生成空白 gold dataset 模板',
722
+ ' validate <dir> 校验数据集结构',
723
+ ' compare <reportId> --gold-dir <dir> 与已有 report 计算 α / κ / Pearson',
724
+ ' [--variant <name>] [--reports-dir <d>]',
725
+ ' [--bootstrap-samples N] [--seed N]',
726
+ '',
727
+ ].join('\n'),
728
+ en: [
729
+ '',
730
+ 'Usage: omk bench gold <subcommand>',
731
+ '',
732
+ 'Subcommands:',
733
+ ' init [--out <dir>] [--annotator <id>] create a blank gold dataset template',
734
+ ' validate <dir> validate dataset structure',
735
+ ' compare <reportId> --gold-dir <dir> compute α / κ / Pearson against an existing report',
736
+ ' [--variant <name>] [--reports-dir <d>]',
737
+ ' [--bootstrap-samples N] [--seed N]',
738
+ '',
739
+ ].join('\n'),
740
+ },
741
+ 'cli.help.debias_validate': {
742
+ zh: [
743
+ '',
744
+ '用法: omk bench debias-validate <kind> <reportId> [options]',
745
+ '',
746
+ '类别:',
747
+ ' length 用相反的长度去偏设置重新评判, 并对分数差出 bootstrap CI。',
748
+ ' judge 成本约为原评判的两倍。',
749
+ '',
750
+ '选项:',
751
+ ' --reports-dir <dir> 报告存储目录 (默认: ~/.oh-my-knowledge/reports)',
752
+ ' --samples <path> 覆盖用例文件 (默认: 从 report.meta.request 读)',
753
+ ' --variant <name> 校验哪个 variant (默认: 第一个)',
754
+ ' --judge-executor <name> 评委调用执行器 (默认: claude)',
755
+ ' --judge-model <model> 评委模型 ID (默认: 沿用 report)',
756
+ ' --bootstrap-samples N bootstrap 迭代次数 (默认 1000)',
757
+ ' --seed N 固定 CI 随机种子',
758
+ '',
759
+ ].join('\n'),
760
+ en: [
761
+ '',
762
+ 'Usage: omk bench debias-validate <kind> <reportId> [options]',
763
+ '',
764
+ 'Kinds:',
765
+ ' length re-judge with the opposite length-debias setting and bootstrap CI',
766
+ ' on the score diff. Cost ~doubles vs the original judge pass.',
767
+ '',
768
+ 'Options:',
769
+ ' --reports-dir <dir> report store dir (default: ~/.oh-my-knowledge/reports)',
770
+ ' --samples <path> override samples file (default: from report.meta.request)',
771
+ ' --variant <name> which variant to validate (default: first)',
772
+ ' --judge-executor <name> executor for judge calls (default: claude)',
773
+ ' --judge-model <model> judge model id (default: from report)',
774
+ ' --bootstrap-samples N bootstrap iterations (default 1000)',
775
+ ' --seed N deterministic CI seed',
776
+ '',
777
+ ].join('\n'),
778
+ },
779
+ 'cli.help.saturation': {
780
+ zh: [
781
+ '',
782
+ '用法: omk bench saturation <reportId> [options]',
783
+ '',
784
+ '回答 "我跑够用例了吗?"。复述已有 report 中持久化的饱和判定。',
785
+ '',
786
+ '注: 本命令读取 run 时跑出的 verdict (运行时已用 method=bootstrap-ci-width',
787
+ '默认阈值 + 3 窗口持续条件)。如要换 method/threshold 重新计算, 需要重跑',
788
+ '`omk bench run --repeat ≥ 5` (运行时持久化的 trace 不含原始分数, 无法',
789
+ '在事后用其他参数复算)。',
790
+ '',
791
+ '选项:',
792
+ ' --reports-dir <dir> 报告存储目录 (默认: ~/.oh-my-knowledge/reports)',
793
+ ' --variant <name> 只看一个 variant (默认: 全部)',
794
+ '',
795
+ ].join('\n'),
796
+ en: [
797
+ '',
798
+ 'Usage: omk bench saturation <reportId> [options]',
799
+ '',
800
+ 'Answers "do I have enough samples?". Replays the saturation verdict',
801
+ 'persisted in an existing report.',
802
+ '',
803
+ 'Note: this command reads the verdict computed at run time (which used',
804
+ 'method=bootstrap-ci-width with default threshold + 3-window sustained',
805
+ 'condition). To re-compute with a different method/threshold, re-run',
806
+ '`omk bench run --repeat ≥ 5` (the persisted trace does not include raw',
807
+ 'scores, so post-hoc parameter sweeps are not possible here).',
808
+ '',
809
+ 'Options:',
810
+ ' --reports-dir <dir> report store dir (default: ~/.oh-my-knowledge/reports)',
811
+ ' --variant <name> only show one variant (default: all)',
812
+ '',
813
+ ].join('\n'),
814
+ },
815
+ 'cli.help.verdict': {
816
+ zh: [
817
+ '',
818
+ '用法: omk bench verdict <reportId> [options]',
819
+ '',
820
+ '聚合 bootstrap CI / 三层 ci-gate / saturation / human α, 给出一行结论。',
821
+ '',
822
+ 'Verdict 等级:',
823
+ ' PROGRESS 显著改进 + 三层全过',
824
+ ' CAUTIOUS 改进真实但有警告 (gate 破 / 幅度太小 / 控制组本身崩)',
825
+ ' REGRESS 显著回退 — 不要 ship',
826
+ ' NOISE CI 跨 0, 无法判定',
827
+ ' UNDERPOWERED 用例不足, 需要扩 N 重测',
828
+ ' SOLO 单 variant 报告, 没有对比对象',
829
+ '',
830
+ '选项:',
831
+ ' --reports-dir <dir> 报告存储目录 (默认: ~/.oh-my-knowledge/reports)',
832
+ ' --threshold <num> 三层 gate 阈值 (默认 3.5, 与 omk bench gate 对齐)',
833
+ ' --trivial-diff <num> "幅度太小" 阈值 (默认 0.1)',
834
+ ' --verbose 展开每个 pair 的详情',
835
+ '',
836
+ ].join('\n'),
837
+ en: [
838
+ '',
839
+ 'Usage: omk bench verdict <reportId> [options]',
840
+ '',
841
+ 'Aggregates bootstrap CI / 3-layer ci-gate / saturation / human α into a one-line verdict.',
842
+ '',
843
+ 'Verdict levels:',
844
+ ' PROGRESS significant improvement + all 3 layers pass',
845
+ ' CAUTIOUS real improvement but with warnings (gate fails / diff too small / control collapsed)',
846
+ ' REGRESS significant regression — do not ship',
847
+ ' NOISE CI crosses 0, no verdict',
848
+ ' UNDERPOWERED not enough samples, expand N and re-run',
849
+ ' SOLO single-variant report, nothing to compare against',
850
+ '',
851
+ 'Options:',
852
+ ' --reports-dir <dir> report store dir (default: ~/.oh-my-knowledge/reports)',
853
+ ' --threshold <num> 3-layer gate threshold (default 3.5, matches omk bench gate)',
854
+ ' --trivial-diff <num> "diff too small" threshold (default 0.1)',
855
+ ' --verbose expand per-pair details',
856
+ '',
857
+ ].join('\n'),
858
+ },
859
+ 'cli.help.diagnose': {
860
+ zh: [
861
+ '',
862
+ '用法: omk bench diagnose <reportId> [options]',
863
+ '',
864
+ '诊断用例集本身的质量问题: 区分度低 / 重复 / 歧义 / 成本异常 / 全 fail。',
865
+ '回答 "测评结论是否被坏用例污染" — 与 omk bench verdict 互补。',
866
+ '',
867
+ '选项:',
868
+ ' --reports-dir <dir> 报告存储目录',
869
+ ' --samples <path> 用例文件路径 (用于 near-duplicate 检测; 默认从 report.meta.request 读)',
870
+ ' --top <n> 每类只显示前 N 个 (默认 10, 0=全部)',
871
+ ' --duplicate-rouge <num> near-duplicate ROUGE-1 阈值 (默认 0.7)',
872
+ ' --ambiguous-stddev <num> 歧义阈值, judge stddev (默认 1.0, 需要 --judge-repeat ≥ 2 数据)',
873
+ ' --cost-k <num> 成本异常倍数 vs 中位数 (默认 3)',
874
+ ' --latency-k <num> 耗时异常倍数 vs 中位数 (默认 3)',
875
+ ' --flat <num> flat_scores 分差阈值 (默认 0.5)',
876
+ '',
877
+ ].join('\n'),
878
+ en: [
879
+ '',
880
+ 'Usage: omk bench diagnose <reportId> [options]',
881
+ '',
882
+ 'Diagnose quality issues in the sample set itself: low discrimination /',
883
+ 'duplicates / ambiguity / cost anomalies / all-fail. Answers "is the verdict',
884
+ 'tainted by bad samples?" — complements omk bench verdict.',
885
+ '',
886
+ 'Options:',
887
+ ' --reports-dir <dir> report store dir',
888
+ ' --samples <path> sample file path (for near-duplicate detection; defaults to report.meta.request)',
889
+ ' --top <n> top N per category (default 10, 0=all)',
890
+ ' --duplicate-rouge <num> near-duplicate ROUGE-1 threshold (default 0.7)',
891
+ ' --ambiguous-stddev <num> ambiguity threshold, judge stddev (default 1.0, requires --judge-repeat ≥ 2)',
892
+ ' --cost-k <num> cost-outlier multiplier vs median (default 3)',
893
+ ' --latency-k <num> latency-outlier multiplier vs median (default 3)',
894
+ ' --flat <num> flat_scores spread threshold (default 0.5)',
895
+ '',
896
+ ].join('\n'),
897
+ },
898
+ 'cli.help.failures': {
899
+ zh: [
900
+ '',
901
+ '用法: omk bench failures <reportId> [options]',
902
+ '',
903
+ '把已有 report 的失败用例喂给一次 LLM 调用, 自动聚类并给出修复建议。',
904
+ '失败定义: compositeScore < threshold 或 ok=false。',
905
+ '',
906
+ '选项:',
907
+ ' --reports-dir <dir> 报告存储目录',
908
+ ' --judge-executor <name> 执行器 (默认: claude)',
909
+ ' --judge-model <id> 聚类用的模型 (默认: 沿用 report.meta.judgeModel)',
910
+ ' --max-clusters <n> 最多聚成几类 (默认 5)',
911
+ ' --threshold <num> compositeScore < threshold 算失败 (默认 3)',
912
+ ' --max-feed <n> 最多喂给 LLM 多少条 (默认 50, 超出取最差)',
913
+ '',
914
+ ].join('\n'),
915
+ en: [
916
+ '',
917
+ 'Usage: omk bench failures <reportId> [options]',
918
+ '',
919
+ 'Feed failing samples from an existing report to a single LLM call, auto-cluster',
920
+ 'them, and produce per-cluster fix suggestions.',
921
+ 'Failure definition: compositeScore < threshold or ok=false.',
922
+ '',
923
+ 'Options:',
924
+ ' --reports-dir <dir> report store dir',
925
+ ' --judge-executor <name> executor (default: claude)',
926
+ ' --judge-model <id> model for clustering (default: from report.meta.judgeModel)',
927
+ ' --max-clusters <n> max number of clusters (default 5)',
928
+ ' --threshold <num> compositeScore < threshold counts as failure (default 3)',
929
+ ' --max-feed <n> max samples to feed the LLM (default 50, takes the worst)',
930
+ '',
931
+ ].join('\n'),
932
+ },
933
+ };
934
+ //# sourceMappingURL=i18n-dict.js.map