oh-my-knowledge 0.20.1 → 0.22.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (200) hide show
  1. package/README.md +10 -4
  2. package/README.zh.md +9 -4
  3. package/dist/src/analysis/coverage-analyzer.d.ts +1 -1
  4. package/dist/src/analysis/coverage-analyzer.d.ts.map +1 -1
  5. package/dist/src/analysis/failure-clusterer.d.ts +1 -1
  6. package/dist/src/analysis/failure-clusterer.d.ts.map +1 -1
  7. package/dist/src/analysis/gap-analyzer.d.ts +2 -2
  8. package/dist/src/analysis/gap-analyzer.d.ts.map +1 -1
  9. package/dist/src/analysis/gap-analyzer.js +4 -4
  10. package/dist/src/analysis/hedging-classifier.d.ts +1 -1
  11. package/dist/src/analysis/hedging-classifier.d.ts.map +1 -1
  12. package/dist/src/analysis/report-diagnostics.d.ts +1 -1
  13. package/dist/src/analysis/report-diagnostics.d.ts.map +1 -1
  14. package/dist/src/analysis/report-diagnostics.js +7 -7
  15. package/dist/src/analysis/sample-diagnostics.d.ts +1 -1
  16. package/dist/src/analysis/sample-diagnostics.d.ts.map +1 -1
  17. package/dist/src/analysis/sample-diagnostics.js +15 -15
  18. package/dist/src/authoring/evolver.d.ts +1 -1
  19. package/dist/src/authoring/evolver.d.ts.map +1 -1
  20. package/dist/src/authoring/evolver.js +5 -5
  21. package/dist/src/authoring/evolver.js.map +1 -1
  22. package/dist/src/authoring/generator.d.ts +1 -1
  23. package/dist/src/authoring/generator.d.ts.map +1 -1
  24. package/dist/src/authoring/generator.js +8 -8
  25. package/dist/src/authoring/generator.js.map +1 -1
  26. package/dist/src/cli/i18n-dict.d.ts +56 -0
  27. package/dist/src/cli/i18n-dict.d.ts.map +1 -0
  28. package/dist/src/cli/i18n-dict.js +934 -0
  29. package/dist/src/cli/i18n-dict.js.map +1 -0
  30. package/dist/src/cli/i18n.d.ts +24 -0
  31. package/dist/src/cli/i18n.d.ts.map +1 -0
  32. package/dist/src/cli/i18n.js +53 -0
  33. package/dist/src/cli/i18n.js.map +1 -0
  34. package/dist/src/cli.js +320 -413
  35. package/dist/src/cli.js.map +1 -1
  36. package/dist/src/eval-core/cache.d.ts +7 -3
  37. package/dist/src/eval-core/cache.d.ts.map +1 -1
  38. package/dist/src/eval-core/cache.js +14 -4
  39. package/dist/src/eval-core/cache.js.map +1 -1
  40. package/dist/src/eval-core/dependency-checker.d.ts +16 -1
  41. package/dist/src/eval-core/dependency-checker.d.ts.map +1 -1
  42. package/dist/src/eval-core/dependency-checker.js +79 -4
  43. package/dist/src/eval-core/dependency-checker.js.map +1 -1
  44. package/dist/src/eval-core/evaluation-execution.d.ts +3 -3
  45. package/dist/src/eval-core/evaluation-execution.d.ts.map +1 -1
  46. package/dist/src/eval-core/evaluation-execution.js +4 -2
  47. package/dist/src/eval-core/evaluation-execution.js.map +1 -1
  48. package/dist/src/eval-core/evaluation-job.d.ts +2 -2
  49. package/dist/src/eval-core/evaluation-job.d.ts.map +1 -1
  50. package/dist/src/eval-core/evaluation-reporting.d.ts +1 -1
  51. package/dist/src/eval-core/evaluation-reporting.d.ts.map +1 -1
  52. package/dist/src/eval-core/evaluation-reporting.js +4 -0
  53. package/dist/src/eval-core/evaluation-reporting.js.map +1 -1
  54. package/dist/src/eval-core/execution-strategy.d.ts +1 -1
  55. package/dist/src/eval-core/execution-strategy.d.ts.map +1 -1
  56. package/dist/src/eval-core/execution-strategy.js +35 -2
  57. package/dist/src/eval-core/execution-strategy.js.map +1 -1
  58. package/dist/src/eval-core/layer-gates.d.ts +17 -0
  59. package/dist/src/eval-core/layer-gates.d.ts.map +1 -0
  60. package/dist/src/eval-core/{ci-gates.js → layer-gates.js} +5 -5
  61. package/dist/src/eval-core/layer-gates.js.map +1 -0
  62. package/dist/src/eval-core/schema.d.ts +1 -1
  63. package/dist/src/eval-core/schema.d.ts.map +1 -1
  64. package/dist/src/eval-core/schema.js +17 -3
  65. package/dist/src/eval-core/schema.js.map +1 -1
  66. package/dist/src/eval-core/task-planner.d.ts +1 -1
  67. package/dist/src/eval-core/task-planner.d.ts.map +1 -1
  68. package/dist/src/eval-core/task-planner.js +1 -1
  69. package/dist/src/eval-core/task-planner.js.map +1 -1
  70. package/dist/src/eval-core/verdict.d.ts +5 -2
  71. package/dist/src/eval-core/verdict.d.ts.map +1 -1
  72. package/dist/src/eval-core/verdict.js +47 -8
  73. package/dist/src/eval-core/verdict.js.map +1 -1
  74. package/dist/src/eval-workflows/each-evaluation-workflow.d.ts +15 -8
  75. package/dist/src/eval-workflows/each-evaluation-workflow.d.ts.map +1 -1
  76. package/dist/src/eval-workflows/each-evaluation-workflow.js +3 -2
  77. package/dist/src/eval-workflows/each-evaluation-workflow.js.map +1 -1
  78. package/dist/src/eval-workflows/evaluation-pipeline.d.ts +33 -4
  79. package/dist/src/eval-workflows/evaluation-pipeline.d.ts.map +1 -1
  80. package/dist/src/eval-workflows/evaluation-pipeline.js +81 -2
  81. package/dist/src/eval-workflows/evaluation-pipeline.js.map +1 -1
  82. package/dist/src/eval-workflows/evaluation-preparation.d.ts +10 -6
  83. package/dist/src/eval-workflows/evaluation-preparation.d.ts.map +1 -1
  84. package/dist/src/eval-workflows/evaluation-preparation.js +2 -2
  85. package/dist/src/eval-workflows/evaluation-preparation.js.map +1 -1
  86. package/dist/src/eval-workflows/run-evaluation.d.ts +16 -5
  87. package/dist/src/eval-workflows/run-evaluation.d.ts.map +1 -1
  88. package/dist/src/eval-workflows/run-evaluation.js +34 -5
  89. package/dist/src/eval-workflows/run-evaluation.js.map +1 -1
  90. package/dist/src/executors/anthropic-api.d.ts +1 -1
  91. package/dist/src/executors/anthropic-api.d.ts.map +1 -1
  92. package/dist/src/executors/anthropic-api.js +1 -1
  93. package/dist/src/executors/anthropic-api.js.map +1 -1
  94. package/dist/src/executors/claude-cli.d.ts +2 -2
  95. package/dist/src/executors/claude-cli.d.ts.map +1 -1
  96. package/dist/src/executors/claude-cli.js +21 -1
  97. package/dist/src/executors/claude-cli.js.map +1 -1
  98. package/dist/src/executors/claude-sdk-trace.d.ts +1 -1
  99. package/dist/src/executors/claude-sdk-trace.d.ts.map +1 -1
  100. package/dist/src/executors/claude-sdk.d.ts +15 -2
  101. package/dist/src/executors/claude-sdk.d.ts.map +1 -1
  102. package/dist/src/executors/claude-sdk.js +19 -1
  103. package/dist/src/executors/claude-sdk.js.map +1 -1
  104. package/dist/src/executors/gemini.d.ts +1 -1
  105. package/dist/src/executors/gemini.d.ts.map +1 -1
  106. package/dist/src/executors/index.d.ts +1 -1
  107. package/dist/src/executors/index.d.ts.map +1 -1
  108. package/dist/src/executors/openai-api.d.ts +1 -1
  109. package/dist/src/executors/openai-api.d.ts.map +1 -1
  110. package/dist/src/executors/openai-api.js +1 -1
  111. package/dist/src/executors/openai-api.js.map +1 -1
  112. package/dist/src/executors/openai-cli.d.ts +1 -1
  113. package/dist/src/executors/openai-cli.d.ts.map +1 -1
  114. package/dist/src/executors/script.d.ts +1 -1
  115. package/dist/src/executors/script.d.ts.map +1 -1
  116. package/dist/src/executors/script.js +10 -1
  117. package/dist/src/executors/script.js.map +1 -1
  118. package/dist/src/executors/shared.d.ts +1 -1
  119. package/dist/src/executors/shared.d.ts.map +1 -1
  120. package/dist/src/grading/assertions.d.ts +1 -1
  121. package/dist/src/grading/assertions.d.ts.map +1 -1
  122. package/dist/src/grading/debias-validate.d.ts +1 -1
  123. package/dist/src/grading/debias-validate.d.ts.map +1 -1
  124. package/dist/src/grading/debias-validate.js +3 -3
  125. package/dist/src/grading/gold-cli.d.ts +1 -1
  126. package/dist/src/grading/gold-cli.d.ts.map +1 -1
  127. package/dist/src/grading/gold-cli.js +3 -3
  128. package/dist/src/grading/gold-cli.js.map +1 -1
  129. package/dist/src/grading/index.d.ts +1 -1
  130. package/dist/src/grading/index.d.ts.map +1 -1
  131. package/dist/src/grading/judge.d.ts +1 -1
  132. package/dist/src/grading/judge.d.ts.map +1 -1
  133. package/dist/src/grading/layered-scores.d.ts +1 -1
  134. package/dist/src/grading/layered-scores.d.ts.map +1 -1
  135. package/dist/src/inputs/eval-config.d.ts +1 -1
  136. package/dist/src/inputs/eval-config.d.ts.map +1 -1
  137. package/dist/src/inputs/eval-config.js +37 -18
  138. package/dist/src/inputs/eval-config.js.map +1 -1
  139. package/dist/src/inputs/load-samples.d.ts +1 -1
  140. package/dist/src/inputs/load-samples.d.ts.map +1 -1
  141. package/dist/src/inputs/load-samples.js +4 -4
  142. package/dist/src/inputs/load-samples.js.map +1 -1
  143. package/dist/src/inputs/mcp-resolver.d.ts +1 -1
  144. package/dist/src/inputs/mcp-resolver.d.ts.map +1 -1
  145. package/dist/src/inputs/mcp-resolver.js +4 -4
  146. package/dist/src/inputs/mcp-resolver.js.map +1 -1
  147. package/dist/src/inputs/skill-loader.d.ts +11 -2
  148. package/dist/src/inputs/skill-loader.d.ts.map +1 -1
  149. package/dist/src/inputs/skill-loader.js +30 -7
  150. package/dist/src/inputs/skill-loader.js.map +1 -1
  151. package/dist/src/inputs/url-fetcher.d.ts +1 -1
  152. package/dist/src/inputs/url-fetcher.d.ts.map +1 -1
  153. package/dist/src/inputs/url-fetcher.js +2 -2
  154. package/dist/src/inputs/url-fetcher.js.map +1 -1
  155. package/dist/src/observability/skill-health-analyzer.d.ts +1 -1
  156. package/dist/src/observability/skill-health-analyzer.d.ts.map +1 -1
  157. package/dist/src/observability/trace-adapter.d.ts +1 -1
  158. package/dist/src/observability/trace-adapter.d.ts.map +1 -1
  159. package/dist/src/renderer/html-renderer.d.ts +1 -1
  160. package/dist/src/renderer/html-renderer.d.ts.map +1 -1
  161. package/dist/src/renderer/html-renderer.js +3 -3
  162. package/dist/src/renderer/html-renderer.js.map +1 -1
  163. package/dist/src/renderer/layout.d.ts +1 -1
  164. package/dist/src/renderer/layout.d.ts.map +1 -1
  165. package/dist/src/renderer/layout.js +1 -1
  166. package/dist/src/renderer/layout.js.map +1 -1
  167. package/dist/src/renderer/skill-health-renderer.d.ts +1 -1
  168. package/dist/src/renderer/skill-health-renderer.d.ts.map +1 -1
  169. package/dist/src/renderer/summary.d.ts +1 -1
  170. package/dist/src/renderer/summary.d.ts.map +1 -1
  171. package/dist/src/renderer/summary.js +22 -20
  172. package/dist/src/renderer/summary.js.map +1 -1
  173. package/dist/src/renderer/table.d.ts +1 -1
  174. package/dist/src/renderer/table.d.ts.map +1 -1
  175. package/dist/src/renderer/trends.d.ts +1 -1
  176. package/dist/src/renderer/trends.d.ts.map +1 -1
  177. package/dist/src/renderer/trends.js +1 -1
  178. package/dist/src/renderer/trends.js.map +1 -1
  179. package/dist/src/server/job-store.d.ts +1 -1
  180. package/dist/src/server/job-store.d.ts.map +1 -1
  181. package/dist/src/server/report-server.d.ts +1 -1
  182. package/dist/src/server/report-server.d.ts.map +1 -1
  183. package/dist/src/server/report-server.js +20 -20
  184. package/dist/src/server/report-server.js.map +1 -1
  185. package/dist/src/server/report-store.d.ts +1 -1
  186. package/dist/src/server/report-store.d.ts.map +1 -1
  187. package/dist/src/types/eval.d.ts +9 -0
  188. package/dist/src/types/eval.d.ts.map +1 -1
  189. package/dist/src/types/executor.d.ts +1 -0
  190. package/dist/src/types/executor.d.ts.map +1 -1
  191. package/dist/src/types/report.d.ts +10 -0
  192. package/dist/src/types/report.d.ts.map +1 -1
  193. package/package.json +2 -2
  194. package/dist/src/eval-core/ci-gates.d.ts +0 -17
  195. package/dist/src/eval-core/ci-gates.d.ts.map +0 -1
  196. package/dist/src/eval-core/ci-gates.js.map +0 -1
  197. package/dist/src/types.d.ts +0 -2
  198. package/dist/src/types.d.ts.map +0 -1
  199. package/dist/src/types.js +0 -6
  200. package/dist/src/types.js.map +0 -1
package/dist/src/cli.js CHANGED
@@ -4,6 +4,7 @@ import { resolve } from 'node:path';
4
4
  import { homedir } from 'node:os';
5
5
  import { join } from 'node:path';
6
6
  import { existsSync } from 'node:fs';
7
+ import { tCli, getCliLang, parseLangFromArgv, langFromArgv } from './cli/i18n.js';
7
8
  import { discoverVariants, parseVariantCwd } from './inputs/skill-loader.js';
8
9
  import { loadEvalConfig, configVariantsToSpecs } from './inputs/eval-config.js';
9
10
  // ---------------------------------------------------------------------------
@@ -14,7 +15,15 @@ const DEFAULT_REPORTS_DIR = join(homedir(), '.oh-my-knowledge', 'reports');
14
15
  // Defaults are applied inside parseRunConfig (after config-file merge) so that
15
16
  // CLI `undefined` can be reliably distinguished from "user passed the default value".
16
17
  // Priority order resolved in parseRunConfig: CLI arg > --config file > hard-coded default.
18
+ /**
19
+ * 所有子命令都接受的通用 flag。新增 --lang 让 parseArgs strict:false 模式下
20
+ * 仍能把值类型化到 values.lang 上(否则未声明的 flag 会被丢弃)。
21
+ */
22
+ const COMMON_OPTIONS = {
23
+ lang: { type: 'string' },
24
+ };
17
25
  const RUN_OPTIONS = {
26
+ ...COMMON_OPTIONS,
18
27
  samples: { type: 'string' },
19
28
  'skill-dir': { type: 'string' },
20
29
  control: { type: 'string' },
@@ -38,6 +47,10 @@ const RUN_OPTIONS = {
38
47
  retry: { type: 'string' },
39
48
  resume: { type: 'string' },
40
49
  'layered-stats': { type: 'boolean' },
50
+ // v0.22 — strict-baseline default true. Declare both forms; reconcile in
51
+ // parseRunConfig (后者赢)。strict-baseline 没传 + no-strict-baseline 没传 = default true。
52
+ 'strict-baseline': { type: 'boolean' },
53
+ 'no-strict-baseline': { type: 'boolean' },
41
54
  };
42
55
  // ---------------------------------------------------------------------------
43
56
  // parseRunConfig
@@ -144,6 +157,21 @@ function parseRunConfig(argv, extraOptions = {}) {
144
157
  const resume = values.resume;
145
158
  const blind = values.blind ?? evalConfig?.blind ?? false;
146
159
  const layeredStats = values['layered-stats'] ?? false;
160
+ // v0.22 — strict-baseline default true. Reconcile both flag forms.
161
+ // Priority: --no-strict-baseline > --strict-baseline > undefined(=true).
162
+ const noStrictFlag = values['no-strict-baseline'];
163
+ const strictFlag = values['strict-baseline'];
164
+ const strictBaseline = noStrictFlag === true ? false : (strictFlag ?? true);
165
+ // v0.22 — extract eval.yaml variant.allowedSkills overrides (per-variant). Always
166
+ // wins over strictBaseline default. Empty object when no eval.yaml or no overrides.
167
+ const variantAllowedSkills = {};
168
+ if (evalConfig?.variants) {
169
+ for (const v of evalConfig.variants) {
170
+ if (v.allowedSkills !== undefined) {
171
+ variantAllowedSkills[v.name] = v.allowedSkills;
172
+ }
173
+ }
174
+ }
147
175
  return {
148
176
  values,
149
177
  config: {
@@ -168,178 +196,56 @@ function parseRunConfig(argv, extraOptions = {}) {
168
196
  blind,
169
197
  layeredStats,
170
198
  budget: evalConfig?.budget,
199
+ strictBaseline,
200
+ ...(Object.keys(variantAllowedSkills).length > 0 && { variantAllowedSkills }),
171
201
  },
172
202
  };
173
203
  }
174
204
  // ---------------------------------------------------------------------------
175
- // Help text
176
- // ---------------------------------------------------------------------------
177
- const HELP = `
178
- oh-my-knowledge — Knowledge artifact evaluation toolkit
179
-
180
- Usage:
181
- omk bench run [options] Run an evaluation
182
- omk bench report [options] Start the report server
183
- omk bench ci [options] Run evaluation and exit with pass/fail code
184
- omk bench init [dir] Scaffold a new eval project
185
- omk bench gen-samples [skill] Generate eval-samples from skill content
186
- omk bench diff <id1> <id2> Compare two evaluation reports
187
- omk bench evolve <skill> Self-improve a skill through iterative evaluation
188
-
189
- omk analyze <dir> Analyze cc session trace(s), produce skill 健康度日报 (v0.18)
190
-
191
- Options for "bench run":
192
-
193
- --samples <path> Sample file (default: eval-samples.json)
194
- --skill-dir <path> Skill definitions directory (default: skills)
195
- --control <expr> Control-group variant expression (experiment role = control)
196
- --treatment <v1,v2> Treatment-group variant expressions (comma-separated; role = treatment)
197
- Each variant expression resolves to an artifact and optional runtime context:
198
- "baseline" — bare model, no artifact injected
199
- "git:name" — artifact from last commit
200
- "git:ref:name" — artifact from specific commit
201
- path with "/" — artifact from file directly (e.g. ./v1.md)
202
- "name@/cwd" — attach runtime context / cwd
203
- At least one of --control / --treatment must be provided.
204
- --config <path> YAML/JSON config file (evaluation-as-code).
205
- Declares samples + variants + model + executor in one file.
206
- CLI flags override config fields when both are provided.
207
- Relative paths inside the config are resolved against its directory.
208
- --model <name> Model under test (default: sonnet)
209
- --judge-model <name> Judge model (default: haiku)
210
- --output-dir <path> Report output directory (default: ~/.oh-my-knowledge/reports/)
211
- --no-judge Skip LLM judging
212
- --no-cache Disable result caching
213
- --dry-run Preview tasks without executing
214
- --blind Blind A/B mode: hide variant names in report
215
- --concurrency <n> Number of parallel tasks (default: 1)
216
- --timeout <seconds> Executor timeout per task in seconds (default: 120)
217
- --repeat <n> Run evaluation N times for variance analysis (default: 1)
218
- --judge-repeat <n> Call LLM judge N times per (sample × dimension) for self-
219
- consistency (default: 1). High stddev across runs = the
220
- judge is unstable on this rubric and the score is noisy.
221
- --judge-models <list> Multi-judge ensemble. Comma-separated executor:model pairs,
222
- e.g. claude:opus,openai:gpt-4o,gemini:pro. Each judge scores
223
- every (sample × dimension); report includes per-judge break-
224
- down + Pearson/MAD inter-judge agreement. Refutes "Claude
225
- judge Claude same-modality bias" critique. Combines with
226
- --judge-repeat. Cost ~ N_judges × N_repeat × N_samples.
227
- --bootstrap Compute bootstrap confidence intervals (distribution-free,
228
- preferred over t-interval for ordinal LLM scores). Adds
229
- per-variant CI on the mean + pairwise CI on treatment-vs-
230
- control difference (significant=0 outside CI). Reports both
231
- t-interval and bootstrap so old tooling still works.
232
- --bootstrap-samples <n> Number of bootstrap resamples (default 1000). N>10000
233
- triggers a stderr warning about runtime cost.
234
- --retry <n> Retry failed tasks up to N times with exponential backoff (default: 0)
235
- --resume <report-id> Resume from a previous report, skipping completed tasks
236
- --executor <name> Executor: claude, openai, gemini, anthropic-api, openai-api,
237
- or any shell command (e.g. "python my_provider.py")
238
- --judge-executor <name> Executor for LLM judge (default: same as --executor)
239
- --each Evaluate each skill independently against baseline
240
- Requires {name}.eval-samples.json paired with each skill
241
- --skip-preflight Skip model connectivity check before evaluation
242
- --mcp-config <path> MCP config file for URL fetching via MCP servers
243
- (default: .mcp.json in current directory)
244
- --no-serve Skip auto-starting report server after evaluation
245
- --verbose Print detailed progress for each sample (exec result, grading phases)
246
- --layered-stats Expand the three-layer (fact/behavior/judge) independent
247
- significance breakdown in the HTML report by default.
248
- Without this flag, the breakdown is collapsed behind a
249
- click-to-expand summary under each comparison.
250
-
251
- Options for "bench ci":
252
- (same as "bench run", plus:)
253
- --threshold <number> Minimum score to pass, applied INDEPENDENTLY to each of
254
- the three layers (fact / behavior / LLM judge). ANY
255
- layer below threshold fails the gate — this prevents
256
- composite averaging from masking a single-layer collapse.
257
- Default: 3.5. If all three layers are absent (no
258
- assertions and no rubric defined in eval-samples), the
259
- gate FAILS with a configuration hint — no composite fallback.
260
-
261
- Options for "bench report":
262
- --port <number> Server port (default: 7799)
263
- --reports-dir <path> Reports directory (default: ~/.oh-my-knowledge/reports/)
264
- --export <id> Export report as standalone HTML file
265
- --dev Dev mode: auto-restart on lib/ file changes
266
-
267
- Options for "bench gen-samples":
268
- --each Generate for all skills missing eval-samples
269
- --count <n> Number of samples to generate per skill (default: 5)
270
- --model <name> Model for generation (default: sonnet)
271
- --skill-dir <path> Skill directory (default: skills), used with --each
272
-
273
- Options for "analyze":
274
- <dir> Input: cc session JSONL file / dir (e.g. ~/.claude/projects/<slug>)
275
- --kb <path> Knowledge base root (default: auto-infer from trace cwd)
276
- --last <duration> Time window like "7d" / "30d" (default: all)
277
- --from <iso> Window start (ISO8601), takes precedence over --last
278
- --to <iso> Window end (ISO8601), takes precedence over --last
279
- --skills <n1,n2,...> Whitelist skills to analyze (default: all)
280
- --output-dir <path> Output dir (default: ~/.oh-my-knowledge/analyses/)
281
-
282
- Options for "bench evolve":
283
- --rounds <n> Maximum evolution rounds (default: 5)
284
- --target <score> Stop early when score reaches this threshold
285
- --samples <path> Sample file (default: eval-samples.json)
286
- --model <name> Model under test (default: sonnet)
287
- --judge-model <name> Judge model (default: haiku)
288
- --improve-model <name> Model for generating improvements (default: sonnet)
289
- --concurrency <n> Parallel eval tasks (default: 1)
290
- --timeout <seconds> Executor timeout per task in seconds (default: 120)
291
- --executor <name> Executor to use (default: claude)
292
-
293
- Examples:
294
- omk bench run --control v1 --treatment v2
295
- omk bench run --control baseline --treatment my-skill
296
- omk bench run --control git:my-skill --treatment my-skill
297
- omk bench run --control ./old-skill.md --treatment ./new-skill.md
298
- omk bench run --control baseline --treatment v1,v2,v3
299
- omk bench run --config eval.yaml
300
- omk bench run --config eval.yaml --model sonnet-4.6 # CLI overrides config
301
- omk bench run --each
302
- omk bench run --dry-run
303
- omk bench report --port 8080
304
- omk bench report --export v1-vs-v2-20260326-1832
305
- omk bench init my-eval
306
- omk bench gen-samples skills/my-skill.md
307
- omk bench gen-samples --each
308
- omk bench diff <report-id-1> <report-id-2>
309
- omk bench evolve skills/my-skill.md --rounds 5
310
- omk analyze ~/.claude/projects/-Users-lizhiyao-Documents-oh-my-knowledge
311
- omk analyze ~/.claude/projects/my-project --last 7d --kb /path/to/project
312
- omk analyze ~/.claude/projects/my-project --skills audit,polish
313
- `.trim();
314
- // ---------------------------------------------------------------------------
315
205
  // Update check
316
206
  // ---------------------------------------------------------------------------
317
- async function checkUpdate() {
207
+ async function checkUpdate(lang) {
318
208
  try {
319
209
  const { readFileSync } = await import('node:fs');
320
210
  const { fileURLToPath } = await import('node:url');
321
211
  const { dirname, join } = await import('node:path');
322
212
  const __dirname = dirname(fileURLToPath(import.meta.url));
323
- const pkg = JSON.parse(readFileSync(join(__dirname, 'package.json'), 'utf-8'));
213
+ const findPackageJson = (startDir) => {
214
+ let dir = startDir;
215
+ for (let i = 0; i < 5; i++) {
216
+ const candidate = join(dir, 'package.json');
217
+ if (existsSync(candidate))
218
+ return candidate;
219
+ dir = dirname(dir);
220
+ }
221
+ return null;
222
+ };
223
+ const pkgPath = findPackageJson(__dirname);
224
+ if (!pkgPath)
225
+ return;
226
+ const pkg = JSON.parse(readFileSync(pkgPath, 'utf-8'));
324
227
  const registry = pkg.publishConfig?.registry || 'https://registry.npmjs.org';
325
228
  const res = await fetch(`${registry}/${pkg.name}/latest`, { signal: AbortSignal.timeout(3000) });
326
229
  if (!res.ok)
327
230
  return;
328
231
  const data = await res.json();
329
232
  if (data.version && data.version !== pkg.version) {
330
- process.stderr.write(`\n💡 新版本可用: ${pkg.version} → ${data.version},运行 npm update ${pkg.name} -g 更新\n\n`);
233
+ process.stderr.write(tCli('cli.update.new_version_available', lang, {
234
+ old: pkg.version, new: data.version, pkg: pkg.name,
235
+ }));
331
236
  }
332
237
  }
333
- catch { /* 静默失败,不影响正常使用 */ }
238
+ catch { /* 静默失败,不影响正常使用 */ }
334
239
  }
335
240
  // ---------------------------------------------------------------------------
336
241
  // Main
337
242
  // ---------------------------------------------------------------------------
338
243
  async function main() {
339
- checkUpdate();
244
+ const lang = getCliLang(parseLangFromArgv(process.argv));
245
+ checkUpdate(lang);
340
246
  const [domain, command, ...rest] = process.argv.slice(2);
341
247
  if (!domain || domain === '--help' || domain === '-h') {
342
- console.log(HELP);
248
+ console.log(tCli('cli.help.main', lang).trim());
343
249
  process.exit(0);
344
250
  }
345
251
  if (domain === 'analyze') {
@@ -348,11 +254,11 @@ async function main() {
348
254
  return;
349
255
  }
350
256
  if (domain !== 'bench') {
351
- console.error(`Unknown domain: ${domain}. Use "omk bench <command>" or "omk analyze <dir>".`);
257
+ console.error(tCli('cli.common.unknown_domain', lang, { domain }));
352
258
  process.exit(1);
353
259
  }
354
260
  if (!command || command === '--help' || command === '-h') {
355
- console.log(HELP);
261
+ console.log(tCli('cli.help.main', lang).trim());
356
262
  process.exit(0);
357
263
  }
358
264
  switch (command) {
@@ -365,8 +271,8 @@ async function main() {
365
271
  case 'init':
366
272
  await handleInit(rest);
367
273
  break;
368
- case 'ci':
369
- await handleCi(rest);
274
+ case 'gate':
275
+ await handleGate(rest);
370
276
  break;
371
277
  case 'gen-samples':
372
278
  await handleGenSamples(rest);
@@ -396,60 +302,77 @@ async function main() {
396
302
  await handleFailures(rest);
397
303
  break;
398
304
  default:
399
- console.error(`Unknown command: bench ${command}. Use "run", "report", "ci", "init", "gen-samples", "evolve", "diff", "gold", "debias-validate", "saturation", "verdict", "diagnose", or "failures".`);
305
+ console.error(tCli('cli.common.unknown_bench_command', lang, { command }));
400
306
  process.exit(1);
401
307
  }
402
308
  }
403
309
  // ---------------------------------------------------------------------------
404
310
  // Progress callback
405
311
  // ---------------------------------------------------------------------------
406
- function defaultOnProgress({ phase, completed, total, sample_id, variant, durationMs, inputTokens, outputTokens, costUSD, score, outputPreview, judgePhase: _judgePhase, judgeDim, skipped, attempt, maxAttempts, error, }) {
407
- if (phase === 'preflight') {
408
- process.stderr.write('⏳ 预检模型连通性...\n');
409
- return;
410
- }
411
- if (phase === 'retry') {
412
- process.stderr.write(`[${completed}/${total}] ${sample_id}/${variant} 🔄 重试 ${attempt}/${maxAttempts}...\n`);
413
- return;
414
- }
415
- if (phase === 'error') {
416
- process.stderr.write(`[${completed}/${total}] ${sample_id}/${variant} ❌ ${error}\n`);
417
- return;
418
- }
419
- if (phase === 'start') {
420
- process.stderr.write(`[${completed}/${total}] ${sample_id}/${variant} ⏳ 执行中...\n`);
421
- }
422
- else if (phase === 'exec_done') {
423
- const costInfo = costUSD != null && costUSD > 0 ? ` $${costUSD.toFixed(4)}` : '';
424
- process.stderr.write(`[${completed}/${total}] ${sample_id}/${variant} 执行完成 ${durationMs}ms ${inputTokens}+${outputTokens} tokens${costInfo}\n`);
425
- if (outputPreview) {
426
- process.stderr.write(` 输出预览: ${outputPreview.slice(0, 150).replace(/\n/g, ' ')}\n`);
312
+ /**
313
+ * Factory: 闭住 lang, 返回 onProgress callback。evaluation engine 回调时不传
314
+ * 上下文, 所以 lang 必须在 handler 入口处通过 closure 传进来。
315
+ */
316
+ function makeOnProgress(lang) {
317
+ return ({ phase, completed, total, sample_id, variant, durationMs, inputTokens, outputTokens, costUSD, score, outputPreview, judgePhase: _judgePhase, judgeDim, skipped, attempt, maxAttempts, error, }) => {
318
+ const ctx = { i: completed ?? '', n: total ?? '', sample: sample_id ?? '', variant: variant ?? '' };
319
+ if (phase === 'preflight') {
320
+ process.stderr.write(tCli('cli.progress.preflight_starting', lang));
321
+ return;
427
322
  }
428
- }
429
- else if (phase === 'grading') {
430
- const dimInfo = judgeDim ? ` [${judgeDim}]` : '';
431
- process.stderr.write(`[${completed}/${total}] ${sample_id}/${variant} 评审中${dimInfo}...\n`);
432
- }
433
- else if (phase === 'judge_done') {
434
- const dimInfo = judgeDim ? ` [${judgeDim}]` : '';
435
- process.stderr.write(`[${completed}/${total}] ${sample_id}/${variant} 评审完成${dimInfo} score=${score}\n`);
436
- }
437
- else if (phase === 'done' && skipped) {
438
- if (sample_id)
439
- process.stderr.write(`[${completed}/${total}] ${sample_id}/${variant} ⏭ 已跳过(已有结果)\n`);
440
- }
441
- else {
442
- const costInfo = costUSD != null && costUSD > 0 ? ` $${costUSD.toFixed(4)}` : '';
443
- const scoreInfo = typeof score === 'number' ? ` score=${score}` : '';
444
- process.stderr.write(`[${completed}/${total}] ${sample_id}/${variant} ✓ ${durationMs}ms ${inputTokens}+${outputTokens} tokens${costInfo}${scoreInfo}\n`);
445
- }
323
+ if (phase === 'retry') {
324
+ process.stderr.write(tCli('cli.progress.sample_retry', lang, {
325
+ ...ctx, attempt: attempt ?? '', max: maxAttempts ?? '',
326
+ }));
327
+ return;
328
+ }
329
+ if (phase === 'error') {
330
+ process.stderr.write(tCli('cli.progress.sample_error', lang, { ...ctx, error: error ?? '' }));
331
+ return;
332
+ }
333
+ if (phase === 'start') {
334
+ process.stderr.write(tCli('cli.progress.sample_executing', lang, ctx));
335
+ }
336
+ else if (phase === 'exec_done') {
337
+ const cost = costUSD != null && costUSD > 0 ? ` $${costUSD.toFixed(4)}` : '';
338
+ process.stderr.write(tCli('cli.progress.sample_exec_done', lang, {
339
+ ...ctx, ms: durationMs ?? '', input: inputTokens ?? '', output: outputTokens ?? '', cost,
340
+ }));
341
+ if (outputPreview) {
342
+ process.stderr.write(tCli('cli.progress.output_preview', lang, {
343
+ preview: outputPreview.slice(0, 150).replace(/\n/g, ' '),
344
+ }));
345
+ }
346
+ }
347
+ else if (phase === 'grading') {
348
+ const dim = judgeDim ? ` [${judgeDim}]` : '';
349
+ process.stderr.write(tCli('cli.progress.judging', lang, { ...ctx, dim }));
350
+ }
351
+ else if (phase === 'judge_done') {
352
+ const dim = judgeDim ? ` [${judgeDim}]` : '';
353
+ process.stderr.write(tCli('cli.progress.judged', lang, { ...ctx, dim, score: score ?? '' }));
354
+ }
355
+ else if (phase === 'done' && skipped) {
356
+ if (sample_id)
357
+ process.stderr.write(tCli('cli.progress.skipped', lang, ctx));
358
+ }
359
+ else {
360
+ const cost = costUSD != null && costUSD > 0 ? ` $${costUSD.toFixed(4)}` : '';
361
+ const scoreInfo = typeof score === 'number' ? ` score=${score}` : '';
362
+ process.stderr.write(tCli('cli.progress.sample_done', lang, {
363
+ ...ctx, ms: durationMs ?? '', input: inputTokens ?? '', output: outputTokens ?? '',
364
+ cost, score: scoreInfo,
365
+ }));
366
+ }
367
+ };
446
368
  }
447
369
  // ---------------------------------------------------------------------------
448
370
  // handleRun
449
371
  // ---------------------------------------------------------------------------
450
372
  async function handleRun(argv) {
373
+ const lang = langFromArgv(argv);
451
374
  const { values, config } = parseRunConfig(argv, {
452
- blind: { type: 'boolean', default: false },
375
+ blind: { type: 'boolean' },
453
376
  repeat: { type: 'string', default: '1' },
454
377
  'judge-repeat': { type: 'string', default: '1' },
455
378
  'judge-models': { type: 'string' },
@@ -462,27 +385,29 @@ async function handleRun(argv) {
462
385
  'budget-per-sample-ms': { type: 'string' },
463
386
  });
464
387
  const { runEvaluation, runMultiple, runEachEvaluation } = await import('./eval-workflows/run-evaluation.js');
465
- config.blind = values.blind;
466
- config.onProgress = defaultOnProgress;
467
- // --repeat 诚实输入校验:非 ≥1 整数时提示并钳到 1,不静默掩盖用户错字/极端输入
468
- // 提前到 --each 分支之前,保证 each 模式也能读到 repeat (曾经 bug: --each 吞 --repeat)
388
+ if (values.blind !== undefined) {
389
+ config.blind = values.blind;
390
+ }
391
+ config.onProgress = makeOnProgress(lang);
392
+ // --repeat 输入校验: 非 ≥1 整数时提示并钳到 1, 不静默掩盖用户错字 / 极端输入。
393
+ // 提前到 --each 分支之前, 保证 each 模式也能读到 repeat (曾经 bug: --each 吞 --repeat)。
469
394
  const repeatRaw = values.repeat;
470
395
  const parsedRepeat = repeatRaw !== undefined ? Number(repeatRaw) : 1;
471
396
  if (repeatRaw !== undefined && (!Number.isFinite(parsedRepeat) || parsedRepeat < 1)) {
472
- process.stderr.write(`⚠ --repeat "${repeatRaw}" 无效(期望 ≥ 1 的整数),已按 1 次评测执行\n`);
397
+ process.stderr.write(tCli('cli.run.invalid_repeat', lang, { value: repeatRaw }));
473
398
  }
474
399
  const repeatCount = Math.max(1, Math.floor(parsedRepeat) || 1);
475
- // --judge-repeat 同样的诚实校验:非 ≥1 整数时钳到 1
400
+ // --judge-repeat 同样校验: 非 ≥1 整数时钳到 1
476
401
  const judgeRepeatRaw = values['judge-repeat'];
477
402
  const parsedJudgeRepeat = judgeRepeatRaw !== undefined ? Number(judgeRepeatRaw) : 1;
478
403
  if (judgeRepeatRaw !== undefined && (!Number.isFinite(parsedJudgeRepeat) || parsedJudgeRepeat < 1)) {
479
- process.stderr.write(`⚠ --judge-repeat "${judgeRepeatRaw}" 无效(期望 ≥ 1 的整数),已按 1 次 judge 执行\n`);
404
+ process.stderr.write(tCli('cli.run.invalid_judge_repeat', lang, { value: judgeRepeatRaw }));
480
405
  }
481
406
  const judgeRepeatCount = Math.max(1, Math.floor(parsedJudgeRepeat) || 1);
482
407
  if (judgeRepeatCount > 1)
483
408
  config.judgeRepeat = judgeRepeatCount;
484
409
  // --judge-models executor:model,executor:model,... -> JudgeConfig[]
485
- // 至少 2 个才进 ensemble 模式,1 个等同于 --judge-model
410
+ // 至少 2 个才进 ensemble 模式, 1 个等同于 --judge-model
486
411
  const judgeModelsRaw = values['judge-models'];
487
412
  if (judgeModelsRaw) {
488
413
  const parts = judgeModelsRaw.split(',').map((s) => s.trim()).filter(Boolean);
@@ -490,7 +415,7 @@ async function handleRun(argv) {
490
415
  const [executor, ...modelParts] = p.split(':');
491
416
  const model = modelParts.join(':');
492
417
  if (!executor || !model) {
493
- throw new Error(`--judge-models 格式错误: "${p}",应为 "executor:model" (如 claude:opus)`);
418
+ throw new Error(tCli('cli.run.invalid_judge_models_format', lang, { part: p }));
494
419
  }
495
420
  return { executor, model };
496
421
  });
@@ -498,8 +423,10 @@ async function handleRun(argv) {
498
423
  config.judgeModels = judges;
499
424
  }
500
425
  else if (judges.length === 1) {
501
- // 单 judge 不走 ensemble,但允许这样写,等同于 --judge-model + --executor
502
- process.stderr.write(`ℹ --judge-models 只指定 1 个 judge (${judges[0].executor}:${judges[0].model}),不触发 ensemble。如需 ensemble 至少给 2 个。\n`);
426
+ // 单 judge 不走 ensemble, 但允许这样写, 等同于 --judge-model + --executor
427
+ process.stderr.write(tCli('cli.run.judge_models_single_warning', lang, {
428
+ executor: judges[0].executor, model: judges[0].model,
429
+ }));
503
430
  }
504
431
  }
505
432
  // --budget-usd / --budget-per-sample-usd / --budget-per-sample-ms:
@@ -521,7 +448,7 @@ async function handleRun(argv) {
521
448
  // off so historical reports (judgePromptHash from v2-cot era) can be reproduced.
522
449
  if (values['no-debias-length']) {
523
450
  config.lengthDebias = false;
524
- process.stderr.write('ℹ --no-debias-length 已生效:judge prompt 退回 v2-cot,与 < v0.21 报告 hash 一致。\n');
451
+ process.stderr.write(tCli('cli.run.no_debias_length_active', lang));
525
452
  }
526
453
  // --bootstrap / --bootstrap-samples
527
454
  if (values.bootstrap) {
@@ -529,11 +456,11 @@ async function handleRun(argv) {
529
456
  const bsRaw = values['bootstrap-samples'];
530
457
  const parsedBs = bsRaw !== undefined ? Number(bsRaw) : 1000;
531
458
  if (bsRaw !== undefined && (!Number.isFinite(parsedBs) || parsedBs < 100)) {
532
- process.stderr.write(`⚠ --bootstrap-samples "${bsRaw}" 无效(期望 ≥ 100 的整数),已按 1000 执行\n`);
459
+ process.stderr.write(tCli('cli.run.invalid_bootstrap_samples', lang, { value: bsRaw }));
533
460
  }
534
461
  const bsCount = Math.max(100, Math.floor(parsedBs) || 1000);
535
462
  if (bsCount > 10000) {
536
- process.stderr.write(`⚠ --bootstrap-samples ${bsCount} 较大,可能耗时数秒。1000 是业内标准,通常已够用。\n`);
463
+ process.stderr.write(tCli('cli.run.bootstrap_samples_too_large', lang, { n: bsCount }));
537
464
  }
538
465
  config.bootstrapSamples = bsCount;
539
466
  }
@@ -545,30 +472,32 @@ async function handleRun(argv) {
545
472
  repeat: repeatCount,
546
473
  onSkillProgress({ phase, skill, current, total }) {
547
474
  if (phase === 'start') {
548
- process.stderr.write(`\n=== [${current}/${total}] Skill: ${skill} ===\n`);
475
+ process.stderr.write(tCli('cli.run.skill_section', lang, {
476
+ i: current ?? '', n: total ?? '', skill: skill ?? '',
477
+ }));
549
478
  }
550
479
  },
551
480
  });
552
481
  console.log(JSON.stringify(report, null, 2));
553
482
  if (filePath) {
554
- process.stderr.write('\n✅ 批量评测完成\n');
555
- process.stderr.write(`📄 Report saved to: ${filePath}\n`);
483
+ process.stderr.write(tCli('cli.run.batch_complete', lang));
484
+ process.stderr.write(tCli('cli.run.report_saved', lang, { path: filePath }));
556
485
  if (!values['no-serve'] && process.stdout.isTTY) {
557
486
  const { createReportServer } = await import('./server/report-server.js');
558
487
  const server = createReportServer({ reportsDir: config.outputDir });
559
488
  const serverUrl = await server.start();
560
- const reportUrl = `${serverUrl}/run/${report.id}`;
561
- process.stderr.write(`\n📊 Report server running at ${serverUrl}\n`);
562
- process.stderr.write(`👉 View report: ${reportUrl}\n`);
563
- process.stderr.write('\nPress Ctrl+C to stop the server\n');
489
+ const reportUrl = `${serverUrl}/reports/${report.id}`;
490
+ process.stderr.write(tCli('cli.run.report_server_running', lang, { url: serverUrl }));
491
+ process.stderr.write(tCli('cli.run.report_server_view', lang, { url: reportUrl }));
492
+ process.stderr.write(tCli('cli.run.report_server_stop', lang));
564
493
  const { platform } = await import('node:os');
565
494
  const openCmd = platform() === 'darwin' ? 'open' : platform() === 'win32' ? 'start' : 'xdg-open';
566
495
  const { execFile: execFileCb } = await import('node:child_process');
567
496
  execFileCb(openCmd, [reportUrl], () => { });
568
497
  }
569
498
  else if (!values['no-serve']) {
570
- process.stderr.write('\n💡 非交互环境,已跳过 report server\n');
571
- process.stderr.write(` 查看报告: omk bench report --reports-dir ${config.outputDir}\n`);
499
+ process.stderr.write(tCli('cli.run.no_serve_in_non_tty', lang));
500
+ process.stderr.write(tCli('cli.run.no_serve_view_hint', lang, { dir: config.outputDir }));
572
501
  }
573
502
  }
574
503
  return;
@@ -580,7 +509,7 @@ async function handleRun(argv) {
580
509
  ...config,
581
510
  repeat: repeatCount,
582
511
  onRepeatProgress({ run, total }) {
583
- process.stderr.write(`\n=== Run ${run}/${total} ===\n`);
512
+ process.stderr.write(tCli('cli.run.run_section', lang, { i: run, n: total }));
584
513
  },
585
514
  });
586
515
  report = result.report;
@@ -604,19 +533,22 @@ async function handleRun(argv) {
604
533
  if (out.result && out.gold) {
605
534
  process.stderr.write(formatGoldCompare(out.result, out.gold));
606
535
  if (out.result.contaminationWarning) {
607
- process.stderr.write(`\n⚠ ${out.result.contaminationWarning}\n`);
536
+ process.stderr.write(tCli('cli.run.contamination_warning', lang, {
537
+ warning: out.result.contaminationWarning,
538
+ }));
608
539
  }
609
540
  }
610
541
  else {
611
- process.stderr.write(`\n⚠ gold dataset 加载失败 (${goldDir}):\n`);
612
- for (const m of out.loadIssues)
613
- process.stderr.write(` - ${m}\n`);
542
+ process.stderr.write(tCli('cli.run.gold_load_failed', lang, { dir: goldDir }));
543
+ for (const m of out.loadIssues) {
544
+ process.stderr.write(tCli('cli.run.gold_load_issue', lang, { message: m }));
545
+ }
614
546
  }
615
547
  }
616
548
  console.log(JSON.stringify(report, null, 2));
617
549
  if (filePath) {
618
- process.stderr.write('\n✅ 评测完成\n');
619
- process.stderr.write(`📄 Report saved to: ${filePath}\n`);
550
+ process.stderr.write(tCli('cli.run.eval_complete', lang));
551
+ process.stderr.write(tCli('cli.run.report_saved', lang, { path: filePath }));
620
552
  if (!values['no-serve'] && process.stdout.isTTY) {
621
553
  // Auto-start report server
622
554
  const { createReportServer } = await import('./server/report-server.js');
@@ -624,10 +556,10 @@ async function handleRun(argv) {
624
556
  reportsDir: config.outputDir,
625
557
  });
626
558
  const serverUrl = await server.start();
627
- const reportUrl = `${serverUrl}/run/${report.id}`;
628
- process.stderr.write(`\n📊 Report server running at ${serverUrl}\n`);
629
- process.stderr.write(`👉 View report: ${reportUrl}\n`);
630
- process.stderr.write('\nPress Ctrl+C to stop the server\n');
559
+ const reportUrl = `${serverUrl}/reports/${report.id}`;
560
+ process.stderr.write(tCli('cli.run.report_server_running', lang, { url: serverUrl }));
561
+ process.stderr.write(tCli('cli.run.report_server_view', lang, { url: reportUrl }));
562
+ process.stderr.write(tCli('cli.run.report_server_stop', lang));
631
563
  // Auto-open report in browser
632
564
  const { platform } = await import('node:os');
633
565
  const openCmd = platform() === 'darwin' ? 'open' : platform() === 'win32' ? 'start' : 'xdg-open';
@@ -635,13 +567,13 @@ async function handleRun(argv) {
635
567
  execFileCb(openCmd, [reportUrl], () => { });
636
568
  }
637
569
  else if (!values['no-serve']) {
638
- process.stderr.write('\n💡 非交互环境,已跳过 report server\n');
639
- process.stderr.write(` 查看报告: omk bench report --reports-dir ${config.outputDir}\n`);
570
+ process.stderr.write(tCli('cli.run.no_serve_in_non_tty', lang));
571
+ process.stderr.write(tCli('cli.run.no_serve_view_hint', lang, { dir: config.outputDir }));
640
572
  }
641
573
  }
642
574
  }
643
575
  catch (err) {
644
- console.error(`Error: ${err.message}`);
576
+ console.error(tCli('cli.common.error_prefix', lang, { message: err.message }));
645
577
  process.exit(1);
646
578
  }
647
579
  }
@@ -649,9 +581,11 @@ async function handleRun(argv) {
649
581
  // handleReport
650
582
  // ---------------------------------------------------------------------------
651
583
  async function handleReport(argv) {
584
+ const lang = langFromArgv(argv);
652
585
  const { values } = parseArgs({
653
586
  args: argv,
654
587
  options: {
588
+ ...COMMON_OPTIONS,
655
589
  port: { type: 'string', default: '7799' },
656
590
  'reports-dir': { type: 'string', default: DEFAULT_REPORTS_DIR },
657
591
  export: { type: 'string' },
@@ -684,7 +618,7 @@ async function handleReport(argv) {
684
618
  const store = createFileStore(resolve(values['reports-dir']));
685
619
  const report = await store.get(values.export);
686
620
  if (!report) {
687
- console.error(`Report not found: ${values.export}`);
621
+ console.error(tCli('cli.common.report_not_found', lang, { id: values.export }));
688
622
  process.exit(1);
689
623
  }
690
624
  const html = report.each ? renderEachRunDetail(report) : renderRunDetail(report);
@@ -779,10 +713,12 @@ function parseLastWindow(spec) {
779
713
  return new Date(Date.now() - ms).toISOString();
780
714
  }
781
715
  async function handleAnalyze(argv) {
716
+ const lang = langFromArgv(argv);
782
717
  const { values, positionals } = parseArgs({
783
718
  args: argv,
784
719
  allowPositionals: true,
785
720
  options: {
721
+ ...COMMON_OPTIONS,
786
722
  kb: { type: 'string' },
787
723
  last: { type: 'string' },
788
724
  from: { type: 'string' },
@@ -793,7 +729,7 @@ async function handleAnalyze(argv) {
793
729
  });
794
730
  const dir = positionals[0];
795
731
  if (!dir) {
796
- console.error('Usage: omk analyze <dir> [--kb <path>] [--last 7d] [--from ISO] [--to ISO] [--skills name1,name2]');
732
+ console.error(tCli('cli.help.analyze_usage', lang));
797
733
  process.exit(1);
798
734
  }
799
735
  const tracePath = resolve(dir);
@@ -842,29 +778,32 @@ async function handleAnalyze(argv) {
842
778
  console.log(skillRows.join('\n'));
843
779
  console.log('');
844
780
  console.log(`report written to: ${jsonPath}`);
845
- console.log(`view in browser: omk bench report # 打开后点首页的 "📊 Skill 健康度日报"`);
781
+ console.log(tCli('cli.analyze.view_in_browser', lang));
846
782
  }
847
783
  async function handleInit(argv) {
784
+ const lang = langFromArgv(argv);
848
785
  const targetDir = resolve(argv[0] || '.');
849
786
  const { writeFileSync, mkdirSync } = await import('node:fs');
850
787
  mkdirSync(join(targetDir, 'skills'), { recursive: true });
851
788
  writeFileSync(join(targetDir, 'eval-samples.json'), INIT_SAMPLES);
852
789
  writeFileSync(join(targetDir, 'skills', 'v1.md'), INIT_SKILL_V1);
853
790
  writeFileSync(join(targetDir, 'skills', 'v2.md'), INIT_SKILL_V2);
854
- console.log(`Eval project scaffolded at: ${targetDir}`);
791
+ console.log(tCli('cli.init.scaffolded', lang, { dir: targetDir }));
855
792
  console.log('');
856
- console.log('Next steps:');
857
- console.log(' 1. Edit eval-samples.json to add your test cases');
858
- console.log(' 2. Edit skills/v1.md and skills/v2.md with your skill versions');
859
- console.log(' 3. Run: omk bench run --control v1 --treatment v2');
793
+ console.log(tCli('cli.init.next_steps_title', lang));
794
+ console.log(tCli('cli.init.next_step_edit_samples', lang));
795
+ console.log(tCli('cli.init.next_step_edit_skills', lang));
796
+ console.log(tCli('cli.init.next_step_run', lang));
860
797
  }
861
798
  // ---------------------------------------------------------------------------
862
799
  // handleGenSamples
863
800
  // ---------------------------------------------------------------------------
864
801
  async function handleGenSamples(argv) {
802
+ const lang = langFromArgv(argv);
865
803
  const { values } = parseArgs({
866
804
  args: argv,
867
805
  options: {
806
+ ...COMMON_OPTIONS,
868
807
  each: { type: 'boolean', default: false },
869
808
  count: { type: 'string', default: '5' },
870
809
  model: { type: 'string', default: 'sonnet' },
@@ -881,7 +820,7 @@ async function handleGenSamples(argv) {
881
820
  // Batch mode: generate for all skills missing eval-samples
882
821
  const skillDir = resolve(values['skill-dir']);
883
822
  if (!existsSync(skillDir)) {
884
- console.error(`Skill directory not found: ${skillDir}`);
823
+ console.error(tCli('cli.common.skill_dir_not_found', lang, { path: skillDir }));
885
824
  process.exit(1);
886
825
  }
887
826
  const { readdirSync, statSync } = await import('node:fs');
@@ -909,55 +848,63 @@ async function handleGenSamples(argv) {
909
848
  continue;
910
849
  }
911
850
  if (existsSync(samplesPath)) {
912
- process.stderr.write(`⏭️ ${name}: eval-samples 已存在,跳过\n`);
851
+ process.stderr.write(tCli('cli.gen.skill_skipped_existing', lang, { name }));
913
852
  continue;
914
853
  }
915
- process.stderr.write(`🔄 ${name}: 正在生成 ${count} 个测试样本...\n`);
854
+ process.stderr.write(tCli('cli.gen.skill_generating', lang, { name, count }));
916
855
  try {
917
856
  const skillContent = readFileSync(skillPath, 'utf-8');
918
857
  const { samples, costUSD } = await generateSamples({ skillContent, count, model });
919
858
  writeFileSync(samplesPath, JSON.stringify(samples, null, 2));
920
- process.stderr.write(`✅ ${name}: 已生成 ${samples.length} 个样本 → ${samplesPath} (${costUSD > 0 ? `$${costUSD.toFixed(4)}` : ''})\n`);
859
+ const cost = costUSD > 0 ? ` $${costUSD.toFixed(4)}` : '';
860
+ process.stderr.write(tCli('cli.gen.skill_done', lang, {
861
+ name, n: samples.length, path: samplesPath, cost,
862
+ }));
921
863
  generated++;
922
864
  }
923
865
  catch (err) {
924
- process.stderr.write(`❌ ${name}: ${err.message}\n`);
866
+ process.stderr.write(tCli('cli.gen.skill_failed', lang, {
867
+ name, message: err.message,
868
+ }));
925
869
  }
926
870
  }
927
871
  if (generated === 0) {
928
- console.log('没有需要生成的 eval-samples(所有 skill 已有配对文件)');
872
+ console.log(tCli('cli.gen.batch_none_needed', lang));
929
873
  }
930
874
  else {
931
- console.log(`\n共生成 ${generated} 份 eval-samples,请审查后运行: omk bench run --each`);
875
+ console.log(tCli('cli.gen.batch_summary', lang, { n: generated }));
932
876
  }
933
877
  }
934
878
  else {
935
879
  // Single skill mode
936
880
  const skillPath = argv.find((a) => !a.startsWith('-'));
937
881
  if (!skillPath) {
938
- console.error('请指定 skill 文件路径,例如: omk bench gen-samples skills/my-skill.md');
882
+ console.error(tCli('cli.gen.specify_skill_path', lang));
939
883
  process.exit(1);
940
884
  }
941
885
  const resolvedPath = resolve(skillPath);
942
886
  if (!existsSync(resolvedPath)) {
943
- console.error(`Skill file not found: ${resolvedPath}`);
887
+ console.error(tCli('cli.common.skill_file_not_found', lang, { path: resolvedPath }));
944
888
  process.exit(1);
945
889
  }
946
890
  const skillContent = readFileSync(resolvedPath, 'utf-8');
947
891
  const outputPath = resolve('eval-samples.json');
948
892
  if (existsSync(outputPath)) {
949
- console.error(`eval-samples.json 已存在。如需覆盖请先删除。`);
893
+ console.error(tCli('cli.gen.samples_already_exists', lang));
950
894
  process.exit(1);
951
895
  }
952
- process.stderr.write(`🔄 正在生成 ${count} 个测试样本...\n`);
896
+ process.stderr.write(tCli('cli.gen.single_generating', lang, { count }));
953
897
  try {
954
898
  const { samples, costUSD } = await generateSamples({ skillContent, count, model });
955
899
  writeFileSync(outputPath, JSON.stringify(samples, null, 2));
956
- process.stderr.write(`✅ 已生成 ${samples.length} 个样本 → ${outputPath} (${costUSD > 0 ? `$${costUSD.toFixed(4)}` : ''})\n`);
957
- console.log('\n请审查生成的测试样本后运行: omk bench run');
900
+ const cost = costUSD > 0 ? ` $${costUSD.toFixed(4)}` : '';
901
+ process.stderr.write(tCli('cli.gen.single_done', lang, {
902
+ n: samples.length, path: outputPath, cost,
903
+ }));
904
+ console.log(tCli('cli.gen.review_hint', lang));
958
905
  }
959
906
  catch (err) {
960
- console.error(`生成失败: ${err.message}`);
907
+ console.error(tCli('cli.gen.failed', lang, { message: err.message }));
961
908
  process.exit(1);
962
909
  }
963
910
  }
@@ -966,9 +913,11 @@ async function handleGenSamples(argv) {
966
913
  // handleEvolve
967
914
  // ---------------------------------------------------------------------------
968
915
  async function handleEvolve(argv) {
916
+ const lang = langFromArgv(argv);
969
917
  const { values } = parseArgs({
970
918
  args: argv,
971
919
  options: {
920
+ ...COMMON_OPTIONS,
972
921
  rounds: { type: 'string', default: '5' },
973
922
  target: { type: 'string' },
974
923
  samples: { type: 'string', default: 'eval-samples.json' },
@@ -985,7 +934,7 @@ async function handleEvolve(argv) {
985
934
  });
986
935
  const skillPath = argv.find((a) => !a.startsWith('-'));
987
936
  if (!skillPath) {
988
- console.error('请指定 skill 文件路径,例如: omk bench evolve skills/my-skill.md');
937
+ console.error(tCli('cli.evolve.specify_skill_path', lang));
989
938
  process.exit(1);
990
939
  }
991
940
  let samplesFile = values.samples ?? 'eval-samples.json';
@@ -996,7 +945,7 @@ async function handleEvolve(argv) {
996
945
  samplesFile = 'eval-samples.yml';
997
946
  }
998
947
  const { evolveSkill } = await import('./authoring/evolver.js');
999
- process.stderr.write(`\n=== Evolution: ${skillPath} ===\n`);
948
+ process.stderr.write(tCli('cli.evolve.section_header', lang, { path: skillPath }));
1000
949
  try {
1001
950
  const result = await evolveSkill({
1002
951
  skillPath: resolve(skillPath),
@@ -1010,59 +959,93 @@ async function handleEvolve(argv) {
1010
959
  concurrency: Math.max(1, Number(values.concurrency) || 1),
1011
960
  timeoutMs: Math.max(1, Number(values.timeout) || 120) * 1000,
1012
961
  skipPreflight: values['skip-preflight'],
1013
- onProgress: defaultOnProgress,
962
+ onProgress: makeOnProgress(lang),
1014
963
  onRoundProgress({ round, totalRounds: _totalRounds, phase, score, delta, accepted, costUSD, error }) {
1015
964
  if (phase === 'baseline') {
1016
- process.stderr.write(`Round 0 (baseline): score=${score.toFixed(2)} ($${costUSD.toFixed(4)})\n`);
965
+ process.stderr.write(tCli('cli.evolve.round_baseline', lang, {
966
+ score: score.toFixed(2), cost: costUSD.toFixed(4),
967
+ }));
1017
968
  }
1018
969
  else if (phase === 'error') {
1019
- process.stderr.write(`Round ${round}: ✗ 改进生成失败: ${error}\n`);
970
+ process.stderr.write(tCli('cli.evolve.round_error', lang, {
971
+ round, error: String(error ?? ''),
972
+ }));
1020
973
  }
1021
974
  else if (phase === 'done') {
1022
- const deltaStr = delta >= 0 ? `+${delta.toFixed(2)}` : delta.toFixed(2);
975
+ const delta_ = delta >= 0 ? `+${delta.toFixed(2)}` : delta.toFixed(2);
1023
976
  const status = accepted ? '✓ ACCEPT' : '✗ REJECT';
1024
- process.stderr.write(`Round ${round}: score=${score.toFixed(2)} (${deltaStr}) ${status} ($${costUSD.toFixed(4)})\n`);
977
+ process.stderr.write(tCli('cli.evolve.round_done', lang, {
978
+ round, score: score.toFixed(2), delta: delta_, status, cost: costUSD.toFixed(4),
979
+ }));
1025
980
  }
1026
981
  },
1027
982
  });
1028
983
  const improvement = result.startScore > 0
1029
984
  ? ((result.finalScore - result.startScore) / result.startScore * 100).toFixed(1)
1030
985
  : '0';
1031
- process.stderr.write(`\n✅ ${result.startScore.toFixed(2)} → ${result.finalScore.toFixed(2)} (+${improvement}%) | ${result.totalRounds} 轮 | $${result.totalCostUSD.toFixed(4)}\n`);
1032
- process.stderr.write(`Best: ${result.bestSkillPath} → ${resolve(skillPath)}\n`);
1033
- process.stderr.write(`所有版本保存在: ${join(resolve(skillPath, '..'), 'evolve')}/\n`);
986
+ process.stderr.write(tCli('cli.evolve.summary', lang, {
987
+ start: result.startScore.toFixed(2), final: result.finalScore.toFixed(2),
988
+ percent: improvement, rounds: result.totalRounds, cost: result.totalCostUSD.toFixed(4),
989
+ }));
990
+ process.stderr.write(tCli('cli.evolve.best_path', lang, {
991
+ best: result.bestSkillPath, target: resolve(skillPath),
992
+ }));
993
+ process.stderr.write(tCli('cli.evolve.versions_saved', lang, {
994
+ dir: join(resolve(skillPath, '..'), 'evolve'),
995
+ }));
1034
996
  if (result.reportId) {
1035
- process.stderr.write(`📊 评测报告: omk bench report (ID: ${result.reportId})\n`);
997
+ process.stderr.write(tCli('cli.evolve.report_link', lang, { id: result.reportId }));
1036
998
  }
1037
999
  console.log(JSON.stringify(result, null, 2));
1038
1000
  }
1039
1001
  catch (err) {
1040
- console.error(`Error: ${err.message}`);
1002
+ console.error(tCli('cli.common.error_prefix', lang, { message: err.message }));
1041
1003
  process.exit(1);
1042
1004
  }
1043
1005
  }
1044
1006
  // ---------------------------------------------------------------------------
1045
- // handleCi
1007
+ // handleGate — 跑评测 + 应用 gate, exit code 0/1 适合 CI/CD pipeline 调用。
1008
+ // 内部 = runEvaluation + computeVerdict + formatVerdictText, 与 bench verdict
1009
+ // 共用决策内核(只是 verdict 读已有报告, gate 跑完再判)。
1046
1010
  // ---------------------------------------------------------------------------
1047
- async function handleCi(argv) {
1011
+ async function handleGate(argv) {
1012
+ const lang = langFromArgv(argv);
1013
+ if (argv[0] === '--help' || argv[0] === '-h') {
1014
+ console.log(tCli('cli.help.main', lang).trim());
1015
+ process.exit(0);
1016
+ }
1048
1017
  const { values, config } = parseRunConfig(argv, {
1049
1018
  threshold: { type: 'string', default: '3.5' },
1019
+ 'trivial-diff': { type: 'string' },
1050
1020
  });
1051
1021
  const { runEvaluation } = await import('./eval-workflows/run-evaluation.js');
1052
- config.onProgress = defaultOnProgress;
1022
+ config.onProgress = makeOnProgress(lang);
1053
1023
  try {
1054
1024
  const { report } = (await runEvaluation(config));
1055
- const threshold = Number(values.threshold);
1056
1025
  if (report.dryRun) {
1057
- console.log('CI dry-run: no scores to check');
1026
+ console.log('Gate dry-run: no scores to check');
1027
+ process.exit(0);
1028
+ }
1029
+ // gate 内核 = run + verdict, 自动覆盖 omk 全部决策维度(三层 layer-gate /
1030
+ // bootstrap diff CI / saturation / Krippendorff α)。computeVerdict 是单一
1031
+ // 决策源, exit code 跟 verdict.level 走 — 数据 underpowered 直接 FAIL,
1032
+ // 堵住"过 PASS 就 deploy"的漏洞。
1033
+ const { computeVerdict, formatVerdictText } = await import('./eval-core/verdict.js');
1034
+ const result = computeVerdict(report, {
1035
+ gateThreshold: Number(values.threshold),
1036
+ triviallySmallDiff: values['trivial-diff'] != null ? Number(values['trivial-diff']) : undefined,
1037
+ });
1038
+ console.log(formatVerdictText(result, { verbose: true }));
1039
+ // exit code 与 handleVerdict 对齐:只有 PROGRESS / SOLO-pass 才 0,
1040
+ // NOISE / UNDERPOWERED / CAUTIOUS / REGRESS 全 1。pipeline `omk bench gate
1041
+ // && deploy` 数据不显著就不会误 deploy。
1042
+ if (result.level === 'PROGRESS') {
1058
1043
  process.exit(0);
1059
1044
  }
1060
- // three-gate 逻辑抽到 src/eval-core/ci-gates.ts 作纯函数,便于测试;此处只做 IO。
1061
- const { evaluateCiGates } = await import('./eval-core/ci-gates.js');
1062
- const { allPass, lines } = evaluateCiGates(report.summary || {}, threshold);
1063
- for (const line of lines)
1064
- console.log(line);
1065
- process.exit(allPass ? 0 : 1);
1045
+ if (result.level === 'SOLO' && result.headline.includes('PASS')) {
1046
+ process.exit(0);
1047
+ }
1048
+ process.exit(1);
1066
1049
  }
1067
1050
  catch (err) {
1068
1051
  console.error(`Error: ${err.message}`);
@@ -1073,6 +1056,7 @@ async function handleCi(argv) {
1073
1056
  // handleDiff
1074
1057
  // ---------------------------------------------------------------------------
1075
1058
  async function handleDiff(argv) {
1059
+ const lang = langFromArgv(argv);
1076
1060
  // Flag-aware split: separate positional report IDs from flags so we can support
1077
1061
  // omk bench diff <id> — within-report sample-level (v0.22)
1078
1062
  // omk bench diff <id1> <id2> — cross-report variant-level (legacy)
@@ -1094,22 +1078,13 @@ async function handleDiff(argv) {
1094
1078
  }
1095
1079
  }
1096
1080
  if (positional.length === 0) {
1097
- console.error([
1098
- 'Usage:',
1099
- ' omk bench diff <reportId> within-report per-sample diff (v0.22)',
1100
- ' omk bench diff <reportId1> <reportId2> cross-report variant-level diff',
1101
- '',
1102
- 'Options:',
1103
- ' --regressions-only 只列 treatment < control 的样本',
1104
- ' --threshold <num> regression 阈值 (default 0,即任一负 Δ 算回退)',
1105
- ' --variant <name> within-report 模式下指定要钻取的 variant (default: variants[1])',
1106
- ' --top <n> 只列差距最大的前 N 个样本',
1107
- ].join('\n'));
1081
+ console.error(tCli('cli.help.diff_usage', lang));
1108
1082
  process.exit(positional.length === 0 ? 1 : 0);
1109
1083
  }
1110
1084
  const { values } = parseArgs({
1111
1085
  args: flagArgs,
1112
1086
  options: {
1087
+ ...COMMON_OPTIONS,
1113
1088
  'regressions-only': { type: 'boolean', default: false },
1114
1089
  threshold: { type: 'string' },
1115
1090
  variant: { type: 'string' },
@@ -1120,18 +1095,18 @@ async function handleDiff(argv) {
1120
1095
  const { createFileStore } = await import('./server/report-store.js');
1121
1096
  const store = createFileStore(resolve(DEFAULT_REPORTS_DIR));
1122
1097
  if (positional.length === 1) {
1123
- await runSampleLevelDiff(positional[0], store, values);
1098
+ await runSampleLevelDiff(positional[0], store, values, lang);
1124
1099
  return;
1125
1100
  }
1126
1101
  const [id1, id2] = positional;
1127
1102
  const r1 = await store.get(id1);
1128
1103
  const r2 = await store.get(id2);
1129
1104
  if (!r1) {
1130
- console.error(`Report not found: ${id1}`);
1105
+ console.error(tCli('cli.common.report_not_found', lang, { id: id1 }));
1131
1106
  process.exit(1);
1132
1107
  }
1133
1108
  if (!r2) {
1134
- console.error(`Report not found: ${id2}`);
1109
+ console.error(tCli('cli.common.report_not_found', lang, { id: id2 }));
1135
1110
  process.exit(1);
1136
1111
  }
1137
1112
  console.log(`\n Diff: ${id1} → ${id2}\n`);
@@ -1187,10 +1162,10 @@ async function handleDiff(argv) {
1187
1162
  * Default focus is variants[0] (control) vs variants[1] (treatment), but
1188
1163
  * `--variant` overrides which variant is the "treatment" side.
1189
1164
  */
1190
- async function runSampleLevelDiff(reportId, store, flags) {
1165
+ async function runSampleLevelDiff(reportId, store, flags, lang) {
1191
1166
  const report = await store.get(reportId);
1192
1167
  if (!report) {
1193
- console.error(`Report not found: ${reportId}`);
1168
+ console.error(tCli('cli.common.report_not_found', lang, { id: reportId }));
1194
1169
  process.exit(1);
1195
1170
  }
1196
1171
  const variants = report.meta?.variants ?? [];
@@ -1267,27 +1242,18 @@ async function runSampleLevelDiff(reportId, store, flags) {
1267
1242
  // handleGold — gold dataset workflow (init / validate / compare)
1268
1243
  // ---------------------------------------------------------------------------
1269
1244
  async function handleGold(argv) {
1245
+ const lang = langFromArgv(argv);
1270
1246
  const sub = argv[0];
1271
1247
  const rest = argv.slice(1);
1272
1248
  if (!sub || sub === '--help' || sub === '-h') {
1273
- console.log([
1274
- '',
1275
- 'Usage: omk bench gold <subcommand>',
1276
- '',
1277
- 'Subcommands:',
1278
- ' init [--out <dir>] [--annotator <id>] 生成空白 gold dataset 模板',
1279
- ' validate <dir> 校验数据集结构',
1280
- ' compare <reportId> --gold-dir <dir> 与已有 report 计算 α/κ/Pearson',
1281
- ' [--variant <name>] [--reports-dir <d>]',
1282
- ' [--bootstrap-samples N] [--seed N]',
1283
- '',
1284
- ].join('\n'));
1249
+ console.log(tCli('cli.help.gold', lang));
1285
1250
  process.exit(sub ? 0 : 1);
1286
1251
  }
1287
1252
  if (sub === 'init') {
1288
1253
  const { values } = parseArgs({
1289
1254
  args: rest,
1290
1255
  options: {
1256
+ ...COMMON_OPTIONS,
1291
1257
  out: { type: 'string', default: './gold-dataset' },
1292
1258
  annotator: { type: 'string' },
1293
1259
  },
@@ -1298,10 +1264,12 @@ async function handleGold(argv) {
1298
1264
  const written = initGoldDataset(values.out, {
1299
1265
  annotator: values.annotator,
1300
1266
  });
1301
- console.log(`Created ${written.length} files in ${values.out}:`);
1267
+ console.log(tCli('cli.gold.created_files', lang, {
1268
+ n: written.length, dir: values.out,
1269
+ }));
1302
1270
  for (const p of written)
1303
1271
  console.log(` ${p}`);
1304
- console.log('\n下一步: 编辑 annotations.yaml 加入真实标注 → 跑 omk bench gold validate');
1272
+ console.log(tCli('cli.gold.next_step_edit_annotations', lang));
1305
1273
  }
1306
1274
  catch (err) {
1307
1275
  console.error(err.message);
@@ -1312,13 +1280,13 @@ async function handleGold(argv) {
1312
1280
  if (sub === 'validate') {
1313
1281
  const dir = rest[0];
1314
1282
  if (!dir) {
1315
- console.error('Usage: omk bench gold validate <dir>');
1283
+ console.error(tCli('cli.common.usage_gold_validate', lang));
1316
1284
  process.exit(1);
1317
1285
  }
1318
1286
  const { validateGoldDataset } = await import('./grading/gold-cli.js');
1319
1287
  const result = validateGoldDataset(dir);
1320
1288
  if (result.ok) {
1321
- console.log(`✓ gold dataset OK — ${result.sampleCount} 条标注`);
1289
+ console.log(tCli('cli.gold.validate_ok', lang, { n: result.sampleCount }));
1322
1290
  return;
1323
1291
  }
1324
1292
  console.error(`✗ gold dataset has ${result.issues.length} issue(s):`);
@@ -1335,6 +1303,7 @@ async function handleGold(argv) {
1335
1303
  const { values } = parseArgs({
1336
1304
  args: rest.slice(1),
1337
1305
  options: {
1306
+ ...COMMON_OPTIONS,
1338
1307
  'gold-dir': { type: 'string' },
1339
1308
  variant: { type: 'string' },
1340
1309
  'reports-dir': { type: 'string', default: DEFAULT_REPORTS_DIR },
@@ -1366,7 +1335,7 @@ async function handleGold(argv) {
1366
1335
  const store = createFileStore(resolve(values['reports-dir']));
1367
1336
  const report = await store.get(reportId);
1368
1337
  if (!report) {
1369
- console.error(`Report not found: ${reportId}`);
1338
+ console.error(tCli('cli.common.report_not_found', lang, { id: reportId }));
1370
1339
  process.exit(1);
1371
1340
  }
1372
1341
  const samples = Math.max(100, Number(values['bootstrap-samples']) || 1000);
@@ -1388,27 +1357,11 @@ async function handleGold(argv) {
1388
1357
  // handleDebiasValidate — measure length-debias prompt sensitivity (Phase 3a)
1389
1358
  // ---------------------------------------------------------------------------
1390
1359
  async function handleDebiasValidate(argv) {
1360
+ const lang = langFromArgv(argv);
1391
1361
  const sub = argv[0];
1392
1362
  const rest = argv.slice(1);
1393
1363
  if (!sub || sub === '--help' || sub === '-h') {
1394
- console.log([
1395
- '',
1396
- 'Usage: omk bench debias-validate <kind> <reportId> [options]',
1397
- '',
1398
- 'Kinds:',
1399
- ' length re-judge with the opposite length-debias setting and bootstrap CI',
1400
- ' on the score diff. Cost ~doubles vs the original judge pass.',
1401
- '',
1402
- 'Options:',
1403
- ' --reports-dir <dir> report store dir (default: ~/.oh-my-knowledge/reports)',
1404
- ' --samples <path> override samples file (default: from report.meta.request)',
1405
- ' --variant <name> which variant to validate (default: first)',
1406
- ' --judge-executor <name> executor for judge calls (default: claude)',
1407
- ' --judge-model <model> judge model id (default: from report)',
1408
- ' --bootstrap-samples N bootstrap iterations (default 1000)',
1409
- ' --seed N deterministic CI seed',
1410
- '',
1411
- ].join('\n'));
1364
+ console.log(tCli('cli.help.debias_validate', lang));
1412
1365
  process.exit(sub ? 0 : 1);
1413
1366
  }
1414
1367
  if (sub !== 'length') {
@@ -1423,6 +1376,7 @@ async function handleDebiasValidate(argv) {
1423
1376
  const { values } = parseArgs({
1424
1377
  args: rest.slice(1),
1425
1378
  options: {
1379
+ ...COMMON_OPTIONS,
1426
1380
  'reports-dir': { type: 'string', default: DEFAULT_REPORTS_DIR },
1427
1381
  samples: { type: 'string' },
1428
1382
  variant: { type: 'string' },
@@ -1437,7 +1391,7 @@ async function handleDebiasValidate(argv) {
1437
1391
  const store = createFileStore(resolve(values['reports-dir']));
1438
1392
  const report = await store.get(reportId);
1439
1393
  if (!report) {
1440
- console.error(`Report not found: ${reportId}`);
1394
+ console.error(tCli('cli.common.report_not_found', lang, { id: reportId }));
1441
1395
  process.exit(1);
1442
1396
  }
1443
1397
  // Resolve samples path: --samples overrides; otherwise read from report.meta.request.
@@ -1452,10 +1406,10 @@ async function handleDebiasValidate(argv) {
1452
1406
  const judgeModel = values['judge-model']
1453
1407
  ?? report.meta?.judgeModel;
1454
1408
  if (!judgeModel) {
1455
- console.error('No judge model. Pass --judge-model <id> or ensure report has meta.judgeModel.');
1409
+ console.error(tCli('cli.common.no_judge_model', lang));
1456
1410
  process.exit(1);
1457
1411
  }
1458
- process.stderr.write('\n⚠ debias-validate 会重判所有 (sample × variant),judge cost 大致翻倍。\n');
1412
+ process.stderr.write(tCli('cli.debias.warn_cost_doubles', lang));
1459
1413
  const { createExecutor } = await import('./executors/index.js');
1460
1414
  const judgeExecutor = createExecutor(values['judge-executor']);
1461
1415
  const { validateLengthDebias, formatDebiasValidate } = await import('./grading/debias-validate.js');
@@ -1479,29 +1433,16 @@ async function handleDebiasValidate(argv) {
1479
1433
  // handleSaturation — re-compute saturation verdict from a finished report
1480
1434
  // ---------------------------------------------------------------------------
1481
1435
  async function handleSaturation(argv) {
1436
+ const lang = langFromArgv(argv);
1482
1437
  const reportId = argv[0];
1483
1438
  if (!reportId || reportId === '--help' || reportId === '-h') {
1484
- console.log([
1485
- '',
1486
- 'Usage: omk bench saturation <reportId> [options]',
1487
- '',
1488
- '回答"我跑够样本了吗?"。复述已有 report 中持久化的饱和判定。',
1489
- '',
1490
- '注:本命令读取 run 时跑出的 verdict(运行时已用 method=bootstrap-ci-width',
1491
- '默认阈值 + 3 窗口 持续条件)。如要换 method/threshold 重新计算,需要重跑',
1492
- '`omk bench run --repeat ≥ 5`(运行时持久化的 trace 不含原始分数,无法',
1493
- '在事后用其他参数复算)。',
1494
- '',
1495
- 'Options:',
1496
- ' --reports-dir <dir> report store dir (default: ~/.oh-my-knowledge/reports)',
1497
- ' --variant <name> 只看一个 variant (default: all)',
1498
- '',
1499
- ].join('\n'));
1439
+ console.log(tCli('cli.help.saturation', lang));
1500
1440
  process.exit(reportId ? 0 : 1);
1501
1441
  }
1502
1442
  const { values } = parseArgs({
1503
1443
  args: argv.slice(1),
1504
1444
  options: {
1445
+ ...COMMON_OPTIONS,
1505
1446
  'reports-dir': { type: 'string', default: DEFAULT_REPORTS_DIR },
1506
1447
  variant: { type: 'string' },
1507
1448
  },
@@ -1511,12 +1452,12 @@ async function handleSaturation(argv) {
1511
1452
  const store = createFileStore(resolve(values['reports-dir']));
1512
1453
  const report = await store.get(reportId);
1513
1454
  if (!report) {
1514
- console.error(`Report not found: ${reportId}`);
1455
+ console.error(tCli('cli.common.report_not_found', lang, { id: reportId }));
1515
1456
  process.exit(1);
1516
1457
  }
1517
1458
  const saturation = report.variance?.saturation;
1518
1459
  if (!saturation) {
1519
- console.error('该 report 无 saturation 数据 (需要 --repeat ≥ 2 才会记录)。');
1460
+ console.error(tCli('cli.saturation.no_data', lang));
1520
1461
  process.exit(1);
1521
1462
  }
1522
1463
  // Print the persisted verdict from the original run. The trace stores
@@ -1527,22 +1468,32 @@ async function handleSaturation(argv) {
1527
1468
  // run time to enable post-hoc parameter sweeps.
1528
1469
  const variants = report.meta.variants ?? [];
1529
1470
  const targetVariants = values.variant ? [values.variant] : variants;
1530
- console.log(`\n Saturation verdict (复述持久化结果)\n`);
1471
+ console.log(tCli('cli.saturation.verdict_header', lang));
1531
1472
  for (const variant of targetVariants) {
1532
1473
  const trace = saturation.perVariant[variant];
1533
1474
  if (!trace || trace.length === 0) {
1534
- console.log(` ${variant}: 无 trace 数据`);
1475
+ console.log(tCli('cli.saturation.variant_no_trace', lang, { variant }));
1535
1476
  continue;
1536
1477
  }
1537
- console.log(` ${variant}:`);
1538
- console.log(` checkpoints: ${trace.length} (N=${trace.map((p) => p.n).join(', ')})`);
1539
- console.log(` 最近一点 mean=${trace[trace.length - 1].mean.toFixed(3)}, CI=[${trace[trace.length - 1].ciLow.toFixed(3)}, ${trace[trace.length - 1].ciHigh.toFixed(3)}]`);
1478
+ console.log(tCli('cli.saturation.variant_label', lang, { variant }));
1479
+ console.log(tCli('cli.saturation.checkpoints', lang, {
1480
+ n: trace.length, list: trace.map((p) => p.n).join(', '),
1481
+ }));
1482
+ const last = trace[trace.length - 1];
1483
+ console.log(tCli('cli.saturation.last_point', lang, {
1484
+ mean: last.mean.toFixed(3), lo: last.ciLow.toFixed(3), hi: last.ciHigh.toFixed(3),
1485
+ }));
1540
1486
  if (saturation.verdicts?.[variant]) {
1541
1487
  const v = saturation.verdicts[variant];
1542
- console.log(` 持久化判定 (${v.method}): ${v.saturated ? `已饱和@N=${v.atN}` : '未饱和'} - ${v.reason}`);
1488
+ const result = v.saturated
1489
+ ? tCli('cli.saturation.persisted_verdict_saturated', lang, { n: v.atN ?? '?' })
1490
+ : tCli('cli.saturation.persisted_verdict_unsaturated', lang);
1491
+ console.log(tCli('cli.saturation.persisted_verdict', lang, {
1492
+ method: v.method, result, reason: v.reason,
1493
+ }));
1543
1494
  }
1544
1495
  else if (trace.length < 5) {
1545
- console.log(` 判定: 数据点 ${trace.length} < 5,跳过 (跑 --repeat 5 以上才输出)`);
1496
+ console.log(tCli('cli.saturation.skipped_too_few_points', lang, { n: trace.length }));
1546
1497
  }
1547
1498
  }
1548
1499
  console.log('');
@@ -1551,34 +1502,16 @@ async function handleSaturation(argv) {
1551
1502
  // handleVerdict — one-line ship/no-ship verdict (v0.22)
1552
1503
  // ---------------------------------------------------------------------------
1553
1504
  async function handleVerdict(argv) {
1505
+ const lang = langFromArgv(argv);
1554
1506
  const reportId = argv[0];
1555
1507
  if (!reportId || reportId === '--help' || reportId === '-h') {
1556
- console.log([
1557
- '',
1558
- 'Usage: omk bench verdict <reportId> [options]',
1559
- '',
1560
- '聚合 bootstrap CI / 三层 ci-gate / saturation / human α 给出一行结论。',
1561
- '',
1562
- 'Verdict 等级:',
1563
- ' PROGRESS 显著改进 + 三层全过',
1564
- ' CAUTIOUS 改进真实但有警告 (gate 破 / 幅度太小 / 控制组本身崩)',
1565
- ' REGRESS 显著回退 — 不要 ship',
1566
- ' NOISE CI 跨 0,无法判定',
1567
- ' UNDERPOWERED 样本不足,需要扩 N 重测',
1568
- ' SOLO 单变体报告,无对比对象',
1569
- '',
1570
- 'Options:',
1571
- ' --reports-dir <dir> report store dir (default: ~/.oh-my-knowledge/reports)',
1572
- ' --threshold <num> 三层 gate 阈值 (default 3.5,匹配 omk bench ci)',
1573
- ' --trivial-diff <num> "幅度太小"阈值 (default 0.1)',
1574
- ' --verbose 展开 per-pair 详情',
1575
- '',
1576
- ].join('\n'));
1508
+ console.log(tCli('cli.help.verdict', lang));
1577
1509
  process.exit(reportId ? 0 : 1);
1578
1510
  }
1579
1511
  const { values } = parseArgs({
1580
1512
  args: argv.slice(1),
1581
1513
  options: {
1514
+ ...COMMON_OPTIONS,
1582
1515
  'reports-dir': { type: 'string', default: DEFAULT_REPORTS_DIR },
1583
1516
  threshold: { type: 'string' },
1584
1517
  'trivial-diff': { type: 'string' },
@@ -1590,12 +1523,12 @@ async function handleVerdict(argv) {
1590
1523
  const store = createFileStore(resolve(values['reports-dir']));
1591
1524
  const report = await store.get(reportId);
1592
1525
  if (!report) {
1593
- console.error(`Report not found: ${reportId}`);
1526
+ console.error(tCli('cli.common.report_not_found', lang, { id: reportId }));
1594
1527
  process.exit(1);
1595
1528
  }
1596
1529
  const { computeVerdict, formatVerdictText } = await import('./eval-core/verdict.js');
1597
1530
  const result = computeVerdict(report, {
1598
- ciThreshold: values.threshold != null ? Number(values.threshold) : undefined,
1531
+ gateThreshold: values.threshold != null ? Number(values.threshold) : undefined,
1599
1532
  triviallySmallDiff: values['trivial-diff'] != null ? Number(values['trivial-diff']) : undefined,
1600
1533
  });
1601
1534
  console.log(formatVerdictText(result, { verbose: Boolean(values.verbose) }));
@@ -1614,31 +1547,16 @@ async function handleVerdict(argv) {
1614
1547
  // handleDiagnose — per-sample quality diagnostics (v0.23 A)
1615
1548
  // ---------------------------------------------------------------------------
1616
1549
  async function handleDiagnose(argv) {
1550
+ const lang = langFromArgv(argv);
1617
1551
  const reportId = argv[0];
1618
1552
  if (!reportId || reportId === '--help' || reportId === '-h') {
1619
- console.log([
1620
- '',
1621
- 'Usage: omk bench diagnose <reportId> [options]',
1622
- '',
1623
- '诊断样本集本身的质量问题:区分度低 / 重复 / 歧义 / 成本异常 / 全 fail。',
1624
- '回答"测评结论是否被坏样本污染"——与 omk bench verdict 互补。',
1625
- '',
1626
- 'Options:',
1627
- ' --reports-dir <dir> report store dir',
1628
- ' --samples <path> 样本文件路径 (用于 near-duplicate 检测;默认从 report.meta.request 读)',
1629
- ' --top <n> 每类只显示前 N 个 (默认 10,0=全部)',
1630
- ' --duplicate-rouge <num> near-duplicate ROUGE-1 阈值 (默认 0.7)',
1631
- ' --ambiguous-stddev <num> 歧义阈值,judge stddev (默认 1.0,需要 --judge-repeat ≥ 2 数据)',
1632
- ' --cost-k <num> 成本异常倍数 vs median (默认 3)',
1633
- ' --latency-k <num> 耗时异常倍数 vs median (默认 3)',
1634
- ' --flat <num> flat_scores 分差阈值 (默认 0.5)',
1635
- '',
1636
- ].join('\n'));
1553
+ console.log(tCli('cli.help.diagnose', lang));
1637
1554
  process.exit(reportId ? 0 : 1);
1638
1555
  }
1639
1556
  const { values } = parseArgs({
1640
1557
  args: argv.slice(1),
1641
1558
  options: {
1559
+ ...COMMON_OPTIONS,
1642
1560
  'reports-dir': { type: 'string', default: DEFAULT_REPORTS_DIR },
1643
1561
  samples: { type: 'string' },
1644
1562
  top: { type: 'string', default: '10' },
@@ -1654,7 +1572,7 @@ async function handleDiagnose(argv) {
1654
1572
  const store = createFileStore(resolve(values['reports-dir']));
1655
1573
  const report = await store.get(reportId);
1656
1574
  if (!report) {
1657
- console.error(`Report not found: ${reportId}`);
1575
+ console.error(tCli('cli.common.report_not_found', lang, { id: reportId }));
1658
1576
  process.exit(1);
1659
1577
  }
1660
1578
  // Try to read the samples file for near-duplicate detection. Source order:
@@ -1669,7 +1587,9 @@ async function handleDiagnose(argv) {
1669
1587
  samples = loadSamples(samplesPath).samples;
1670
1588
  }
1671
1589
  catch (err) {
1672
- process.stderr.write(`warn: 加载 samples 文件失败 (${samplesPath}): ${err.message}\n`);
1590
+ process.stderr.write(tCli('cli.common.warn_load_samples_failed', lang, {
1591
+ path: samplesPath, message: err.message,
1592
+ }));
1673
1593
  }
1674
1594
  }
1675
1595
  const topRaw = Number(values.top);
@@ -1694,29 +1614,16 @@ async function handleDiagnose(argv) {
1694
1614
  // handleFailures — LLM-driven failure clustering (v0.23 B)
1695
1615
  // ---------------------------------------------------------------------------
1696
1616
  async function handleFailures(argv) {
1617
+ const lang = langFromArgv(argv);
1697
1618
  const reportId = argv[0];
1698
1619
  if (!reportId || reportId === '--help' || reportId === '-h') {
1699
- console.log([
1700
- '',
1701
- 'Usage: omk bench failures <reportId> [options]',
1702
- '',
1703
- '把已有 report 的失败样本喂给一次 LLM 调用,自动聚类 + 给修复建议。',
1704
- '失败定义:compositeScore < threshold 或 ok=false。',
1705
- '',
1706
- 'Options:',
1707
- ' --reports-dir <dir> report store dir',
1708
- ' --judge-executor <name> 执行器 (default: claude)',
1709
- ' --judge-model <id> 聚类用的 model (default: 沿用 report.meta.judgeModel)',
1710
- ' --max-clusters <n> 最多多少 cluster (default 5)',
1711
- ' --threshold <num> compositeScore < threshold 算失败 (default 3)',
1712
- ' --max-feed <n> 最多喂给 LLM 多少条 (default 50,超出取最差)',
1713
- '',
1714
- ].join('\n'));
1620
+ console.log(tCli('cli.help.failures', lang));
1715
1621
  process.exit(reportId ? 0 : 1);
1716
1622
  }
1717
1623
  const { values } = parseArgs({
1718
1624
  args: argv.slice(1),
1719
1625
  options: {
1626
+ ...COMMON_OPTIONS,
1720
1627
  'reports-dir': { type: 'string', default: DEFAULT_REPORTS_DIR },
1721
1628
  'judge-executor': { type: 'string', default: 'claude' },
1722
1629
  'judge-model': { type: 'string' },
@@ -1730,12 +1637,12 @@ async function handleFailures(argv) {
1730
1637
  const store = createFileStore(resolve(values['reports-dir']));
1731
1638
  const report = await store.get(reportId);
1732
1639
  if (!report) {
1733
- console.error(`Report not found: ${reportId}`);
1640
+ console.error(tCli('cli.common.report_not_found', lang, { id: reportId }));
1734
1641
  process.exit(1);
1735
1642
  }
1736
1643
  const judgeModel = values['judge-model'] ?? report.meta?.judgeModel;
1737
1644
  if (!judgeModel) {
1738
- console.error('No judge model. Pass --judge-model <id> or ensure report has meta.judgeModel.');
1645
+ console.error(tCli('cli.common.no_judge_model', lang));
1739
1646
  process.exit(1);
1740
1647
  }
1741
1648
  const { createExecutor } = await import('./executors/index.js');