oh-my-knowledge 0.20.1 → 0.22.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +10 -4
- package/README.zh.md +9 -4
- package/dist/src/analysis/coverage-analyzer.d.ts +1 -1
- package/dist/src/analysis/coverage-analyzer.d.ts.map +1 -1
- package/dist/src/analysis/failure-clusterer.d.ts +1 -1
- package/dist/src/analysis/failure-clusterer.d.ts.map +1 -1
- package/dist/src/analysis/gap-analyzer.d.ts +2 -2
- package/dist/src/analysis/gap-analyzer.d.ts.map +1 -1
- package/dist/src/analysis/gap-analyzer.js +4 -4
- package/dist/src/analysis/hedging-classifier.d.ts +1 -1
- package/dist/src/analysis/hedging-classifier.d.ts.map +1 -1
- package/dist/src/analysis/report-diagnostics.d.ts +1 -1
- package/dist/src/analysis/report-diagnostics.d.ts.map +1 -1
- package/dist/src/analysis/report-diagnostics.js +7 -7
- package/dist/src/analysis/sample-diagnostics.d.ts +1 -1
- package/dist/src/analysis/sample-diagnostics.d.ts.map +1 -1
- package/dist/src/analysis/sample-diagnostics.js +15 -15
- package/dist/src/authoring/evolver.d.ts +1 -1
- package/dist/src/authoring/evolver.d.ts.map +1 -1
- package/dist/src/authoring/evolver.js +5 -5
- package/dist/src/authoring/evolver.js.map +1 -1
- package/dist/src/authoring/generator.d.ts +1 -1
- package/dist/src/authoring/generator.d.ts.map +1 -1
- package/dist/src/authoring/generator.js +8 -8
- package/dist/src/authoring/generator.js.map +1 -1
- package/dist/src/cli/i18n-dict.d.ts +56 -0
- package/dist/src/cli/i18n-dict.d.ts.map +1 -0
- package/dist/src/cli/i18n-dict.js +934 -0
- package/dist/src/cli/i18n-dict.js.map +1 -0
- package/dist/src/cli/i18n.d.ts +24 -0
- package/dist/src/cli/i18n.d.ts.map +1 -0
- package/dist/src/cli/i18n.js +53 -0
- package/dist/src/cli/i18n.js.map +1 -0
- package/dist/src/cli.js +320 -413
- package/dist/src/cli.js.map +1 -1
- package/dist/src/eval-core/cache.d.ts +7 -3
- package/dist/src/eval-core/cache.d.ts.map +1 -1
- package/dist/src/eval-core/cache.js +14 -4
- package/dist/src/eval-core/cache.js.map +1 -1
- package/dist/src/eval-core/dependency-checker.d.ts +16 -1
- package/dist/src/eval-core/dependency-checker.d.ts.map +1 -1
- package/dist/src/eval-core/dependency-checker.js +79 -4
- package/dist/src/eval-core/dependency-checker.js.map +1 -1
- package/dist/src/eval-core/evaluation-execution.d.ts +3 -3
- package/dist/src/eval-core/evaluation-execution.d.ts.map +1 -1
- package/dist/src/eval-core/evaluation-execution.js +4 -2
- package/dist/src/eval-core/evaluation-execution.js.map +1 -1
- package/dist/src/eval-core/evaluation-job.d.ts +2 -2
- package/dist/src/eval-core/evaluation-job.d.ts.map +1 -1
- package/dist/src/eval-core/evaluation-reporting.d.ts +1 -1
- package/dist/src/eval-core/evaluation-reporting.d.ts.map +1 -1
- package/dist/src/eval-core/evaluation-reporting.js +4 -0
- package/dist/src/eval-core/evaluation-reporting.js.map +1 -1
- package/dist/src/eval-core/execution-strategy.d.ts +1 -1
- package/dist/src/eval-core/execution-strategy.d.ts.map +1 -1
- package/dist/src/eval-core/execution-strategy.js +35 -2
- package/dist/src/eval-core/execution-strategy.js.map +1 -1
- package/dist/src/eval-core/layer-gates.d.ts +17 -0
- package/dist/src/eval-core/layer-gates.d.ts.map +1 -0
- package/dist/src/eval-core/{ci-gates.js → layer-gates.js} +5 -5
- package/dist/src/eval-core/layer-gates.js.map +1 -0
- package/dist/src/eval-core/schema.d.ts +1 -1
- package/dist/src/eval-core/schema.d.ts.map +1 -1
- package/dist/src/eval-core/schema.js +17 -3
- package/dist/src/eval-core/schema.js.map +1 -1
- package/dist/src/eval-core/task-planner.d.ts +1 -1
- package/dist/src/eval-core/task-planner.d.ts.map +1 -1
- package/dist/src/eval-core/task-planner.js +1 -1
- package/dist/src/eval-core/task-planner.js.map +1 -1
- package/dist/src/eval-core/verdict.d.ts +5 -2
- package/dist/src/eval-core/verdict.d.ts.map +1 -1
- package/dist/src/eval-core/verdict.js +47 -8
- package/dist/src/eval-core/verdict.js.map +1 -1
- package/dist/src/eval-workflows/each-evaluation-workflow.d.ts +15 -8
- package/dist/src/eval-workflows/each-evaluation-workflow.d.ts.map +1 -1
- package/dist/src/eval-workflows/each-evaluation-workflow.js +3 -2
- package/dist/src/eval-workflows/each-evaluation-workflow.js.map +1 -1
- package/dist/src/eval-workflows/evaluation-pipeline.d.ts +33 -4
- package/dist/src/eval-workflows/evaluation-pipeline.d.ts.map +1 -1
- package/dist/src/eval-workflows/evaluation-pipeline.js +81 -2
- package/dist/src/eval-workflows/evaluation-pipeline.js.map +1 -1
- package/dist/src/eval-workflows/evaluation-preparation.d.ts +10 -6
- package/dist/src/eval-workflows/evaluation-preparation.d.ts.map +1 -1
- package/dist/src/eval-workflows/evaluation-preparation.js +2 -2
- package/dist/src/eval-workflows/evaluation-preparation.js.map +1 -1
- package/dist/src/eval-workflows/run-evaluation.d.ts +16 -5
- package/dist/src/eval-workflows/run-evaluation.d.ts.map +1 -1
- package/dist/src/eval-workflows/run-evaluation.js +34 -5
- package/dist/src/eval-workflows/run-evaluation.js.map +1 -1
- package/dist/src/executors/anthropic-api.d.ts +1 -1
- package/dist/src/executors/anthropic-api.d.ts.map +1 -1
- package/dist/src/executors/anthropic-api.js +1 -1
- package/dist/src/executors/anthropic-api.js.map +1 -1
- package/dist/src/executors/claude-cli.d.ts +2 -2
- package/dist/src/executors/claude-cli.d.ts.map +1 -1
- package/dist/src/executors/claude-cli.js +21 -1
- package/dist/src/executors/claude-cli.js.map +1 -1
- package/dist/src/executors/claude-sdk-trace.d.ts +1 -1
- package/dist/src/executors/claude-sdk-trace.d.ts.map +1 -1
- package/dist/src/executors/claude-sdk.d.ts +15 -2
- package/dist/src/executors/claude-sdk.d.ts.map +1 -1
- package/dist/src/executors/claude-sdk.js +19 -1
- package/dist/src/executors/claude-sdk.js.map +1 -1
- package/dist/src/executors/gemini.d.ts +1 -1
- package/dist/src/executors/gemini.d.ts.map +1 -1
- package/dist/src/executors/index.d.ts +1 -1
- package/dist/src/executors/index.d.ts.map +1 -1
- package/dist/src/executors/openai-api.d.ts +1 -1
- package/dist/src/executors/openai-api.d.ts.map +1 -1
- package/dist/src/executors/openai-api.js +1 -1
- package/dist/src/executors/openai-api.js.map +1 -1
- package/dist/src/executors/openai-cli.d.ts +1 -1
- package/dist/src/executors/openai-cli.d.ts.map +1 -1
- package/dist/src/executors/script.d.ts +1 -1
- package/dist/src/executors/script.d.ts.map +1 -1
- package/dist/src/executors/script.js +10 -1
- package/dist/src/executors/script.js.map +1 -1
- package/dist/src/executors/shared.d.ts +1 -1
- package/dist/src/executors/shared.d.ts.map +1 -1
- package/dist/src/grading/assertions.d.ts +1 -1
- package/dist/src/grading/assertions.d.ts.map +1 -1
- package/dist/src/grading/debias-validate.d.ts +1 -1
- package/dist/src/grading/debias-validate.d.ts.map +1 -1
- package/dist/src/grading/debias-validate.js +3 -3
- package/dist/src/grading/gold-cli.d.ts +1 -1
- package/dist/src/grading/gold-cli.d.ts.map +1 -1
- package/dist/src/grading/gold-cli.js +3 -3
- package/dist/src/grading/gold-cli.js.map +1 -1
- package/dist/src/grading/index.d.ts +1 -1
- package/dist/src/grading/index.d.ts.map +1 -1
- package/dist/src/grading/judge.d.ts +1 -1
- package/dist/src/grading/judge.d.ts.map +1 -1
- package/dist/src/grading/layered-scores.d.ts +1 -1
- package/dist/src/grading/layered-scores.d.ts.map +1 -1
- package/dist/src/inputs/eval-config.d.ts +1 -1
- package/dist/src/inputs/eval-config.d.ts.map +1 -1
- package/dist/src/inputs/eval-config.js +37 -18
- package/dist/src/inputs/eval-config.js.map +1 -1
- package/dist/src/inputs/load-samples.d.ts +1 -1
- package/dist/src/inputs/load-samples.d.ts.map +1 -1
- package/dist/src/inputs/load-samples.js +4 -4
- package/dist/src/inputs/load-samples.js.map +1 -1
- package/dist/src/inputs/mcp-resolver.d.ts +1 -1
- package/dist/src/inputs/mcp-resolver.d.ts.map +1 -1
- package/dist/src/inputs/mcp-resolver.js +4 -4
- package/dist/src/inputs/mcp-resolver.js.map +1 -1
- package/dist/src/inputs/skill-loader.d.ts +11 -2
- package/dist/src/inputs/skill-loader.d.ts.map +1 -1
- package/dist/src/inputs/skill-loader.js +30 -7
- package/dist/src/inputs/skill-loader.js.map +1 -1
- package/dist/src/inputs/url-fetcher.d.ts +1 -1
- package/dist/src/inputs/url-fetcher.d.ts.map +1 -1
- package/dist/src/inputs/url-fetcher.js +2 -2
- package/dist/src/inputs/url-fetcher.js.map +1 -1
- package/dist/src/observability/skill-health-analyzer.d.ts +1 -1
- package/dist/src/observability/skill-health-analyzer.d.ts.map +1 -1
- package/dist/src/observability/trace-adapter.d.ts +1 -1
- package/dist/src/observability/trace-adapter.d.ts.map +1 -1
- package/dist/src/renderer/html-renderer.d.ts +1 -1
- package/dist/src/renderer/html-renderer.d.ts.map +1 -1
- package/dist/src/renderer/html-renderer.js +3 -3
- package/dist/src/renderer/html-renderer.js.map +1 -1
- package/dist/src/renderer/layout.d.ts +1 -1
- package/dist/src/renderer/layout.d.ts.map +1 -1
- package/dist/src/renderer/layout.js +1 -1
- package/dist/src/renderer/layout.js.map +1 -1
- package/dist/src/renderer/skill-health-renderer.d.ts +1 -1
- package/dist/src/renderer/skill-health-renderer.d.ts.map +1 -1
- package/dist/src/renderer/summary.d.ts +1 -1
- package/dist/src/renderer/summary.d.ts.map +1 -1
- package/dist/src/renderer/summary.js +22 -20
- package/dist/src/renderer/summary.js.map +1 -1
- package/dist/src/renderer/table.d.ts +1 -1
- package/dist/src/renderer/table.d.ts.map +1 -1
- package/dist/src/renderer/trends.d.ts +1 -1
- package/dist/src/renderer/trends.d.ts.map +1 -1
- package/dist/src/renderer/trends.js +1 -1
- package/dist/src/renderer/trends.js.map +1 -1
- package/dist/src/server/job-store.d.ts +1 -1
- package/dist/src/server/job-store.d.ts.map +1 -1
- package/dist/src/server/report-server.d.ts +1 -1
- package/dist/src/server/report-server.d.ts.map +1 -1
- package/dist/src/server/report-server.js +20 -20
- package/dist/src/server/report-server.js.map +1 -1
- package/dist/src/server/report-store.d.ts +1 -1
- package/dist/src/server/report-store.d.ts.map +1 -1
- package/dist/src/types/eval.d.ts +9 -0
- package/dist/src/types/eval.d.ts.map +1 -1
- package/dist/src/types/executor.d.ts +1 -0
- package/dist/src/types/executor.d.ts.map +1 -1
- package/dist/src/types/report.d.ts +10 -0
- package/dist/src/types/report.d.ts.map +1 -1
- package/package.json +2 -2
- package/dist/src/eval-core/ci-gates.d.ts +0 -17
- package/dist/src/eval-core/ci-gates.d.ts.map +0 -1
- package/dist/src/eval-core/ci-gates.js.map +0 -1
- package/dist/src/types.d.ts +0 -2
- package/dist/src/types.d.ts.map +0 -1
- package/dist/src/types.js +0 -6
- package/dist/src/types.js.map +0 -1
package/dist/src/cli.js
CHANGED
|
@@ -4,6 +4,7 @@ import { resolve } from 'node:path';
|
|
|
4
4
|
import { homedir } from 'node:os';
|
|
5
5
|
import { join } from 'node:path';
|
|
6
6
|
import { existsSync } from 'node:fs';
|
|
7
|
+
import { tCli, getCliLang, parseLangFromArgv, langFromArgv } from './cli/i18n.js';
|
|
7
8
|
import { discoverVariants, parseVariantCwd } from './inputs/skill-loader.js';
|
|
8
9
|
import { loadEvalConfig, configVariantsToSpecs } from './inputs/eval-config.js';
|
|
9
10
|
// ---------------------------------------------------------------------------
|
|
@@ -14,7 +15,15 @@ const DEFAULT_REPORTS_DIR = join(homedir(), '.oh-my-knowledge', 'reports');
|
|
|
14
15
|
// Defaults are applied inside parseRunConfig (after config-file merge) so that
|
|
15
16
|
// CLI `undefined` can be reliably distinguished from "user passed the default value".
|
|
16
17
|
// Priority order resolved in parseRunConfig: CLI arg > --config file > hard-coded default.
|
|
18
|
+
/**
|
|
19
|
+
* 所有子命令都接受的通用 flag。新增 --lang 让 parseArgs strict:false 模式下
|
|
20
|
+
* 仍能把值类型化到 values.lang 上(否则未声明的 flag 会被丢弃)。
|
|
21
|
+
*/
|
|
22
|
+
const COMMON_OPTIONS = {
|
|
23
|
+
lang: { type: 'string' },
|
|
24
|
+
};
|
|
17
25
|
const RUN_OPTIONS = {
|
|
26
|
+
...COMMON_OPTIONS,
|
|
18
27
|
samples: { type: 'string' },
|
|
19
28
|
'skill-dir': { type: 'string' },
|
|
20
29
|
control: { type: 'string' },
|
|
@@ -38,6 +47,10 @@ const RUN_OPTIONS = {
|
|
|
38
47
|
retry: { type: 'string' },
|
|
39
48
|
resume: { type: 'string' },
|
|
40
49
|
'layered-stats': { type: 'boolean' },
|
|
50
|
+
// v0.22 — strict-baseline default true. Declare both forms; reconcile in
|
|
51
|
+
// parseRunConfig (后者赢)。strict-baseline 没传 + no-strict-baseline 没传 = default true。
|
|
52
|
+
'strict-baseline': { type: 'boolean' },
|
|
53
|
+
'no-strict-baseline': { type: 'boolean' },
|
|
41
54
|
};
|
|
42
55
|
// ---------------------------------------------------------------------------
|
|
43
56
|
// parseRunConfig
|
|
@@ -144,6 +157,21 @@ function parseRunConfig(argv, extraOptions = {}) {
|
|
|
144
157
|
const resume = values.resume;
|
|
145
158
|
const blind = values.blind ?? evalConfig?.blind ?? false;
|
|
146
159
|
const layeredStats = values['layered-stats'] ?? false;
|
|
160
|
+
// v0.22 — strict-baseline default true. Reconcile both flag forms.
|
|
161
|
+
// Priority: --no-strict-baseline > --strict-baseline > undefined(=true).
|
|
162
|
+
const noStrictFlag = values['no-strict-baseline'];
|
|
163
|
+
const strictFlag = values['strict-baseline'];
|
|
164
|
+
const strictBaseline = noStrictFlag === true ? false : (strictFlag ?? true);
|
|
165
|
+
// v0.22 — extract eval.yaml variant.allowedSkills overrides (per-variant). Always
|
|
166
|
+
// wins over strictBaseline default. Empty object when no eval.yaml or no overrides.
|
|
167
|
+
const variantAllowedSkills = {};
|
|
168
|
+
if (evalConfig?.variants) {
|
|
169
|
+
for (const v of evalConfig.variants) {
|
|
170
|
+
if (v.allowedSkills !== undefined) {
|
|
171
|
+
variantAllowedSkills[v.name] = v.allowedSkills;
|
|
172
|
+
}
|
|
173
|
+
}
|
|
174
|
+
}
|
|
147
175
|
return {
|
|
148
176
|
values,
|
|
149
177
|
config: {
|
|
@@ -168,178 +196,56 @@ function parseRunConfig(argv, extraOptions = {}) {
|
|
|
168
196
|
blind,
|
|
169
197
|
layeredStats,
|
|
170
198
|
budget: evalConfig?.budget,
|
|
199
|
+
strictBaseline,
|
|
200
|
+
...(Object.keys(variantAllowedSkills).length > 0 && { variantAllowedSkills }),
|
|
171
201
|
},
|
|
172
202
|
};
|
|
173
203
|
}
|
|
174
204
|
// ---------------------------------------------------------------------------
|
|
175
|
-
// Help text
|
|
176
|
-
// ---------------------------------------------------------------------------
|
|
177
|
-
const HELP = `
|
|
178
|
-
oh-my-knowledge — Knowledge artifact evaluation toolkit
|
|
179
|
-
|
|
180
|
-
Usage:
|
|
181
|
-
omk bench run [options] Run an evaluation
|
|
182
|
-
omk bench report [options] Start the report server
|
|
183
|
-
omk bench ci [options] Run evaluation and exit with pass/fail code
|
|
184
|
-
omk bench init [dir] Scaffold a new eval project
|
|
185
|
-
omk bench gen-samples [skill] Generate eval-samples from skill content
|
|
186
|
-
omk bench diff <id1> <id2> Compare two evaluation reports
|
|
187
|
-
omk bench evolve <skill> Self-improve a skill through iterative evaluation
|
|
188
|
-
|
|
189
|
-
omk analyze <dir> Analyze cc session trace(s), produce skill 健康度日报 (v0.18)
|
|
190
|
-
|
|
191
|
-
Options for "bench run":
|
|
192
|
-
|
|
193
|
-
--samples <path> Sample file (default: eval-samples.json)
|
|
194
|
-
--skill-dir <path> Skill definitions directory (default: skills)
|
|
195
|
-
--control <expr> Control-group variant expression (experiment role = control)
|
|
196
|
-
--treatment <v1,v2> Treatment-group variant expressions (comma-separated; role = treatment)
|
|
197
|
-
Each variant expression resolves to an artifact and optional runtime context:
|
|
198
|
-
"baseline" — bare model, no artifact injected
|
|
199
|
-
"git:name" — artifact from last commit
|
|
200
|
-
"git:ref:name" — artifact from specific commit
|
|
201
|
-
path with "/" — artifact from file directly (e.g. ./v1.md)
|
|
202
|
-
"name@/cwd" — attach runtime context / cwd
|
|
203
|
-
At least one of --control / --treatment must be provided.
|
|
204
|
-
--config <path> YAML/JSON config file (evaluation-as-code).
|
|
205
|
-
Declares samples + variants + model + executor in one file.
|
|
206
|
-
CLI flags override config fields when both are provided.
|
|
207
|
-
Relative paths inside the config are resolved against its directory.
|
|
208
|
-
--model <name> Model under test (default: sonnet)
|
|
209
|
-
--judge-model <name> Judge model (default: haiku)
|
|
210
|
-
--output-dir <path> Report output directory (default: ~/.oh-my-knowledge/reports/)
|
|
211
|
-
--no-judge Skip LLM judging
|
|
212
|
-
--no-cache Disable result caching
|
|
213
|
-
--dry-run Preview tasks without executing
|
|
214
|
-
--blind Blind A/B mode: hide variant names in report
|
|
215
|
-
--concurrency <n> Number of parallel tasks (default: 1)
|
|
216
|
-
--timeout <seconds> Executor timeout per task in seconds (default: 120)
|
|
217
|
-
--repeat <n> Run evaluation N times for variance analysis (default: 1)
|
|
218
|
-
--judge-repeat <n> Call LLM judge N times per (sample × dimension) for self-
|
|
219
|
-
consistency (default: 1). High stddev across runs = the
|
|
220
|
-
judge is unstable on this rubric and the score is noisy.
|
|
221
|
-
--judge-models <list> Multi-judge ensemble. Comma-separated executor:model pairs,
|
|
222
|
-
e.g. claude:opus,openai:gpt-4o,gemini:pro. Each judge scores
|
|
223
|
-
every (sample × dimension); report includes per-judge break-
|
|
224
|
-
down + Pearson/MAD inter-judge agreement. Refutes "Claude
|
|
225
|
-
judge Claude same-modality bias" critique. Combines with
|
|
226
|
-
--judge-repeat. Cost ~ N_judges × N_repeat × N_samples.
|
|
227
|
-
--bootstrap Compute bootstrap confidence intervals (distribution-free,
|
|
228
|
-
preferred over t-interval for ordinal LLM scores). Adds
|
|
229
|
-
per-variant CI on the mean + pairwise CI on treatment-vs-
|
|
230
|
-
control difference (significant=0 outside CI). Reports both
|
|
231
|
-
t-interval and bootstrap so old tooling still works.
|
|
232
|
-
--bootstrap-samples <n> Number of bootstrap resamples (default 1000). N>10000
|
|
233
|
-
triggers a stderr warning about runtime cost.
|
|
234
|
-
--retry <n> Retry failed tasks up to N times with exponential backoff (default: 0)
|
|
235
|
-
--resume <report-id> Resume from a previous report, skipping completed tasks
|
|
236
|
-
--executor <name> Executor: claude, openai, gemini, anthropic-api, openai-api,
|
|
237
|
-
or any shell command (e.g. "python my_provider.py")
|
|
238
|
-
--judge-executor <name> Executor for LLM judge (default: same as --executor)
|
|
239
|
-
--each Evaluate each skill independently against baseline
|
|
240
|
-
Requires {name}.eval-samples.json paired with each skill
|
|
241
|
-
--skip-preflight Skip model connectivity check before evaluation
|
|
242
|
-
--mcp-config <path> MCP config file for URL fetching via MCP servers
|
|
243
|
-
(default: .mcp.json in current directory)
|
|
244
|
-
--no-serve Skip auto-starting report server after evaluation
|
|
245
|
-
--verbose Print detailed progress for each sample (exec result, grading phases)
|
|
246
|
-
--layered-stats Expand the three-layer (fact/behavior/judge) independent
|
|
247
|
-
significance breakdown in the HTML report by default.
|
|
248
|
-
Without this flag, the breakdown is collapsed behind a
|
|
249
|
-
click-to-expand summary under each comparison.
|
|
250
|
-
|
|
251
|
-
Options for "bench ci":
|
|
252
|
-
(same as "bench run", plus:)
|
|
253
|
-
--threshold <number> Minimum score to pass, applied INDEPENDENTLY to each of
|
|
254
|
-
the three layers (fact / behavior / LLM judge). ANY
|
|
255
|
-
layer below threshold fails the gate — this prevents
|
|
256
|
-
composite averaging from masking a single-layer collapse.
|
|
257
|
-
Default: 3.5. If all three layers are absent (no
|
|
258
|
-
assertions and no rubric defined in eval-samples), the
|
|
259
|
-
gate FAILS with a configuration hint — no composite fallback.
|
|
260
|
-
|
|
261
|
-
Options for "bench report":
|
|
262
|
-
--port <number> Server port (default: 7799)
|
|
263
|
-
--reports-dir <path> Reports directory (default: ~/.oh-my-knowledge/reports/)
|
|
264
|
-
--export <id> Export report as standalone HTML file
|
|
265
|
-
--dev Dev mode: auto-restart on lib/ file changes
|
|
266
|
-
|
|
267
|
-
Options for "bench gen-samples":
|
|
268
|
-
--each Generate for all skills missing eval-samples
|
|
269
|
-
--count <n> Number of samples to generate per skill (default: 5)
|
|
270
|
-
--model <name> Model for generation (default: sonnet)
|
|
271
|
-
--skill-dir <path> Skill directory (default: skills), used with --each
|
|
272
|
-
|
|
273
|
-
Options for "analyze":
|
|
274
|
-
<dir> Input: cc session JSONL file / dir (e.g. ~/.claude/projects/<slug>)
|
|
275
|
-
--kb <path> Knowledge base root (default: auto-infer from trace cwd)
|
|
276
|
-
--last <duration> Time window like "7d" / "30d" (default: all)
|
|
277
|
-
--from <iso> Window start (ISO8601), takes precedence over --last
|
|
278
|
-
--to <iso> Window end (ISO8601), takes precedence over --last
|
|
279
|
-
--skills <n1,n2,...> Whitelist skills to analyze (default: all)
|
|
280
|
-
--output-dir <path> Output dir (default: ~/.oh-my-knowledge/analyses/)
|
|
281
|
-
|
|
282
|
-
Options for "bench evolve":
|
|
283
|
-
--rounds <n> Maximum evolution rounds (default: 5)
|
|
284
|
-
--target <score> Stop early when score reaches this threshold
|
|
285
|
-
--samples <path> Sample file (default: eval-samples.json)
|
|
286
|
-
--model <name> Model under test (default: sonnet)
|
|
287
|
-
--judge-model <name> Judge model (default: haiku)
|
|
288
|
-
--improve-model <name> Model for generating improvements (default: sonnet)
|
|
289
|
-
--concurrency <n> Parallel eval tasks (default: 1)
|
|
290
|
-
--timeout <seconds> Executor timeout per task in seconds (default: 120)
|
|
291
|
-
--executor <name> Executor to use (default: claude)
|
|
292
|
-
|
|
293
|
-
Examples:
|
|
294
|
-
omk bench run --control v1 --treatment v2
|
|
295
|
-
omk bench run --control baseline --treatment my-skill
|
|
296
|
-
omk bench run --control git:my-skill --treatment my-skill
|
|
297
|
-
omk bench run --control ./old-skill.md --treatment ./new-skill.md
|
|
298
|
-
omk bench run --control baseline --treatment v1,v2,v3
|
|
299
|
-
omk bench run --config eval.yaml
|
|
300
|
-
omk bench run --config eval.yaml --model sonnet-4.6 # CLI overrides config
|
|
301
|
-
omk bench run --each
|
|
302
|
-
omk bench run --dry-run
|
|
303
|
-
omk bench report --port 8080
|
|
304
|
-
omk bench report --export v1-vs-v2-20260326-1832
|
|
305
|
-
omk bench init my-eval
|
|
306
|
-
omk bench gen-samples skills/my-skill.md
|
|
307
|
-
omk bench gen-samples --each
|
|
308
|
-
omk bench diff <report-id-1> <report-id-2>
|
|
309
|
-
omk bench evolve skills/my-skill.md --rounds 5
|
|
310
|
-
omk analyze ~/.claude/projects/-Users-lizhiyao-Documents-oh-my-knowledge
|
|
311
|
-
omk analyze ~/.claude/projects/my-project --last 7d --kb /path/to/project
|
|
312
|
-
omk analyze ~/.claude/projects/my-project --skills audit,polish
|
|
313
|
-
`.trim();
|
|
314
|
-
// ---------------------------------------------------------------------------
|
|
315
205
|
// Update check
|
|
316
206
|
// ---------------------------------------------------------------------------
|
|
317
|
-
async function checkUpdate() {
|
|
207
|
+
async function checkUpdate(lang) {
|
|
318
208
|
try {
|
|
319
209
|
const { readFileSync } = await import('node:fs');
|
|
320
210
|
const { fileURLToPath } = await import('node:url');
|
|
321
211
|
const { dirname, join } = await import('node:path');
|
|
322
212
|
const __dirname = dirname(fileURLToPath(import.meta.url));
|
|
323
|
-
const
|
|
213
|
+
const findPackageJson = (startDir) => {
|
|
214
|
+
let dir = startDir;
|
|
215
|
+
for (let i = 0; i < 5; i++) {
|
|
216
|
+
const candidate = join(dir, 'package.json');
|
|
217
|
+
if (existsSync(candidate))
|
|
218
|
+
return candidate;
|
|
219
|
+
dir = dirname(dir);
|
|
220
|
+
}
|
|
221
|
+
return null;
|
|
222
|
+
};
|
|
223
|
+
const pkgPath = findPackageJson(__dirname);
|
|
224
|
+
if (!pkgPath)
|
|
225
|
+
return;
|
|
226
|
+
const pkg = JSON.parse(readFileSync(pkgPath, 'utf-8'));
|
|
324
227
|
const registry = pkg.publishConfig?.registry || 'https://registry.npmjs.org';
|
|
325
228
|
const res = await fetch(`${registry}/${pkg.name}/latest`, { signal: AbortSignal.timeout(3000) });
|
|
326
229
|
if (!res.ok)
|
|
327
230
|
return;
|
|
328
231
|
const data = await res.json();
|
|
329
232
|
if (data.version && data.version !== pkg.version) {
|
|
330
|
-
process.stderr.write(
|
|
233
|
+
process.stderr.write(tCli('cli.update.new_version_available', lang, {
|
|
234
|
+
old: pkg.version, new: data.version, pkg: pkg.name,
|
|
235
|
+
}));
|
|
331
236
|
}
|
|
332
237
|
}
|
|
333
|
-
catch { /*
|
|
238
|
+
catch { /* 静默失败,不影响正常使用 */ }
|
|
334
239
|
}
|
|
335
240
|
// ---------------------------------------------------------------------------
|
|
336
241
|
// Main
|
|
337
242
|
// ---------------------------------------------------------------------------
|
|
338
243
|
async function main() {
|
|
339
|
-
|
|
244
|
+
const lang = getCliLang(parseLangFromArgv(process.argv));
|
|
245
|
+
checkUpdate(lang);
|
|
340
246
|
const [domain, command, ...rest] = process.argv.slice(2);
|
|
341
247
|
if (!domain || domain === '--help' || domain === '-h') {
|
|
342
|
-
console.log(
|
|
248
|
+
console.log(tCli('cli.help.main', lang).trim());
|
|
343
249
|
process.exit(0);
|
|
344
250
|
}
|
|
345
251
|
if (domain === 'analyze') {
|
|
@@ -348,11 +254,11 @@ async function main() {
|
|
|
348
254
|
return;
|
|
349
255
|
}
|
|
350
256
|
if (domain !== 'bench') {
|
|
351
|
-
console.error(
|
|
257
|
+
console.error(tCli('cli.common.unknown_domain', lang, { domain }));
|
|
352
258
|
process.exit(1);
|
|
353
259
|
}
|
|
354
260
|
if (!command || command === '--help' || command === '-h') {
|
|
355
|
-
console.log(
|
|
261
|
+
console.log(tCli('cli.help.main', lang).trim());
|
|
356
262
|
process.exit(0);
|
|
357
263
|
}
|
|
358
264
|
switch (command) {
|
|
@@ -365,8 +271,8 @@ async function main() {
|
|
|
365
271
|
case 'init':
|
|
366
272
|
await handleInit(rest);
|
|
367
273
|
break;
|
|
368
|
-
case '
|
|
369
|
-
await
|
|
274
|
+
case 'gate':
|
|
275
|
+
await handleGate(rest);
|
|
370
276
|
break;
|
|
371
277
|
case 'gen-samples':
|
|
372
278
|
await handleGenSamples(rest);
|
|
@@ -396,60 +302,77 @@ async function main() {
|
|
|
396
302
|
await handleFailures(rest);
|
|
397
303
|
break;
|
|
398
304
|
default:
|
|
399
|
-
console.error(
|
|
305
|
+
console.error(tCli('cli.common.unknown_bench_command', lang, { command }));
|
|
400
306
|
process.exit(1);
|
|
401
307
|
}
|
|
402
308
|
}
|
|
403
309
|
// ---------------------------------------------------------------------------
|
|
404
310
|
// Progress callback
|
|
405
311
|
// ---------------------------------------------------------------------------
|
|
406
|
-
|
|
407
|
-
|
|
408
|
-
|
|
409
|
-
|
|
410
|
-
|
|
411
|
-
|
|
412
|
-
|
|
413
|
-
|
|
414
|
-
|
|
415
|
-
|
|
416
|
-
process.stderr.write(`[${completed}/${total}] ${sample_id}/${variant} ❌ ${error}\n`);
|
|
417
|
-
return;
|
|
418
|
-
}
|
|
419
|
-
if (phase === 'start') {
|
|
420
|
-
process.stderr.write(`[${completed}/${total}] ${sample_id}/${variant} ⏳ 执行中...\n`);
|
|
421
|
-
}
|
|
422
|
-
else if (phase === 'exec_done') {
|
|
423
|
-
const costInfo = costUSD != null && costUSD > 0 ? ` $${costUSD.toFixed(4)}` : '';
|
|
424
|
-
process.stderr.write(`[${completed}/${total}] ${sample_id}/${variant} 执行完成 ${durationMs}ms ${inputTokens}+${outputTokens} tokens${costInfo}\n`);
|
|
425
|
-
if (outputPreview) {
|
|
426
|
-
process.stderr.write(` 输出预览: ${outputPreview.slice(0, 150).replace(/\n/g, ' ')}\n`);
|
|
312
|
+
/**
|
|
313
|
+
* Factory: 闭住 lang, 返回 onProgress callback。evaluation engine 回调时不传
|
|
314
|
+
* 上下文, 所以 lang 必须在 handler 入口处通过 closure 传进来。
|
|
315
|
+
*/
|
|
316
|
+
function makeOnProgress(lang) {
|
|
317
|
+
return ({ phase, completed, total, sample_id, variant, durationMs, inputTokens, outputTokens, costUSD, score, outputPreview, judgePhase: _judgePhase, judgeDim, skipped, attempt, maxAttempts, error, }) => {
|
|
318
|
+
const ctx = { i: completed ?? '', n: total ?? '', sample: sample_id ?? '', variant: variant ?? '' };
|
|
319
|
+
if (phase === 'preflight') {
|
|
320
|
+
process.stderr.write(tCli('cli.progress.preflight_starting', lang));
|
|
321
|
+
return;
|
|
427
322
|
}
|
|
428
|
-
|
|
429
|
-
|
|
430
|
-
|
|
431
|
-
|
|
432
|
-
|
|
433
|
-
|
|
434
|
-
|
|
435
|
-
|
|
436
|
-
|
|
437
|
-
|
|
438
|
-
if (
|
|
439
|
-
process.stderr.write(
|
|
440
|
-
|
|
441
|
-
|
|
442
|
-
|
|
443
|
-
|
|
444
|
-
|
|
445
|
-
|
|
323
|
+
if (phase === 'retry') {
|
|
324
|
+
process.stderr.write(tCli('cli.progress.sample_retry', lang, {
|
|
325
|
+
...ctx, attempt: attempt ?? '', max: maxAttempts ?? '',
|
|
326
|
+
}));
|
|
327
|
+
return;
|
|
328
|
+
}
|
|
329
|
+
if (phase === 'error') {
|
|
330
|
+
process.stderr.write(tCli('cli.progress.sample_error', lang, { ...ctx, error: error ?? '' }));
|
|
331
|
+
return;
|
|
332
|
+
}
|
|
333
|
+
if (phase === 'start') {
|
|
334
|
+
process.stderr.write(tCli('cli.progress.sample_executing', lang, ctx));
|
|
335
|
+
}
|
|
336
|
+
else if (phase === 'exec_done') {
|
|
337
|
+
const cost = costUSD != null && costUSD > 0 ? ` $${costUSD.toFixed(4)}` : '';
|
|
338
|
+
process.stderr.write(tCli('cli.progress.sample_exec_done', lang, {
|
|
339
|
+
...ctx, ms: durationMs ?? '', input: inputTokens ?? '', output: outputTokens ?? '', cost,
|
|
340
|
+
}));
|
|
341
|
+
if (outputPreview) {
|
|
342
|
+
process.stderr.write(tCli('cli.progress.output_preview', lang, {
|
|
343
|
+
preview: outputPreview.slice(0, 150).replace(/\n/g, ' '),
|
|
344
|
+
}));
|
|
345
|
+
}
|
|
346
|
+
}
|
|
347
|
+
else if (phase === 'grading') {
|
|
348
|
+
const dim = judgeDim ? ` [${judgeDim}]` : '';
|
|
349
|
+
process.stderr.write(tCli('cli.progress.judging', lang, { ...ctx, dim }));
|
|
350
|
+
}
|
|
351
|
+
else if (phase === 'judge_done') {
|
|
352
|
+
const dim = judgeDim ? ` [${judgeDim}]` : '';
|
|
353
|
+
process.stderr.write(tCli('cli.progress.judged', lang, { ...ctx, dim, score: score ?? '' }));
|
|
354
|
+
}
|
|
355
|
+
else if (phase === 'done' && skipped) {
|
|
356
|
+
if (sample_id)
|
|
357
|
+
process.stderr.write(tCli('cli.progress.skipped', lang, ctx));
|
|
358
|
+
}
|
|
359
|
+
else {
|
|
360
|
+
const cost = costUSD != null && costUSD > 0 ? ` $${costUSD.toFixed(4)}` : '';
|
|
361
|
+
const scoreInfo = typeof score === 'number' ? ` score=${score}` : '';
|
|
362
|
+
process.stderr.write(tCli('cli.progress.sample_done', lang, {
|
|
363
|
+
...ctx, ms: durationMs ?? '', input: inputTokens ?? '', output: outputTokens ?? '',
|
|
364
|
+
cost, score: scoreInfo,
|
|
365
|
+
}));
|
|
366
|
+
}
|
|
367
|
+
};
|
|
446
368
|
}
|
|
447
369
|
// ---------------------------------------------------------------------------
|
|
448
370
|
// handleRun
|
|
449
371
|
// ---------------------------------------------------------------------------
|
|
450
372
|
async function handleRun(argv) {
|
|
373
|
+
const lang = langFromArgv(argv);
|
|
451
374
|
const { values, config } = parseRunConfig(argv, {
|
|
452
|
-
blind: { type: 'boolean'
|
|
375
|
+
blind: { type: 'boolean' },
|
|
453
376
|
repeat: { type: 'string', default: '1' },
|
|
454
377
|
'judge-repeat': { type: 'string', default: '1' },
|
|
455
378
|
'judge-models': { type: 'string' },
|
|
@@ -462,27 +385,29 @@ async function handleRun(argv) {
|
|
|
462
385
|
'budget-per-sample-ms': { type: 'string' },
|
|
463
386
|
});
|
|
464
387
|
const { runEvaluation, runMultiple, runEachEvaluation } = await import('./eval-workflows/run-evaluation.js');
|
|
465
|
-
|
|
466
|
-
|
|
467
|
-
|
|
468
|
-
|
|
388
|
+
if (values.blind !== undefined) {
|
|
389
|
+
config.blind = values.blind;
|
|
390
|
+
}
|
|
391
|
+
config.onProgress = makeOnProgress(lang);
|
|
392
|
+
// --repeat 输入校验: 非 ≥1 整数时提示并钳到 1, 不静默掩盖用户错字 / 极端输入。
|
|
393
|
+
// 提前到 --each 分支之前, 保证 each 模式也能读到 repeat (曾经 bug: --each 吞 --repeat)。
|
|
469
394
|
const repeatRaw = values.repeat;
|
|
470
395
|
const parsedRepeat = repeatRaw !== undefined ? Number(repeatRaw) : 1;
|
|
471
396
|
if (repeatRaw !== undefined && (!Number.isFinite(parsedRepeat) || parsedRepeat < 1)) {
|
|
472
|
-
process.stderr.write(
|
|
397
|
+
process.stderr.write(tCli('cli.run.invalid_repeat', lang, { value: repeatRaw }));
|
|
473
398
|
}
|
|
474
399
|
const repeatCount = Math.max(1, Math.floor(parsedRepeat) || 1);
|
|
475
|
-
// --judge-repeat
|
|
400
|
+
// --judge-repeat 同样校验: 非 ≥1 整数时钳到 1
|
|
476
401
|
const judgeRepeatRaw = values['judge-repeat'];
|
|
477
402
|
const parsedJudgeRepeat = judgeRepeatRaw !== undefined ? Number(judgeRepeatRaw) : 1;
|
|
478
403
|
if (judgeRepeatRaw !== undefined && (!Number.isFinite(parsedJudgeRepeat) || parsedJudgeRepeat < 1)) {
|
|
479
|
-
process.stderr.write(
|
|
404
|
+
process.stderr.write(tCli('cli.run.invalid_judge_repeat', lang, { value: judgeRepeatRaw }));
|
|
480
405
|
}
|
|
481
406
|
const judgeRepeatCount = Math.max(1, Math.floor(parsedJudgeRepeat) || 1);
|
|
482
407
|
if (judgeRepeatCount > 1)
|
|
483
408
|
config.judgeRepeat = judgeRepeatCount;
|
|
484
409
|
// --judge-models executor:model,executor:model,... -> JudgeConfig[]
|
|
485
|
-
// 至少 2 个才进 ensemble 模式,1 个等同于 --judge-model
|
|
410
|
+
// 至少 2 个才进 ensemble 模式, 1 个等同于 --judge-model
|
|
486
411
|
const judgeModelsRaw = values['judge-models'];
|
|
487
412
|
if (judgeModelsRaw) {
|
|
488
413
|
const parts = judgeModelsRaw.split(',').map((s) => s.trim()).filter(Boolean);
|
|
@@ -490,7 +415,7 @@ async function handleRun(argv) {
|
|
|
490
415
|
const [executor, ...modelParts] = p.split(':');
|
|
491
416
|
const model = modelParts.join(':');
|
|
492
417
|
if (!executor || !model) {
|
|
493
|
-
throw new Error(
|
|
418
|
+
throw new Error(tCli('cli.run.invalid_judge_models_format', lang, { part: p }));
|
|
494
419
|
}
|
|
495
420
|
return { executor, model };
|
|
496
421
|
});
|
|
@@ -498,8 +423,10 @@ async function handleRun(argv) {
|
|
|
498
423
|
config.judgeModels = judges;
|
|
499
424
|
}
|
|
500
425
|
else if (judges.length === 1) {
|
|
501
|
-
// 单 judge 不走 ensemble
|
|
502
|
-
process.stderr.write(
|
|
426
|
+
// 单 judge 不走 ensemble, 但允许这样写, 等同于 --judge-model + --executor
|
|
427
|
+
process.stderr.write(tCli('cli.run.judge_models_single_warning', lang, {
|
|
428
|
+
executor: judges[0].executor, model: judges[0].model,
|
|
429
|
+
}));
|
|
503
430
|
}
|
|
504
431
|
}
|
|
505
432
|
// --budget-usd / --budget-per-sample-usd / --budget-per-sample-ms:
|
|
@@ -521,7 +448,7 @@ async function handleRun(argv) {
|
|
|
521
448
|
// off so historical reports (judgePromptHash from v2-cot era) can be reproduced.
|
|
522
449
|
if (values['no-debias-length']) {
|
|
523
450
|
config.lengthDebias = false;
|
|
524
|
-
process.stderr.write('
|
|
451
|
+
process.stderr.write(tCli('cli.run.no_debias_length_active', lang));
|
|
525
452
|
}
|
|
526
453
|
// --bootstrap / --bootstrap-samples
|
|
527
454
|
if (values.bootstrap) {
|
|
@@ -529,11 +456,11 @@ async function handleRun(argv) {
|
|
|
529
456
|
const bsRaw = values['bootstrap-samples'];
|
|
530
457
|
const parsedBs = bsRaw !== undefined ? Number(bsRaw) : 1000;
|
|
531
458
|
if (bsRaw !== undefined && (!Number.isFinite(parsedBs) || parsedBs < 100)) {
|
|
532
|
-
process.stderr.write(
|
|
459
|
+
process.stderr.write(tCli('cli.run.invalid_bootstrap_samples', lang, { value: bsRaw }));
|
|
533
460
|
}
|
|
534
461
|
const bsCount = Math.max(100, Math.floor(parsedBs) || 1000);
|
|
535
462
|
if (bsCount > 10000) {
|
|
536
|
-
process.stderr.write(
|
|
463
|
+
process.stderr.write(tCli('cli.run.bootstrap_samples_too_large', lang, { n: bsCount }));
|
|
537
464
|
}
|
|
538
465
|
config.bootstrapSamples = bsCount;
|
|
539
466
|
}
|
|
@@ -545,30 +472,32 @@ async function handleRun(argv) {
|
|
|
545
472
|
repeat: repeatCount,
|
|
546
473
|
onSkillProgress({ phase, skill, current, total }) {
|
|
547
474
|
if (phase === 'start') {
|
|
548
|
-
process.stderr.write(
|
|
475
|
+
process.stderr.write(tCli('cli.run.skill_section', lang, {
|
|
476
|
+
i: current ?? '', n: total ?? '', skill: skill ?? '',
|
|
477
|
+
}));
|
|
549
478
|
}
|
|
550
479
|
},
|
|
551
480
|
});
|
|
552
481
|
console.log(JSON.stringify(report, null, 2));
|
|
553
482
|
if (filePath) {
|
|
554
|
-
process.stderr.write('
|
|
555
|
-
process.stderr.write(
|
|
483
|
+
process.stderr.write(tCli('cli.run.batch_complete', lang));
|
|
484
|
+
process.stderr.write(tCli('cli.run.report_saved', lang, { path: filePath }));
|
|
556
485
|
if (!values['no-serve'] && process.stdout.isTTY) {
|
|
557
486
|
const { createReportServer } = await import('./server/report-server.js');
|
|
558
487
|
const server = createReportServer({ reportsDir: config.outputDir });
|
|
559
488
|
const serverUrl = await server.start();
|
|
560
|
-
const reportUrl = `${serverUrl}/
|
|
561
|
-
process.stderr.write(
|
|
562
|
-
process.stderr.write(
|
|
563
|
-
process.stderr.write('
|
|
489
|
+
const reportUrl = `${serverUrl}/reports/${report.id}`;
|
|
490
|
+
process.stderr.write(tCli('cli.run.report_server_running', lang, { url: serverUrl }));
|
|
491
|
+
process.stderr.write(tCli('cli.run.report_server_view', lang, { url: reportUrl }));
|
|
492
|
+
process.stderr.write(tCli('cli.run.report_server_stop', lang));
|
|
564
493
|
const { platform } = await import('node:os');
|
|
565
494
|
const openCmd = platform() === 'darwin' ? 'open' : platform() === 'win32' ? 'start' : 'xdg-open';
|
|
566
495
|
const { execFile: execFileCb } = await import('node:child_process');
|
|
567
496
|
execFileCb(openCmd, [reportUrl], () => { });
|
|
568
497
|
}
|
|
569
498
|
else if (!values['no-serve']) {
|
|
570
|
-
process.stderr.write('
|
|
571
|
-
process.stderr.write(
|
|
499
|
+
process.stderr.write(tCli('cli.run.no_serve_in_non_tty', lang));
|
|
500
|
+
process.stderr.write(tCli('cli.run.no_serve_view_hint', lang, { dir: config.outputDir }));
|
|
572
501
|
}
|
|
573
502
|
}
|
|
574
503
|
return;
|
|
@@ -580,7 +509,7 @@ async function handleRun(argv) {
|
|
|
580
509
|
...config,
|
|
581
510
|
repeat: repeatCount,
|
|
582
511
|
onRepeatProgress({ run, total }) {
|
|
583
|
-
process.stderr.write(
|
|
512
|
+
process.stderr.write(tCli('cli.run.run_section', lang, { i: run, n: total }));
|
|
584
513
|
},
|
|
585
514
|
});
|
|
586
515
|
report = result.report;
|
|
@@ -604,19 +533,22 @@ async function handleRun(argv) {
|
|
|
604
533
|
if (out.result && out.gold) {
|
|
605
534
|
process.stderr.write(formatGoldCompare(out.result, out.gold));
|
|
606
535
|
if (out.result.contaminationWarning) {
|
|
607
|
-
process.stderr.write(
|
|
536
|
+
process.stderr.write(tCli('cli.run.contamination_warning', lang, {
|
|
537
|
+
warning: out.result.contaminationWarning,
|
|
538
|
+
}));
|
|
608
539
|
}
|
|
609
540
|
}
|
|
610
541
|
else {
|
|
611
|
-
process.stderr.write(
|
|
612
|
-
for (const m of out.loadIssues)
|
|
613
|
-
process.stderr.write(
|
|
542
|
+
process.stderr.write(tCli('cli.run.gold_load_failed', lang, { dir: goldDir }));
|
|
543
|
+
for (const m of out.loadIssues) {
|
|
544
|
+
process.stderr.write(tCli('cli.run.gold_load_issue', lang, { message: m }));
|
|
545
|
+
}
|
|
614
546
|
}
|
|
615
547
|
}
|
|
616
548
|
console.log(JSON.stringify(report, null, 2));
|
|
617
549
|
if (filePath) {
|
|
618
|
-
process.stderr.write('
|
|
619
|
-
process.stderr.write(
|
|
550
|
+
process.stderr.write(tCli('cli.run.eval_complete', lang));
|
|
551
|
+
process.stderr.write(tCli('cli.run.report_saved', lang, { path: filePath }));
|
|
620
552
|
if (!values['no-serve'] && process.stdout.isTTY) {
|
|
621
553
|
// Auto-start report server
|
|
622
554
|
const { createReportServer } = await import('./server/report-server.js');
|
|
@@ -624,10 +556,10 @@ async function handleRun(argv) {
|
|
|
624
556
|
reportsDir: config.outputDir,
|
|
625
557
|
});
|
|
626
558
|
const serverUrl = await server.start();
|
|
627
|
-
const reportUrl = `${serverUrl}/
|
|
628
|
-
process.stderr.write(
|
|
629
|
-
process.stderr.write(
|
|
630
|
-
process.stderr.write('
|
|
559
|
+
const reportUrl = `${serverUrl}/reports/${report.id}`;
|
|
560
|
+
process.stderr.write(tCli('cli.run.report_server_running', lang, { url: serverUrl }));
|
|
561
|
+
process.stderr.write(tCli('cli.run.report_server_view', lang, { url: reportUrl }));
|
|
562
|
+
process.stderr.write(tCli('cli.run.report_server_stop', lang));
|
|
631
563
|
// Auto-open report in browser
|
|
632
564
|
const { platform } = await import('node:os');
|
|
633
565
|
const openCmd = platform() === 'darwin' ? 'open' : platform() === 'win32' ? 'start' : 'xdg-open';
|
|
@@ -635,13 +567,13 @@ async function handleRun(argv) {
|
|
|
635
567
|
execFileCb(openCmd, [reportUrl], () => { });
|
|
636
568
|
}
|
|
637
569
|
else if (!values['no-serve']) {
|
|
638
|
-
process.stderr.write('
|
|
639
|
-
process.stderr.write(
|
|
570
|
+
process.stderr.write(tCli('cli.run.no_serve_in_non_tty', lang));
|
|
571
|
+
process.stderr.write(tCli('cli.run.no_serve_view_hint', lang, { dir: config.outputDir }));
|
|
640
572
|
}
|
|
641
573
|
}
|
|
642
574
|
}
|
|
643
575
|
catch (err) {
|
|
644
|
-
console.error(
|
|
576
|
+
console.error(tCli('cli.common.error_prefix', lang, { message: err.message }));
|
|
645
577
|
process.exit(1);
|
|
646
578
|
}
|
|
647
579
|
}
|
|
@@ -649,9 +581,11 @@ async function handleRun(argv) {
|
|
|
649
581
|
// handleReport
|
|
650
582
|
// ---------------------------------------------------------------------------
|
|
651
583
|
async function handleReport(argv) {
|
|
584
|
+
const lang = langFromArgv(argv);
|
|
652
585
|
const { values } = parseArgs({
|
|
653
586
|
args: argv,
|
|
654
587
|
options: {
|
|
588
|
+
...COMMON_OPTIONS,
|
|
655
589
|
port: { type: 'string', default: '7799' },
|
|
656
590
|
'reports-dir': { type: 'string', default: DEFAULT_REPORTS_DIR },
|
|
657
591
|
export: { type: 'string' },
|
|
@@ -684,7 +618,7 @@ async function handleReport(argv) {
|
|
|
684
618
|
const store = createFileStore(resolve(values['reports-dir']));
|
|
685
619
|
const report = await store.get(values.export);
|
|
686
620
|
if (!report) {
|
|
687
|
-
console.error(
|
|
621
|
+
console.error(tCli('cli.common.report_not_found', lang, { id: values.export }));
|
|
688
622
|
process.exit(1);
|
|
689
623
|
}
|
|
690
624
|
const html = report.each ? renderEachRunDetail(report) : renderRunDetail(report);
|
|
@@ -779,10 +713,12 @@ function parseLastWindow(spec) {
|
|
|
779
713
|
return new Date(Date.now() - ms).toISOString();
|
|
780
714
|
}
|
|
781
715
|
async function handleAnalyze(argv) {
|
|
716
|
+
const lang = langFromArgv(argv);
|
|
782
717
|
const { values, positionals } = parseArgs({
|
|
783
718
|
args: argv,
|
|
784
719
|
allowPositionals: true,
|
|
785
720
|
options: {
|
|
721
|
+
...COMMON_OPTIONS,
|
|
786
722
|
kb: { type: 'string' },
|
|
787
723
|
last: { type: 'string' },
|
|
788
724
|
from: { type: 'string' },
|
|
@@ -793,7 +729,7 @@ async function handleAnalyze(argv) {
|
|
|
793
729
|
});
|
|
794
730
|
const dir = positionals[0];
|
|
795
731
|
if (!dir) {
|
|
796
|
-
console.error('
|
|
732
|
+
console.error(tCli('cli.help.analyze_usage', lang));
|
|
797
733
|
process.exit(1);
|
|
798
734
|
}
|
|
799
735
|
const tracePath = resolve(dir);
|
|
@@ -842,29 +778,32 @@ async function handleAnalyze(argv) {
|
|
|
842
778
|
console.log(skillRows.join('\n'));
|
|
843
779
|
console.log('');
|
|
844
780
|
console.log(`report written to: ${jsonPath}`);
|
|
845
|
-
console.log(
|
|
781
|
+
console.log(tCli('cli.analyze.view_in_browser', lang));
|
|
846
782
|
}
|
|
847
783
|
async function handleInit(argv) {
|
|
784
|
+
const lang = langFromArgv(argv);
|
|
848
785
|
const targetDir = resolve(argv[0] || '.');
|
|
849
786
|
const { writeFileSync, mkdirSync } = await import('node:fs');
|
|
850
787
|
mkdirSync(join(targetDir, 'skills'), { recursive: true });
|
|
851
788
|
writeFileSync(join(targetDir, 'eval-samples.json'), INIT_SAMPLES);
|
|
852
789
|
writeFileSync(join(targetDir, 'skills', 'v1.md'), INIT_SKILL_V1);
|
|
853
790
|
writeFileSync(join(targetDir, 'skills', 'v2.md'), INIT_SKILL_V2);
|
|
854
|
-
console.log(
|
|
791
|
+
console.log(tCli('cli.init.scaffolded', lang, { dir: targetDir }));
|
|
855
792
|
console.log('');
|
|
856
|
-
console.log('
|
|
857
|
-
console.log('
|
|
858
|
-
console.log('
|
|
859
|
-
console.log('
|
|
793
|
+
console.log(tCli('cli.init.next_steps_title', lang));
|
|
794
|
+
console.log(tCli('cli.init.next_step_edit_samples', lang));
|
|
795
|
+
console.log(tCli('cli.init.next_step_edit_skills', lang));
|
|
796
|
+
console.log(tCli('cli.init.next_step_run', lang));
|
|
860
797
|
}
|
|
861
798
|
// ---------------------------------------------------------------------------
|
|
862
799
|
// handleGenSamples
|
|
863
800
|
// ---------------------------------------------------------------------------
|
|
864
801
|
async function handleGenSamples(argv) {
|
|
802
|
+
const lang = langFromArgv(argv);
|
|
865
803
|
const { values } = parseArgs({
|
|
866
804
|
args: argv,
|
|
867
805
|
options: {
|
|
806
|
+
...COMMON_OPTIONS,
|
|
868
807
|
each: { type: 'boolean', default: false },
|
|
869
808
|
count: { type: 'string', default: '5' },
|
|
870
809
|
model: { type: 'string', default: 'sonnet' },
|
|
@@ -881,7 +820,7 @@ async function handleGenSamples(argv) {
|
|
|
881
820
|
// Batch mode: generate for all skills missing eval-samples
|
|
882
821
|
const skillDir = resolve(values['skill-dir']);
|
|
883
822
|
if (!existsSync(skillDir)) {
|
|
884
|
-
console.error(
|
|
823
|
+
console.error(tCli('cli.common.skill_dir_not_found', lang, { path: skillDir }));
|
|
885
824
|
process.exit(1);
|
|
886
825
|
}
|
|
887
826
|
const { readdirSync, statSync } = await import('node:fs');
|
|
@@ -909,55 +848,63 @@ async function handleGenSamples(argv) {
|
|
|
909
848
|
continue;
|
|
910
849
|
}
|
|
911
850
|
if (existsSync(samplesPath)) {
|
|
912
|
-
process.stderr.write(
|
|
851
|
+
process.stderr.write(tCli('cli.gen.skill_skipped_existing', lang, { name }));
|
|
913
852
|
continue;
|
|
914
853
|
}
|
|
915
|
-
process.stderr.write(
|
|
854
|
+
process.stderr.write(tCli('cli.gen.skill_generating', lang, { name, count }));
|
|
916
855
|
try {
|
|
917
856
|
const skillContent = readFileSync(skillPath, 'utf-8');
|
|
918
857
|
const { samples, costUSD } = await generateSamples({ skillContent, count, model });
|
|
919
858
|
writeFileSync(samplesPath, JSON.stringify(samples, null, 2));
|
|
920
|
-
|
|
859
|
+
const cost = costUSD > 0 ? ` $${costUSD.toFixed(4)}` : '';
|
|
860
|
+
process.stderr.write(tCli('cli.gen.skill_done', lang, {
|
|
861
|
+
name, n: samples.length, path: samplesPath, cost,
|
|
862
|
+
}));
|
|
921
863
|
generated++;
|
|
922
864
|
}
|
|
923
865
|
catch (err) {
|
|
924
|
-
process.stderr.write(
|
|
866
|
+
process.stderr.write(tCli('cli.gen.skill_failed', lang, {
|
|
867
|
+
name, message: err.message,
|
|
868
|
+
}));
|
|
925
869
|
}
|
|
926
870
|
}
|
|
927
871
|
if (generated === 0) {
|
|
928
|
-
console.log('
|
|
872
|
+
console.log(tCli('cli.gen.batch_none_needed', lang));
|
|
929
873
|
}
|
|
930
874
|
else {
|
|
931
|
-
console.log(
|
|
875
|
+
console.log(tCli('cli.gen.batch_summary', lang, { n: generated }));
|
|
932
876
|
}
|
|
933
877
|
}
|
|
934
878
|
else {
|
|
935
879
|
// Single skill mode
|
|
936
880
|
const skillPath = argv.find((a) => !a.startsWith('-'));
|
|
937
881
|
if (!skillPath) {
|
|
938
|
-
console.error('
|
|
882
|
+
console.error(tCli('cli.gen.specify_skill_path', lang));
|
|
939
883
|
process.exit(1);
|
|
940
884
|
}
|
|
941
885
|
const resolvedPath = resolve(skillPath);
|
|
942
886
|
if (!existsSync(resolvedPath)) {
|
|
943
|
-
console.error(
|
|
887
|
+
console.error(tCli('cli.common.skill_file_not_found', lang, { path: resolvedPath }));
|
|
944
888
|
process.exit(1);
|
|
945
889
|
}
|
|
946
890
|
const skillContent = readFileSync(resolvedPath, 'utf-8');
|
|
947
891
|
const outputPath = resolve('eval-samples.json');
|
|
948
892
|
if (existsSync(outputPath)) {
|
|
949
|
-
console.error(
|
|
893
|
+
console.error(tCli('cli.gen.samples_already_exists', lang));
|
|
950
894
|
process.exit(1);
|
|
951
895
|
}
|
|
952
|
-
process.stderr.write(
|
|
896
|
+
process.stderr.write(tCli('cli.gen.single_generating', lang, { count }));
|
|
953
897
|
try {
|
|
954
898
|
const { samples, costUSD } = await generateSamples({ skillContent, count, model });
|
|
955
899
|
writeFileSync(outputPath, JSON.stringify(samples, null, 2));
|
|
956
|
-
|
|
957
|
-
|
|
900
|
+
const cost = costUSD > 0 ? ` $${costUSD.toFixed(4)}` : '';
|
|
901
|
+
process.stderr.write(tCli('cli.gen.single_done', lang, {
|
|
902
|
+
n: samples.length, path: outputPath, cost,
|
|
903
|
+
}));
|
|
904
|
+
console.log(tCli('cli.gen.review_hint', lang));
|
|
958
905
|
}
|
|
959
906
|
catch (err) {
|
|
960
|
-
console.error(
|
|
907
|
+
console.error(tCli('cli.gen.failed', lang, { message: err.message }));
|
|
961
908
|
process.exit(1);
|
|
962
909
|
}
|
|
963
910
|
}
|
|
@@ -966,9 +913,11 @@ async function handleGenSamples(argv) {
|
|
|
966
913
|
// handleEvolve
|
|
967
914
|
// ---------------------------------------------------------------------------
|
|
968
915
|
async function handleEvolve(argv) {
|
|
916
|
+
const lang = langFromArgv(argv);
|
|
969
917
|
const { values } = parseArgs({
|
|
970
918
|
args: argv,
|
|
971
919
|
options: {
|
|
920
|
+
...COMMON_OPTIONS,
|
|
972
921
|
rounds: { type: 'string', default: '5' },
|
|
973
922
|
target: { type: 'string' },
|
|
974
923
|
samples: { type: 'string', default: 'eval-samples.json' },
|
|
@@ -985,7 +934,7 @@ async function handleEvolve(argv) {
|
|
|
985
934
|
});
|
|
986
935
|
const skillPath = argv.find((a) => !a.startsWith('-'));
|
|
987
936
|
if (!skillPath) {
|
|
988
|
-
console.error('
|
|
937
|
+
console.error(tCli('cli.evolve.specify_skill_path', lang));
|
|
989
938
|
process.exit(1);
|
|
990
939
|
}
|
|
991
940
|
let samplesFile = values.samples ?? 'eval-samples.json';
|
|
@@ -996,7 +945,7 @@ async function handleEvolve(argv) {
|
|
|
996
945
|
samplesFile = 'eval-samples.yml';
|
|
997
946
|
}
|
|
998
947
|
const { evolveSkill } = await import('./authoring/evolver.js');
|
|
999
|
-
process.stderr.write(
|
|
948
|
+
process.stderr.write(tCli('cli.evolve.section_header', lang, { path: skillPath }));
|
|
1000
949
|
try {
|
|
1001
950
|
const result = await evolveSkill({
|
|
1002
951
|
skillPath: resolve(skillPath),
|
|
@@ -1010,59 +959,93 @@ async function handleEvolve(argv) {
|
|
|
1010
959
|
concurrency: Math.max(1, Number(values.concurrency) || 1),
|
|
1011
960
|
timeoutMs: Math.max(1, Number(values.timeout) || 120) * 1000,
|
|
1012
961
|
skipPreflight: values['skip-preflight'],
|
|
1013
|
-
onProgress:
|
|
962
|
+
onProgress: makeOnProgress(lang),
|
|
1014
963
|
onRoundProgress({ round, totalRounds: _totalRounds, phase, score, delta, accepted, costUSD, error }) {
|
|
1015
964
|
if (phase === 'baseline') {
|
|
1016
|
-
process.stderr.write(
|
|
965
|
+
process.stderr.write(tCli('cli.evolve.round_baseline', lang, {
|
|
966
|
+
score: score.toFixed(2), cost: costUSD.toFixed(4),
|
|
967
|
+
}));
|
|
1017
968
|
}
|
|
1018
969
|
else if (phase === 'error') {
|
|
1019
|
-
process.stderr.write(
|
|
970
|
+
process.stderr.write(tCli('cli.evolve.round_error', lang, {
|
|
971
|
+
round, error: String(error ?? ''),
|
|
972
|
+
}));
|
|
1020
973
|
}
|
|
1021
974
|
else if (phase === 'done') {
|
|
1022
|
-
const
|
|
975
|
+
const delta_ = delta >= 0 ? `+${delta.toFixed(2)}` : delta.toFixed(2);
|
|
1023
976
|
const status = accepted ? '✓ ACCEPT' : '✗ REJECT';
|
|
1024
|
-
process.stderr.write(
|
|
977
|
+
process.stderr.write(tCli('cli.evolve.round_done', lang, {
|
|
978
|
+
round, score: score.toFixed(2), delta: delta_, status, cost: costUSD.toFixed(4),
|
|
979
|
+
}));
|
|
1025
980
|
}
|
|
1026
981
|
},
|
|
1027
982
|
});
|
|
1028
983
|
const improvement = result.startScore > 0
|
|
1029
984
|
? ((result.finalScore - result.startScore) / result.startScore * 100).toFixed(1)
|
|
1030
985
|
: '0';
|
|
1031
|
-
process.stderr.write(
|
|
1032
|
-
|
|
1033
|
-
|
|
986
|
+
process.stderr.write(tCli('cli.evolve.summary', lang, {
|
|
987
|
+
start: result.startScore.toFixed(2), final: result.finalScore.toFixed(2),
|
|
988
|
+
percent: improvement, rounds: result.totalRounds, cost: result.totalCostUSD.toFixed(4),
|
|
989
|
+
}));
|
|
990
|
+
process.stderr.write(tCli('cli.evolve.best_path', lang, {
|
|
991
|
+
best: result.bestSkillPath, target: resolve(skillPath),
|
|
992
|
+
}));
|
|
993
|
+
process.stderr.write(tCli('cli.evolve.versions_saved', lang, {
|
|
994
|
+
dir: join(resolve(skillPath, '..'), 'evolve'),
|
|
995
|
+
}));
|
|
1034
996
|
if (result.reportId) {
|
|
1035
|
-
process.stderr.write(
|
|
997
|
+
process.stderr.write(tCli('cli.evolve.report_link', lang, { id: result.reportId }));
|
|
1036
998
|
}
|
|
1037
999
|
console.log(JSON.stringify(result, null, 2));
|
|
1038
1000
|
}
|
|
1039
1001
|
catch (err) {
|
|
1040
|
-
console.error(
|
|
1002
|
+
console.error(tCli('cli.common.error_prefix', lang, { message: err.message }));
|
|
1041
1003
|
process.exit(1);
|
|
1042
1004
|
}
|
|
1043
1005
|
}
|
|
1044
1006
|
// ---------------------------------------------------------------------------
|
|
1045
|
-
//
|
|
1007
|
+
// handleGate — 跑评测 + 应用 gate, exit code 0/1 适合 CI/CD pipeline 调用。
|
|
1008
|
+
// 内部 = runEvaluation + computeVerdict + formatVerdictText, 与 bench verdict
|
|
1009
|
+
// 共用决策内核(只是 verdict 读已有报告, gate 跑完再判)。
|
|
1046
1010
|
// ---------------------------------------------------------------------------
|
|
1047
|
-
async function
|
|
1011
|
+
async function handleGate(argv) {
|
|
1012
|
+
const lang = langFromArgv(argv);
|
|
1013
|
+
if (argv[0] === '--help' || argv[0] === '-h') {
|
|
1014
|
+
console.log(tCli('cli.help.main', lang).trim());
|
|
1015
|
+
process.exit(0);
|
|
1016
|
+
}
|
|
1048
1017
|
const { values, config } = parseRunConfig(argv, {
|
|
1049
1018
|
threshold: { type: 'string', default: '3.5' },
|
|
1019
|
+
'trivial-diff': { type: 'string' },
|
|
1050
1020
|
});
|
|
1051
1021
|
const { runEvaluation } = await import('./eval-workflows/run-evaluation.js');
|
|
1052
|
-
config.onProgress =
|
|
1022
|
+
config.onProgress = makeOnProgress(lang);
|
|
1053
1023
|
try {
|
|
1054
1024
|
const { report } = (await runEvaluation(config));
|
|
1055
|
-
const threshold = Number(values.threshold);
|
|
1056
1025
|
if (report.dryRun) {
|
|
1057
|
-
console.log('
|
|
1026
|
+
console.log('Gate dry-run: no scores to check');
|
|
1027
|
+
process.exit(0);
|
|
1028
|
+
}
|
|
1029
|
+
// gate 内核 = run + verdict, 自动覆盖 omk 全部决策维度(三层 layer-gate /
|
|
1030
|
+
// bootstrap diff CI / saturation / Krippendorff α)。computeVerdict 是单一
|
|
1031
|
+
// 决策源, exit code 跟 verdict.level 走 — 数据 underpowered 直接 FAIL,
|
|
1032
|
+
// 堵住"过 PASS 就 deploy"的漏洞。
|
|
1033
|
+
const { computeVerdict, formatVerdictText } = await import('./eval-core/verdict.js');
|
|
1034
|
+
const result = computeVerdict(report, {
|
|
1035
|
+
gateThreshold: Number(values.threshold),
|
|
1036
|
+
triviallySmallDiff: values['trivial-diff'] != null ? Number(values['trivial-diff']) : undefined,
|
|
1037
|
+
});
|
|
1038
|
+
console.log(formatVerdictText(result, { verbose: true }));
|
|
1039
|
+
// exit code 与 handleVerdict 对齐:只有 PROGRESS / SOLO-pass 才 0,
|
|
1040
|
+
// NOISE / UNDERPOWERED / CAUTIOUS / REGRESS 全 1。pipeline `omk bench gate
|
|
1041
|
+
// && deploy` 数据不显著就不会误 deploy。
|
|
1042
|
+
if (result.level === 'PROGRESS') {
|
|
1058
1043
|
process.exit(0);
|
|
1059
1044
|
}
|
|
1060
|
-
|
|
1061
|
-
|
|
1062
|
-
|
|
1063
|
-
|
|
1064
|
-
console.log(line);
|
|
1065
|
-
process.exit(allPass ? 0 : 1);
|
|
1045
|
+
if (result.level === 'SOLO' && result.headline.includes('PASS')) {
|
|
1046
|
+
process.exit(0);
|
|
1047
|
+
}
|
|
1048
|
+
process.exit(1);
|
|
1066
1049
|
}
|
|
1067
1050
|
catch (err) {
|
|
1068
1051
|
console.error(`Error: ${err.message}`);
|
|
@@ -1073,6 +1056,7 @@ async function handleCi(argv) {
|
|
|
1073
1056
|
// handleDiff
|
|
1074
1057
|
// ---------------------------------------------------------------------------
|
|
1075
1058
|
async function handleDiff(argv) {
|
|
1059
|
+
const lang = langFromArgv(argv);
|
|
1076
1060
|
// Flag-aware split: separate positional report IDs from flags so we can support
|
|
1077
1061
|
// omk bench diff <id> — within-report sample-level (v0.22)
|
|
1078
1062
|
// omk bench diff <id1> <id2> — cross-report variant-level (legacy)
|
|
@@ -1094,22 +1078,13 @@ async function handleDiff(argv) {
|
|
|
1094
1078
|
}
|
|
1095
1079
|
}
|
|
1096
1080
|
if (positional.length === 0) {
|
|
1097
|
-
console.error(
|
|
1098
|
-
'Usage:',
|
|
1099
|
-
' omk bench diff <reportId> within-report per-sample diff (v0.22)',
|
|
1100
|
-
' omk bench diff <reportId1> <reportId2> cross-report variant-level diff',
|
|
1101
|
-
'',
|
|
1102
|
-
'Options:',
|
|
1103
|
-
' --regressions-only 只列 treatment < control 的样本',
|
|
1104
|
-
' --threshold <num> regression 阈值 (default 0,即任一负 Δ 算回退)',
|
|
1105
|
-
' --variant <name> within-report 模式下指定要钻取的 variant (default: variants[1])',
|
|
1106
|
-
' --top <n> 只列差距最大的前 N 个样本',
|
|
1107
|
-
].join('\n'));
|
|
1081
|
+
console.error(tCli('cli.help.diff_usage', lang));
|
|
1108
1082
|
process.exit(positional.length === 0 ? 1 : 0);
|
|
1109
1083
|
}
|
|
1110
1084
|
const { values } = parseArgs({
|
|
1111
1085
|
args: flagArgs,
|
|
1112
1086
|
options: {
|
|
1087
|
+
...COMMON_OPTIONS,
|
|
1113
1088
|
'regressions-only': { type: 'boolean', default: false },
|
|
1114
1089
|
threshold: { type: 'string' },
|
|
1115
1090
|
variant: { type: 'string' },
|
|
@@ -1120,18 +1095,18 @@ async function handleDiff(argv) {
|
|
|
1120
1095
|
const { createFileStore } = await import('./server/report-store.js');
|
|
1121
1096
|
const store = createFileStore(resolve(DEFAULT_REPORTS_DIR));
|
|
1122
1097
|
if (positional.length === 1) {
|
|
1123
|
-
await runSampleLevelDiff(positional[0], store, values);
|
|
1098
|
+
await runSampleLevelDiff(positional[0], store, values, lang);
|
|
1124
1099
|
return;
|
|
1125
1100
|
}
|
|
1126
1101
|
const [id1, id2] = positional;
|
|
1127
1102
|
const r1 = await store.get(id1);
|
|
1128
1103
|
const r2 = await store.get(id2);
|
|
1129
1104
|
if (!r1) {
|
|
1130
|
-
console.error(
|
|
1105
|
+
console.error(tCli('cli.common.report_not_found', lang, { id: id1 }));
|
|
1131
1106
|
process.exit(1);
|
|
1132
1107
|
}
|
|
1133
1108
|
if (!r2) {
|
|
1134
|
-
console.error(
|
|
1109
|
+
console.error(tCli('cli.common.report_not_found', lang, { id: id2 }));
|
|
1135
1110
|
process.exit(1);
|
|
1136
1111
|
}
|
|
1137
1112
|
console.log(`\n Diff: ${id1} → ${id2}\n`);
|
|
@@ -1187,10 +1162,10 @@ async function handleDiff(argv) {
|
|
|
1187
1162
|
* Default focus is variants[0] (control) vs variants[1] (treatment), but
|
|
1188
1163
|
* `--variant` overrides which variant is the "treatment" side.
|
|
1189
1164
|
*/
|
|
1190
|
-
async function runSampleLevelDiff(reportId, store, flags) {
|
|
1165
|
+
async function runSampleLevelDiff(reportId, store, flags, lang) {
|
|
1191
1166
|
const report = await store.get(reportId);
|
|
1192
1167
|
if (!report) {
|
|
1193
|
-
console.error(
|
|
1168
|
+
console.error(tCli('cli.common.report_not_found', lang, { id: reportId }));
|
|
1194
1169
|
process.exit(1);
|
|
1195
1170
|
}
|
|
1196
1171
|
const variants = report.meta?.variants ?? [];
|
|
@@ -1267,27 +1242,18 @@ async function runSampleLevelDiff(reportId, store, flags) {
|
|
|
1267
1242
|
// handleGold — gold dataset workflow (init / validate / compare)
|
|
1268
1243
|
// ---------------------------------------------------------------------------
|
|
1269
1244
|
async function handleGold(argv) {
|
|
1245
|
+
const lang = langFromArgv(argv);
|
|
1270
1246
|
const sub = argv[0];
|
|
1271
1247
|
const rest = argv.slice(1);
|
|
1272
1248
|
if (!sub || sub === '--help' || sub === '-h') {
|
|
1273
|
-
console.log(
|
|
1274
|
-
'',
|
|
1275
|
-
'Usage: omk bench gold <subcommand>',
|
|
1276
|
-
'',
|
|
1277
|
-
'Subcommands:',
|
|
1278
|
-
' init [--out <dir>] [--annotator <id>] 生成空白 gold dataset 模板',
|
|
1279
|
-
' validate <dir> 校验数据集结构',
|
|
1280
|
-
' compare <reportId> --gold-dir <dir> 与已有 report 计算 α/κ/Pearson',
|
|
1281
|
-
' [--variant <name>] [--reports-dir <d>]',
|
|
1282
|
-
' [--bootstrap-samples N] [--seed N]',
|
|
1283
|
-
'',
|
|
1284
|
-
].join('\n'));
|
|
1249
|
+
console.log(tCli('cli.help.gold', lang));
|
|
1285
1250
|
process.exit(sub ? 0 : 1);
|
|
1286
1251
|
}
|
|
1287
1252
|
if (sub === 'init') {
|
|
1288
1253
|
const { values } = parseArgs({
|
|
1289
1254
|
args: rest,
|
|
1290
1255
|
options: {
|
|
1256
|
+
...COMMON_OPTIONS,
|
|
1291
1257
|
out: { type: 'string', default: './gold-dataset' },
|
|
1292
1258
|
annotator: { type: 'string' },
|
|
1293
1259
|
},
|
|
@@ -1298,10 +1264,12 @@ async function handleGold(argv) {
|
|
|
1298
1264
|
const written = initGoldDataset(values.out, {
|
|
1299
1265
|
annotator: values.annotator,
|
|
1300
1266
|
});
|
|
1301
|
-
console.log(
|
|
1267
|
+
console.log(tCli('cli.gold.created_files', lang, {
|
|
1268
|
+
n: written.length, dir: values.out,
|
|
1269
|
+
}));
|
|
1302
1270
|
for (const p of written)
|
|
1303
1271
|
console.log(` ${p}`);
|
|
1304
|
-
console.log('
|
|
1272
|
+
console.log(tCli('cli.gold.next_step_edit_annotations', lang));
|
|
1305
1273
|
}
|
|
1306
1274
|
catch (err) {
|
|
1307
1275
|
console.error(err.message);
|
|
@@ -1312,13 +1280,13 @@ async function handleGold(argv) {
|
|
|
1312
1280
|
if (sub === 'validate') {
|
|
1313
1281
|
const dir = rest[0];
|
|
1314
1282
|
if (!dir) {
|
|
1315
|
-
console.error('
|
|
1283
|
+
console.error(tCli('cli.common.usage_gold_validate', lang));
|
|
1316
1284
|
process.exit(1);
|
|
1317
1285
|
}
|
|
1318
1286
|
const { validateGoldDataset } = await import('./grading/gold-cli.js');
|
|
1319
1287
|
const result = validateGoldDataset(dir);
|
|
1320
1288
|
if (result.ok) {
|
|
1321
|
-
console.log(
|
|
1289
|
+
console.log(tCli('cli.gold.validate_ok', lang, { n: result.sampleCount }));
|
|
1322
1290
|
return;
|
|
1323
1291
|
}
|
|
1324
1292
|
console.error(`✗ gold dataset has ${result.issues.length} issue(s):`);
|
|
@@ -1335,6 +1303,7 @@ async function handleGold(argv) {
|
|
|
1335
1303
|
const { values } = parseArgs({
|
|
1336
1304
|
args: rest.slice(1),
|
|
1337
1305
|
options: {
|
|
1306
|
+
...COMMON_OPTIONS,
|
|
1338
1307
|
'gold-dir': { type: 'string' },
|
|
1339
1308
|
variant: { type: 'string' },
|
|
1340
1309
|
'reports-dir': { type: 'string', default: DEFAULT_REPORTS_DIR },
|
|
@@ -1366,7 +1335,7 @@ async function handleGold(argv) {
|
|
|
1366
1335
|
const store = createFileStore(resolve(values['reports-dir']));
|
|
1367
1336
|
const report = await store.get(reportId);
|
|
1368
1337
|
if (!report) {
|
|
1369
|
-
console.error(
|
|
1338
|
+
console.error(tCli('cli.common.report_not_found', lang, { id: reportId }));
|
|
1370
1339
|
process.exit(1);
|
|
1371
1340
|
}
|
|
1372
1341
|
const samples = Math.max(100, Number(values['bootstrap-samples']) || 1000);
|
|
@@ -1388,27 +1357,11 @@ async function handleGold(argv) {
|
|
|
1388
1357
|
// handleDebiasValidate — measure length-debias prompt sensitivity (Phase 3a)
|
|
1389
1358
|
// ---------------------------------------------------------------------------
|
|
1390
1359
|
async function handleDebiasValidate(argv) {
|
|
1360
|
+
const lang = langFromArgv(argv);
|
|
1391
1361
|
const sub = argv[0];
|
|
1392
1362
|
const rest = argv.slice(1);
|
|
1393
1363
|
if (!sub || sub === '--help' || sub === '-h') {
|
|
1394
|
-
console.log(
|
|
1395
|
-
'',
|
|
1396
|
-
'Usage: omk bench debias-validate <kind> <reportId> [options]',
|
|
1397
|
-
'',
|
|
1398
|
-
'Kinds:',
|
|
1399
|
-
' length re-judge with the opposite length-debias setting and bootstrap CI',
|
|
1400
|
-
' on the score diff. Cost ~doubles vs the original judge pass.',
|
|
1401
|
-
'',
|
|
1402
|
-
'Options:',
|
|
1403
|
-
' --reports-dir <dir> report store dir (default: ~/.oh-my-knowledge/reports)',
|
|
1404
|
-
' --samples <path> override samples file (default: from report.meta.request)',
|
|
1405
|
-
' --variant <name> which variant to validate (default: first)',
|
|
1406
|
-
' --judge-executor <name> executor for judge calls (default: claude)',
|
|
1407
|
-
' --judge-model <model> judge model id (default: from report)',
|
|
1408
|
-
' --bootstrap-samples N bootstrap iterations (default 1000)',
|
|
1409
|
-
' --seed N deterministic CI seed',
|
|
1410
|
-
'',
|
|
1411
|
-
].join('\n'));
|
|
1364
|
+
console.log(tCli('cli.help.debias_validate', lang));
|
|
1412
1365
|
process.exit(sub ? 0 : 1);
|
|
1413
1366
|
}
|
|
1414
1367
|
if (sub !== 'length') {
|
|
@@ -1423,6 +1376,7 @@ async function handleDebiasValidate(argv) {
|
|
|
1423
1376
|
const { values } = parseArgs({
|
|
1424
1377
|
args: rest.slice(1),
|
|
1425
1378
|
options: {
|
|
1379
|
+
...COMMON_OPTIONS,
|
|
1426
1380
|
'reports-dir': { type: 'string', default: DEFAULT_REPORTS_DIR },
|
|
1427
1381
|
samples: { type: 'string' },
|
|
1428
1382
|
variant: { type: 'string' },
|
|
@@ -1437,7 +1391,7 @@ async function handleDebiasValidate(argv) {
|
|
|
1437
1391
|
const store = createFileStore(resolve(values['reports-dir']));
|
|
1438
1392
|
const report = await store.get(reportId);
|
|
1439
1393
|
if (!report) {
|
|
1440
|
-
console.error(
|
|
1394
|
+
console.error(tCli('cli.common.report_not_found', lang, { id: reportId }));
|
|
1441
1395
|
process.exit(1);
|
|
1442
1396
|
}
|
|
1443
1397
|
// Resolve samples path: --samples overrides; otherwise read from report.meta.request.
|
|
@@ -1452,10 +1406,10 @@ async function handleDebiasValidate(argv) {
|
|
|
1452
1406
|
const judgeModel = values['judge-model']
|
|
1453
1407
|
?? report.meta?.judgeModel;
|
|
1454
1408
|
if (!judgeModel) {
|
|
1455
|
-
console.error('
|
|
1409
|
+
console.error(tCli('cli.common.no_judge_model', lang));
|
|
1456
1410
|
process.exit(1);
|
|
1457
1411
|
}
|
|
1458
|
-
process.stderr.write('
|
|
1412
|
+
process.stderr.write(tCli('cli.debias.warn_cost_doubles', lang));
|
|
1459
1413
|
const { createExecutor } = await import('./executors/index.js');
|
|
1460
1414
|
const judgeExecutor = createExecutor(values['judge-executor']);
|
|
1461
1415
|
const { validateLengthDebias, formatDebiasValidate } = await import('./grading/debias-validate.js');
|
|
@@ -1479,29 +1433,16 @@ async function handleDebiasValidate(argv) {
|
|
|
1479
1433
|
// handleSaturation — re-compute saturation verdict from a finished report
|
|
1480
1434
|
// ---------------------------------------------------------------------------
|
|
1481
1435
|
async function handleSaturation(argv) {
|
|
1436
|
+
const lang = langFromArgv(argv);
|
|
1482
1437
|
const reportId = argv[0];
|
|
1483
1438
|
if (!reportId || reportId === '--help' || reportId === '-h') {
|
|
1484
|
-
console.log(
|
|
1485
|
-
'',
|
|
1486
|
-
'Usage: omk bench saturation <reportId> [options]',
|
|
1487
|
-
'',
|
|
1488
|
-
'回答"我跑够样本了吗?"。复述已有 report 中持久化的饱和判定。',
|
|
1489
|
-
'',
|
|
1490
|
-
'注:本命令读取 run 时跑出的 verdict(运行时已用 method=bootstrap-ci-width',
|
|
1491
|
-
'默认阈值 + 3 窗口 持续条件)。如要换 method/threshold 重新计算,需要重跑',
|
|
1492
|
-
'`omk bench run --repeat ≥ 5`(运行时持久化的 trace 不含原始分数,无法',
|
|
1493
|
-
'在事后用其他参数复算)。',
|
|
1494
|
-
'',
|
|
1495
|
-
'Options:',
|
|
1496
|
-
' --reports-dir <dir> report store dir (default: ~/.oh-my-knowledge/reports)',
|
|
1497
|
-
' --variant <name> 只看一个 variant (default: all)',
|
|
1498
|
-
'',
|
|
1499
|
-
].join('\n'));
|
|
1439
|
+
console.log(tCli('cli.help.saturation', lang));
|
|
1500
1440
|
process.exit(reportId ? 0 : 1);
|
|
1501
1441
|
}
|
|
1502
1442
|
const { values } = parseArgs({
|
|
1503
1443
|
args: argv.slice(1),
|
|
1504
1444
|
options: {
|
|
1445
|
+
...COMMON_OPTIONS,
|
|
1505
1446
|
'reports-dir': { type: 'string', default: DEFAULT_REPORTS_DIR },
|
|
1506
1447
|
variant: { type: 'string' },
|
|
1507
1448
|
},
|
|
@@ -1511,12 +1452,12 @@ async function handleSaturation(argv) {
|
|
|
1511
1452
|
const store = createFileStore(resolve(values['reports-dir']));
|
|
1512
1453
|
const report = await store.get(reportId);
|
|
1513
1454
|
if (!report) {
|
|
1514
|
-
console.error(
|
|
1455
|
+
console.error(tCli('cli.common.report_not_found', lang, { id: reportId }));
|
|
1515
1456
|
process.exit(1);
|
|
1516
1457
|
}
|
|
1517
1458
|
const saturation = report.variance?.saturation;
|
|
1518
1459
|
if (!saturation) {
|
|
1519
|
-
console.error('
|
|
1460
|
+
console.error(tCli('cli.saturation.no_data', lang));
|
|
1520
1461
|
process.exit(1);
|
|
1521
1462
|
}
|
|
1522
1463
|
// Print the persisted verdict from the original run. The trace stores
|
|
@@ -1527,22 +1468,32 @@ async function handleSaturation(argv) {
|
|
|
1527
1468
|
// run time to enable post-hoc parameter sweeps.
|
|
1528
1469
|
const variants = report.meta.variants ?? [];
|
|
1529
1470
|
const targetVariants = values.variant ? [values.variant] : variants;
|
|
1530
|
-
console.log(
|
|
1471
|
+
console.log(tCli('cli.saturation.verdict_header', lang));
|
|
1531
1472
|
for (const variant of targetVariants) {
|
|
1532
1473
|
const trace = saturation.perVariant[variant];
|
|
1533
1474
|
if (!trace || trace.length === 0) {
|
|
1534
|
-
console.log(
|
|
1475
|
+
console.log(tCli('cli.saturation.variant_no_trace', lang, { variant }));
|
|
1535
1476
|
continue;
|
|
1536
1477
|
}
|
|
1537
|
-
console.log(
|
|
1538
|
-
console.log(
|
|
1539
|
-
|
|
1478
|
+
console.log(tCli('cli.saturation.variant_label', lang, { variant }));
|
|
1479
|
+
console.log(tCli('cli.saturation.checkpoints', lang, {
|
|
1480
|
+
n: trace.length, list: trace.map((p) => p.n).join(', '),
|
|
1481
|
+
}));
|
|
1482
|
+
const last = trace[trace.length - 1];
|
|
1483
|
+
console.log(tCli('cli.saturation.last_point', lang, {
|
|
1484
|
+
mean: last.mean.toFixed(3), lo: last.ciLow.toFixed(3), hi: last.ciHigh.toFixed(3),
|
|
1485
|
+
}));
|
|
1540
1486
|
if (saturation.verdicts?.[variant]) {
|
|
1541
1487
|
const v = saturation.verdicts[variant];
|
|
1542
|
-
|
|
1488
|
+
const result = v.saturated
|
|
1489
|
+
? tCli('cli.saturation.persisted_verdict_saturated', lang, { n: v.atN ?? '?' })
|
|
1490
|
+
: tCli('cli.saturation.persisted_verdict_unsaturated', lang);
|
|
1491
|
+
console.log(tCli('cli.saturation.persisted_verdict', lang, {
|
|
1492
|
+
method: v.method, result, reason: v.reason,
|
|
1493
|
+
}));
|
|
1543
1494
|
}
|
|
1544
1495
|
else if (trace.length < 5) {
|
|
1545
|
-
console.log(
|
|
1496
|
+
console.log(tCli('cli.saturation.skipped_too_few_points', lang, { n: trace.length }));
|
|
1546
1497
|
}
|
|
1547
1498
|
}
|
|
1548
1499
|
console.log('');
|
|
@@ -1551,34 +1502,16 @@ async function handleSaturation(argv) {
|
|
|
1551
1502
|
// handleVerdict — one-line ship/no-ship verdict (v0.22)
|
|
1552
1503
|
// ---------------------------------------------------------------------------
|
|
1553
1504
|
async function handleVerdict(argv) {
|
|
1505
|
+
const lang = langFromArgv(argv);
|
|
1554
1506
|
const reportId = argv[0];
|
|
1555
1507
|
if (!reportId || reportId === '--help' || reportId === '-h') {
|
|
1556
|
-
console.log(
|
|
1557
|
-
'',
|
|
1558
|
-
'Usage: omk bench verdict <reportId> [options]',
|
|
1559
|
-
'',
|
|
1560
|
-
'聚合 bootstrap CI / 三层 ci-gate / saturation / human α 给出一行结论。',
|
|
1561
|
-
'',
|
|
1562
|
-
'Verdict 等级:',
|
|
1563
|
-
' PROGRESS 显著改进 + 三层全过',
|
|
1564
|
-
' CAUTIOUS 改进真实但有警告 (gate 破 / 幅度太小 / 控制组本身崩)',
|
|
1565
|
-
' REGRESS 显著回退 — 不要 ship',
|
|
1566
|
-
' NOISE CI 跨 0,无法判定',
|
|
1567
|
-
' UNDERPOWERED 样本不足,需要扩 N 重测',
|
|
1568
|
-
' SOLO 单变体报告,无对比对象',
|
|
1569
|
-
'',
|
|
1570
|
-
'Options:',
|
|
1571
|
-
' --reports-dir <dir> report store dir (default: ~/.oh-my-knowledge/reports)',
|
|
1572
|
-
' --threshold <num> 三层 gate 阈值 (default 3.5,匹配 omk bench ci)',
|
|
1573
|
-
' --trivial-diff <num> "幅度太小"阈值 (default 0.1)',
|
|
1574
|
-
' --verbose 展开 per-pair 详情',
|
|
1575
|
-
'',
|
|
1576
|
-
].join('\n'));
|
|
1508
|
+
console.log(tCli('cli.help.verdict', lang));
|
|
1577
1509
|
process.exit(reportId ? 0 : 1);
|
|
1578
1510
|
}
|
|
1579
1511
|
const { values } = parseArgs({
|
|
1580
1512
|
args: argv.slice(1),
|
|
1581
1513
|
options: {
|
|
1514
|
+
...COMMON_OPTIONS,
|
|
1582
1515
|
'reports-dir': { type: 'string', default: DEFAULT_REPORTS_DIR },
|
|
1583
1516
|
threshold: { type: 'string' },
|
|
1584
1517
|
'trivial-diff': { type: 'string' },
|
|
@@ -1590,12 +1523,12 @@ async function handleVerdict(argv) {
|
|
|
1590
1523
|
const store = createFileStore(resolve(values['reports-dir']));
|
|
1591
1524
|
const report = await store.get(reportId);
|
|
1592
1525
|
if (!report) {
|
|
1593
|
-
console.error(
|
|
1526
|
+
console.error(tCli('cli.common.report_not_found', lang, { id: reportId }));
|
|
1594
1527
|
process.exit(1);
|
|
1595
1528
|
}
|
|
1596
1529
|
const { computeVerdict, formatVerdictText } = await import('./eval-core/verdict.js');
|
|
1597
1530
|
const result = computeVerdict(report, {
|
|
1598
|
-
|
|
1531
|
+
gateThreshold: values.threshold != null ? Number(values.threshold) : undefined,
|
|
1599
1532
|
triviallySmallDiff: values['trivial-diff'] != null ? Number(values['trivial-diff']) : undefined,
|
|
1600
1533
|
});
|
|
1601
1534
|
console.log(formatVerdictText(result, { verbose: Boolean(values.verbose) }));
|
|
@@ -1614,31 +1547,16 @@ async function handleVerdict(argv) {
|
|
|
1614
1547
|
// handleDiagnose — per-sample quality diagnostics (v0.23 A)
|
|
1615
1548
|
// ---------------------------------------------------------------------------
|
|
1616
1549
|
async function handleDiagnose(argv) {
|
|
1550
|
+
const lang = langFromArgv(argv);
|
|
1617
1551
|
const reportId = argv[0];
|
|
1618
1552
|
if (!reportId || reportId === '--help' || reportId === '-h') {
|
|
1619
|
-
console.log(
|
|
1620
|
-
'',
|
|
1621
|
-
'Usage: omk bench diagnose <reportId> [options]',
|
|
1622
|
-
'',
|
|
1623
|
-
'诊断样本集本身的质量问题:区分度低 / 重复 / 歧义 / 成本异常 / 全 fail。',
|
|
1624
|
-
'回答"测评结论是否被坏样本污染"——与 omk bench verdict 互补。',
|
|
1625
|
-
'',
|
|
1626
|
-
'Options:',
|
|
1627
|
-
' --reports-dir <dir> report store dir',
|
|
1628
|
-
' --samples <path> 样本文件路径 (用于 near-duplicate 检测;默认从 report.meta.request 读)',
|
|
1629
|
-
' --top <n> 每类只显示前 N 个 (默认 10,0=全部)',
|
|
1630
|
-
' --duplicate-rouge <num> near-duplicate ROUGE-1 阈值 (默认 0.7)',
|
|
1631
|
-
' --ambiguous-stddev <num> 歧义阈值,judge stddev (默认 1.0,需要 --judge-repeat ≥ 2 数据)',
|
|
1632
|
-
' --cost-k <num> 成本异常倍数 vs median (默认 3)',
|
|
1633
|
-
' --latency-k <num> 耗时异常倍数 vs median (默认 3)',
|
|
1634
|
-
' --flat <num> flat_scores 分差阈值 (默认 0.5)',
|
|
1635
|
-
'',
|
|
1636
|
-
].join('\n'));
|
|
1553
|
+
console.log(tCli('cli.help.diagnose', lang));
|
|
1637
1554
|
process.exit(reportId ? 0 : 1);
|
|
1638
1555
|
}
|
|
1639
1556
|
const { values } = parseArgs({
|
|
1640
1557
|
args: argv.slice(1),
|
|
1641
1558
|
options: {
|
|
1559
|
+
...COMMON_OPTIONS,
|
|
1642
1560
|
'reports-dir': { type: 'string', default: DEFAULT_REPORTS_DIR },
|
|
1643
1561
|
samples: { type: 'string' },
|
|
1644
1562
|
top: { type: 'string', default: '10' },
|
|
@@ -1654,7 +1572,7 @@ async function handleDiagnose(argv) {
|
|
|
1654
1572
|
const store = createFileStore(resolve(values['reports-dir']));
|
|
1655
1573
|
const report = await store.get(reportId);
|
|
1656
1574
|
if (!report) {
|
|
1657
|
-
console.error(
|
|
1575
|
+
console.error(tCli('cli.common.report_not_found', lang, { id: reportId }));
|
|
1658
1576
|
process.exit(1);
|
|
1659
1577
|
}
|
|
1660
1578
|
// Try to read the samples file for near-duplicate detection. Source order:
|
|
@@ -1669,7 +1587,9 @@ async function handleDiagnose(argv) {
|
|
|
1669
1587
|
samples = loadSamples(samplesPath).samples;
|
|
1670
1588
|
}
|
|
1671
1589
|
catch (err) {
|
|
1672
|
-
process.stderr.write(
|
|
1590
|
+
process.stderr.write(tCli('cli.common.warn_load_samples_failed', lang, {
|
|
1591
|
+
path: samplesPath, message: err.message,
|
|
1592
|
+
}));
|
|
1673
1593
|
}
|
|
1674
1594
|
}
|
|
1675
1595
|
const topRaw = Number(values.top);
|
|
@@ -1694,29 +1614,16 @@ async function handleDiagnose(argv) {
|
|
|
1694
1614
|
// handleFailures — LLM-driven failure clustering (v0.23 B)
|
|
1695
1615
|
// ---------------------------------------------------------------------------
|
|
1696
1616
|
async function handleFailures(argv) {
|
|
1617
|
+
const lang = langFromArgv(argv);
|
|
1697
1618
|
const reportId = argv[0];
|
|
1698
1619
|
if (!reportId || reportId === '--help' || reportId === '-h') {
|
|
1699
|
-
console.log(
|
|
1700
|
-
'',
|
|
1701
|
-
'Usage: omk bench failures <reportId> [options]',
|
|
1702
|
-
'',
|
|
1703
|
-
'把已有 report 的失败样本喂给一次 LLM 调用,自动聚类 + 给修复建议。',
|
|
1704
|
-
'失败定义:compositeScore < threshold 或 ok=false。',
|
|
1705
|
-
'',
|
|
1706
|
-
'Options:',
|
|
1707
|
-
' --reports-dir <dir> report store dir',
|
|
1708
|
-
' --judge-executor <name> 执行器 (default: claude)',
|
|
1709
|
-
' --judge-model <id> 聚类用的 model (default: 沿用 report.meta.judgeModel)',
|
|
1710
|
-
' --max-clusters <n> 最多多少 cluster (default 5)',
|
|
1711
|
-
' --threshold <num> compositeScore < threshold 算失败 (default 3)',
|
|
1712
|
-
' --max-feed <n> 最多喂给 LLM 多少条 (default 50,超出取最差)',
|
|
1713
|
-
'',
|
|
1714
|
-
].join('\n'));
|
|
1620
|
+
console.log(tCli('cli.help.failures', lang));
|
|
1715
1621
|
process.exit(reportId ? 0 : 1);
|
|
1716
1622
|
}
|
|
1717
1623
|
const { values } = parseArgs({
|
|
1718
1624
|
args: argv.slice(1),
|
|
1719
1625
|
options: {
|
|
1626
|
+
...COMMON_OPTIONS,
|
|
1720
1627
|
'reports-dir': { type: 'string', default: DEFAULT_REPORTS_DIR },
|
|
1721
1628
|
'judge-executor': { type: 'string', default: 'claude' },
|
|
1722
1629
|
'judge-model': { type: 'string' },
|
|
@@ -1730,12 +1637,12 @@ async function handleFailures(argv) {
|
|
|
1730
1637
|
const store = createFileStore(resolve(values['reports-dir']));
|
|
1731
1638
|
const report = await store.get(reportId);
|
|
1732
1639
|
if (!report) {
|
|
1733
|
-
console.error(
|
|
1640
|
+
console.error(tCli('cli.common.report_not_found', lang, { id: reportId }));
|
|
1734
1641
|
process.exit(1);
|
|
1735
1642
|
}
|
|
1736
1643
|
const judgeModel = values['judge-model'] ?? report.meta?.judgeModel;
|
|
1737
1644
|
if (!judgeModel) {
|
|
1738
|
-
console.error('
|
|
1645
|
+
console.error(tCli('cli.common.no_judge_model', lang));
|
|
1739
1646
|
process.exit(1);
|
|
1740
1647
|
}
|
|
1741
1648
|
const { createExecutor } = await import('./executors/index.js');
|