oh-my-knowledge 0.0.1 → 0.18.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +537 -332
- package/dist/src/analysis/coverage-analyzer.d.ts +57 -0
- package/dist/src/analysis/coverage-analyzer.d.ts.map +1 -0
- package/dist/src/analysis/coverage-analyzer.js +262 -0
- package/dist/src/analysis/coverage-analyzer.js.map +1 -0
- package/dist/src/analysis/gap-analyzer.d.ts +114 -0
- package/dist/src/analysis/gap-analyzer.d.ts.map +1 -0
- package/dist/src/analysis/gap-analyzer.js +439 -0
- package/dist/src/analysis/gap-analyzer.js.map +1 -0
- package/dist/src/analysis/hedging-classifier.d.ts +34 -0
- package/dist/src/analysis/hedging-classifier.d.ts.map +1 -0
- package/dist/src/analysis/hedging-classifier.js +144 -0
- package/dist/src/analysis/hedging-classifier.js.map +1 -0
- package/dist/src/analysis/report-diagnostics.d.ts +9 -0
- package/dist/src/analysis/report-diagnostics.d.ts.map +1 -0
- package/dist/src/analysis/report-diagnostics.js +631 -0
- package/dist/src/analysis/report-diagnostics.js.map +1 -0
- package/dist/src/authoring/evolver.d.ts +64 -0
- package/dist/src/authoring/evolver.d.ts.map +1 -0
- package/dist/src/authoring/evolver.js +275 -0
- package/dist/src/authoring/evolver.js.map +1 -0
- package/dist/src/authoring/generator.d.ts +13 -0
- package/dist/src/authoring/generator.d.ts.map +1 -0
- package/dist/src/authoring/generator.js +60 -0
- package/dist/src/authoring/generator.js.map +1 -0
- package/dist/src/cli.d.ts +3 -0
- package/dist/src/cli.d.ts.map +1 -0
- package/dist/src/cli.js +1004 -0
- package/dist/src/cli.js.map +1 -0
- package/dist/src/eval-core/cache.d.ts +11 -0
- package/dist/src/eval-core/cache.d.ts.map +1 -0
- package/dist/src/eval-core/cache.js +53 -0
- package/dist/src/eval-core/cache.js.map +1 -0
- package/dist/src/eval-core/ci-gates.d.ts +17 -0
- package/dist/src/eval-core/ci-gates.d.ts.map +1 -0
- package/dist/src/eval-core/ci-gates.js +42 -0
- package/dist/src/eval-core/ci-gates.js.map +1 -0
- package/dist/src/eval-core/dependency-checker.d.ts +37 -0
- package/dist/src/eval-core/dependency-checker.d.ts.map +1 -0
- package/dist/src/eval-core/dependency-checker.js +280 -0
- package/dist/src/eval-core/dependency-checker.js.map +1 -0
- package/dist/src/eval-core/evaluation-execution.d.ts +26 -0
- package/dist/src/eval-core/evaluation-execution.d.ts.map +1 -0
- package/dist/src/eval-core/evaluation-execution.js +196 -0
- package/dist/src/eval-core/evaluation-execution.js.map +1 -0
- package/dist/src/eval-core/evaluation-job.d.ts +48 -0
- package/dist/src/eval-core/evaluation-job.d.ts.map +1 -0
- package/dist/src/eval-core/evaluation-job.js +112 -0
- package/dist/src/eval-core/evaluation-job.js.map +1 -0
- package/dist/src/eval-core/evaluation-reporting.d.ts +28 -0
- package/dist/src/eval-core/evaluation-reporting.d.ts.map +1 -0
- package/dist/src/eval-core/evaluation-reporting.js +123 -0
- package/dist/src/eval-core/evaluation-reporting.js.map +1 -0
- package/dist/src/eval-core/execution-strategy.d.ts +11 -0
- package/dist/src/eval-core/execution-strategy.d.ts.map +1 -0
- package/dist/src/eval-core/execution-strategy.js +121 -0
- package/dist/src/eval-core/execution-strategy.js.map +1 -0
- package/dist/src/eval-core/fact-checker.d.ts +25 -0
- package/dist/src/eval-core/fact-checker.d.ts.map +1 -0
- package/dist/src/eval-core/fact-checker.js +66 -0
- package/dist/src/eval-core/fact-checker.js.map +1 -0
- package/dist/src/eval-core/schema.d.ts +30 -0
- package/dist/src/eval-core/schema.d.ts.map +1 -0
- package/dist/src/eval-core/schema.js +195 -0
- package/dist/src/eval-core/schema.js.map +1 -0
- package/dist/src/eval-core/statistics.d.ts +86 -0
- package/dist/src/eval-core/statistics.d.ts.map +1 -0
- package/dist/src/eval-core/statistics.js +210 -0
- package/dist/src/eval-core/statistics.js.map +1 -0
- package/dist/src/eval-core/task-planner.d.ts +4 -0
- package/dist/src/eval-core/task-planner.d.ts.map +1 -0
- package/dist/src/eval-core/task-planner.js +33 -0
- package/dist/src/eval-core/task-planner.js.map +1 -0
- package/dist/src/eval-workflows/each-evaluation-workflow.d.ts +133 -0
- package/dist/src/eval-workflows/each-evaluation-workflow.d.ts.map +1 -0
- package/dist/src/eval-workflows/each-evaluation-workflow.js +169 -0
- package/dist/src/eval-workflows/each-evaluation-workflow.js.map +1 -0
- package/dist/src/eval-workflows/evaluation-pipeline.d.ts +38 -0
- package/dist/src/eval-workflows/evaluation-pipeline.d.ts.map +1 -0
- package/dist/src/eval-workflows/evaluation-pipeline.js +214 -0
- package/dist/src/eval-workflows/evaluation-pipeline.js.map +1 -0
- package/dist/src/eval-workflows/evaluation-preparation.d.ts +64 -0
- package/dist/src/eval-workflows/evaluation-preparation.d.ts.map +1 -0
- package/dist/src/eval-workflows/evaluation-preparation.js +90 -0
- package/dist/src/eval-workflows/evaluation-preparation.js.map +1 -0
- package/dist/src/eval-workflows/run-evaluation.d.ts +105 -0
- package/dist/src/eval-workflows/run-evaluation.d.ts.map +1 -0
- package/dist/src/eval-workflows/run-evaluation.js +264 -0
- package/dist/src/eval-workflows/run-evaluation.js.map +1 -0
- package/dist/src/executors/anthropic-api.d.ts +3 -0
- package/dist/src/executors/anthropic-api.d.ts.map +1 -0
- package/dist/src/executors/anthropic-api.js +43 -0
- package/dist/src/executors/anthropic-api.js.map +1 -0
- package/dist/src/executors/claude-cli.d.ts +3 -0
- package/dist/src/executors/claude-cli.d.ts.map +1 -0
- package/dist/src/executors/claude-cli.js +103 -0
- package/dist/src/executors/claude-cli.js.map +1 -0
- package/dist/src/executors/claude-sdk-trace.d.ts +10 -0
- package/dist/src/executors/claude-sdk-trace.d.ts.map +1 -0
- package/dist/src/executors/claude-sdk-trace.js +100 -0
- package/dist/src/executors/claude-sdk-trace.js.map +1 -0
- package/dist/src/executors/claude-sdk.d.ts +3 -0
- package/dist/src/executors/claude-sdk.d.ts.map +1 -0
- package/dist/src/executors/claude-sdk.js +160 -0
- package/dist/src/executors/claude-sdk.js.map +1 -0
- package/dist/src/executors/gemini.d.ts +3 -0
- package/dist/src/executors/gemini.d.ts.map +1 -0
- package/dist/src/executors/gemini.js +78 -0
- package/dist/src/executors/gemini.js.map +1 -0
- package/dist/src/executors/index.d.ts +7 -0
- package/dist/src/executors/index.d.ts.map +1 -0
- package/dist/src/executors/index.js +22 -0
- package/dist/src/executors/index.js.map +1 -0
- package/dist/src/executors/openai-api.d.ts +3 -0
- package/dist/src/executors/openai-api.d.ts.map +1 -0
- package/dist/src/executors/openai-api.js +40 -0
- package/dist/src/executors/openai-api.js.map +1 -0
- package/dist/src/executors/openai-cli.d.ts +3 -0
- package/dist/src/executors/openai-cli.d.ts.map +1 -0
- package/dist/src/executors/openai-cli.js +60 -0
- package/dist/src/executors/openai-cli.js.map +1 -0
- package/dist/src/executors/script.d.ts +3 -0
- package/dist/src/executors/script.d.ts.map +1 -0
- package/dist/src/executors/script.js +63 -0
- package/dist/src/executors/script.js.map +1 -0
- package/dist/src/executors/shared.d.ts +117 -0
- package/dist/src/executors/shared.d.ts.map +1 -0
- package/dist/src/executors/shared.js +49 -0
- package/dist/src/executors/shared.js.map +1 -0
- package/dist/src/grading/assertions.d.ts +18 -0
- package/dist/src/grading/assertions.d.ts.map +1 -0
- package/dist/src/grading/assertions.js +239 -0
- package/dist/src/grading/assertions.js.map +1 -0
- package/dist/src/grading/index.d.ts +26 -0
- package/dist/src/grading/index.d.ts.map +1 -0
- package/dist/src/grading/index.js +98 -0
- package/dist/src/grading/index.js.map +1 -0
- package/dist/src/grading/judge.d.ts +13 -0
- package/dist/src/grading/judge.d.ts.map +1 -0
- package/dist/src/grading/judge.js +98 -0
- package/dist/src/grading/judge.js.map +1 -0
- package/dist/src/grading/layered-scores.d.ts +13 -0
- package/dist/src/grading/layered-scores.d.ts.map +1 -0
- package/dist/src/grading/layered-scores.js +62 -0
- package/dist/src/grading/layered-scores.js.map +1 -0
- package/dist/src/inputs/eval-config.d.ts +13 -0
- package/dist/src/inputs/eval-config.d.ts.map +1 -0
- package/dist/src/inputs/eval-config.js +136 -0
- package/dist/src/inputs/eval-config.js.map +1 -0
- package/dist/src/inputs/load-samples.d.ts +16 -0
- package/dist/src/inputs/load-samples.d.ts.map +1 -0
- package/dist/src/inputs/load-samples.js +51 -0
- package/dist/src/inputs/load-samples.js.map +1 -0
- package/dist/src/inputs/mcp-resolver.d.ts +50 -0
- package/dist/src/inputs/mcp-resolver.d.ts.map +1 -0
- package/dist/src/inputs/mcp-resolver.js +307 -0
- package/dist/src/inputs/mcp-resolver.js.map +1 -0
- package/dist/src/inputs/skill-loader.d.ts +18 -0
- package/dist/src/inputs/skill-loader.d.ts.map +1 -0
- package/dist/src/inputs/skill-loader.js +222 -0
- package/dist/src/inputs/skill-loader.js.map +1 -0
- package/dist/src/inputs/url-fetcher.d.ts +19 -0
- package/dist/src/inputs/url-fetcher.d.ts.map +1 -0
- package/dist/src/inputs/url-fetcher.js +241 -0
- package/dist/src/inputs/url-fetcher.js.map +1 -0
- package/dist/src/observability/production-analyzer.d.ts +61 -0
- package/dist/src/observability/production-analyzer.d.ts.map +1 -0
- package/dist/src/observability/production-analyzer.js +183 -0
- package/dist/src/observability/production-analyzer.js.map +1 -0
- package/dist/src/observability/trace-adapter.d.ts +75 -0
- package/dist/src/observability/trace-adapter.d.ts.map +1 -0
- package/dist/src/observability/trace-adapter.js +341 -0
- package/dist/src/observability/trace-adapter.js.map +1 -0
- package/dist/src/renderer/html-renderer.d.ts +9 -0
- package/dist/src/renderer/html-renderer.d.ts.map +1 -0
- package/dist/src/renderer/html-renderer.js +338 -0
- package/dist/src/renderer/html-renderer.js.map +1 -0
- package/dist/src/renderer/layout.d.ts +13 -0
- package/dist/src/renderer/layout.d.ts.map +1 -0
- package/dist/src/renderer/layout.js +476 -0
- package/dist/src/renderer/layout.js.map +1 -0
- package/dist/src/renderer/skill-health-renderer.d.ts +20 -0
- package/dist/src/renderer/skill-health-renderer.d.ts.map +1 -0
- package/dist/src/renderer/skill-health-renderer.js +218 -0
- package/dist/src/renderer/skill-health-renderer.js.map +1 -0
- package/dist/src/renderer/summary.d.ts +21 -0
- package/dist/src/renderer/summary.d.ts.map +1 -0
- package/dist/src/renderer/summary.js +1072 -0
- package/dist/src/renderer/summary.js.map +1 -0
- package/dist/src/renderer/table.d.ts +3 -0
- package/dist/src/renderer/table.d.ts.map +1 -0
- package/dist/src/renderer/table.js +134 -0
- package/dist/src/renderer/table.js.map +1 -0
- package/dist/src/renderer/trends.d.ts +6 -0
- package/dist/src/renderer/trends.d.ts.map +1 -0
- package/dist/src/renderer/trends.js +127 -0
- package/dist/src/renderer/trends.js.map +1 -0
- package/dist/src/server/job-store.d.ts +4 -0
- package/dist/src/server/job-store.d.ts.map +1 -0
- package/dist/src/server/job-store.js +68 -0
- package/dist/src/server/job-store.js.map +1 -0
- package/dist/src/server/report-server.d.ts +16 -0
- package/dist/src/server/report-server.d.ts.map +1 -0
- package/dist/src/server/report-server.js +196 -0
- package/dist/src/server/report-server.js.map +1 -0
- package/dist/src/server/report-store.d.ts +44 -0
- package/dist/src/server/report-store.d.ts.map +1 -0
- package/dist/src/server/report-store.js +190 -0
- package/dist/src/server/report-store.js.map +1 -0
- package/dist/src/types.d.ts +553 -0
- package/dist/src/types.d.ts.map +1 -0
- package/dist/src/types.js +2 -0
- package/dist/src/types.js.map +1 -0
- package/package.json +44 -3
package/dist/src/cli.js
ADDED
|
@@ -0,0 +1,1004 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
import { parseArgs } from 'node:util';
|
|
3
|
+
import { resolve } from 'node:path';
|
|
4
|
+
import { homedir } from 'node:os';
|
|
5
|
+
import { join } from 'node:path';
|
|
6
|
+
import { existsSync } from 'node:fs';
|
|
7
|
+
import { discoverVariants, parseVariantCwd } from './inputs/skill-loader.js';
|
|
8
|
+
import { loadEvalConfig, configVariantsToSpecs } from './inputs/eval-config.js';
|
|
9
|
+
// ---------------------------------------------------------------------------
|
|
10
|
+
// Constants
|
|
11
|
+
// ---------------------------------------------------------------------------
|
|
12
|
+
const DEFAULT_REPORTS_DIR = join(homedir(), '.oh-my-knowledge', 'reports');
|
|
13
|
+
// Shared CLI options for run/ci commands.
|
|
14
|
+
// Defaults are applied inside parseRunConfig (after config-file merge) so that
|
|
15
|
+
// CLI `undefined` can be reliably distinguished from "user passed the default value".
|
|
16
|
+
// Priority order resolved in parseRunConfig: CLI arg > --config file > hard-coded default.
|
|
17
|
+
const RUN_OPTIONS = {
|
|
18
|
+
samples: { type: 'string' },
|
|
19
|
+
'skill-dir': { type: 'string' },
|
|
20
|
+
control: { type: 'string' },
|
|
21
|
+
treatment: { type: 'string' },
|
|
22
|
+
config: { type: 'string' },
|
|
23
|
+
model: { type: 'string' },
|
|
24
|
+
'judge-model': { type: 'string' },
|
|
25
|
+
'output-dir': { type: 'string' },
|
|
26
|
+
'no-judge': { type: 'boolean' },
|
|
27
|
+
'no-cache': { type: 'boolean' },
|
|
28
|
+
'dry-run': { type: 'boolean' },
|
|
29
|
+
concurrency: { type: 'string' },
|
|
30
|
+
timeout: { type: 'string' },
|
|
31
|
+
executor: { type: 'string' },
|
|
32
|
+
'judge-executor': { type: 'string' },
|
|
33
|
+
each: { type: 'boolean' },
|
|
34
|
+
'skip-preflight': { type: 'boolean' },
|
|
35
|
+
'mcp-config': { type: 'string' },
|
|
36
|
+
'no-serve': { type: 'boolean' },
|
|
37
|
+
verbose: { type: 'boolean' },
|
|
38
|
+
retry: { type: 'string' },
|
|
39
|
+
resume: { type: 'string' },
|
|
40
|
+
'layered-stats': { type: 'boolean' },
|
|
41
|
+
};
|
|
42
|
+
// ---------------------------------------------------------------------------
|
|
43
|
+
// parseRunConfig
|
|
44
|
+
// ---------------------------------------------------------------------------
|
|
45
|
+
function parseRunConfig(argv, extraOptions = {}) {
|
|
46
|
+
const { values } = parseArgs({
|
|
47
|
+
args: argv,
|
|
48
|
+
options: { ...RUN_OPTIONS, ...extraOptions },
|
|
49
|
+
strict: false,
|
|
50
|
+
});
|
|
51
|
+
if (values.variants !== undefined) {
|
|
52
|
+
throw new Error(`--variants 已在 v0.16 废除,请改用 --control <expr> 与 --treatment <v1,v2,...>\n`
|
|
53
|
+
+ ` 迁移示例:--variants baseline,my-skill → --control baseline --treatment my-skill\n`
|
|
54
|
+
+ ` 复杂场景可用 --config eval.yaml(参见 docs/terminology-spec.md)`);
|
|
55
|
+
}
|
|
56
|
+
// 1) Load --config (if provided). All subsequent fields fall back to it when CLI is silent.
|
|
57
|
+
const evalConfig = values.config
|
|
58
|
+
? loadEvalConfig(values.config)
|
|
59
|
+
: null;
|
|
60
|
+
// 2) Resolve samples path: CLI > config > auto-detect .json/.yaml/.yml in cwd.
|
|
61
|
+
const cliSamples = values.samples;
|
|
62
|
+
let samplesFile;
|
|
63
|
+
if (cliSamples) {
|
|
64
|
+
samplesFile = cliSamples;
|
|
65
|
+
}
|
|
66
|
+
else if (evalConfig?.samples) {
|
|
67
|
+
samplesFile = evalConfig.samples; // already resolved against config file dir
|
|
68
|
+
}
|
|
69
|
+
else {
|
|
70
|
+
samplesFile = 'eval-samples.json';
|
|
71
|
+
if (!existsSync(resolve(samplesFile))) {
|
|
72
|
+
if (existsSync(resolve('eval-samples.yaml')))
|
|
73
|
+
samplesFile = 'eval-samples.yaml';
|
|
74
|
+
else if (existsSync(resolve('eval-samples.yml')))
|
|
75
|
+
samplesFile = 'eval-samples.yml';
|
|
76
|
+
}
|
|
77
|
+
}
|
|
78
|
+
const skillDir = resolve(values['skill-dir'] ?? 'skills');
|
|
79
|
+
// 3) Resolve variantSpecs: CLI > config. If neither, error with a helpful hint.
|
|
80
|
+
const controlExpr = values.control;
|
|
81
|
+
const treatmentExprs = values.treatment
|
|
82
|
+
? values.treatment.split(',').map((v) => v.trim()).filter(Boolean)
|
|
83
|
+
: [];
|
|
84
|
+
let variantSpecs;
|
|
85
|
+
if (controlExpr || treatmentExprs.length > 0) {
|
|
86
|
+
// CLI roles present → CLI entirely replaces config.variants (no merging).
|
|
87
|
+
variantSpecs = [];
|
|
88
|
+
if (controlExpr) {
|
|
89
|
+
variantSpecs.push({ name: parseVariantCwd(controlExpr).name, role: 'control', expr: controlExpr });
|
|
90
|
+
}
|
|
91
|
+
for (const expr of treatmentExprs) {
|
|
92
|
+
variantSpecs.push({ name: parseVariantCwd(expr).name, role: 'treatment', expr });
|
|
93
|
+
}
|
|
94
|
+
}
|
|
95
|
+
else if (evalConfig) {
|
|
96
|
+
variantSpecs = configVariantsToSpecs(evalConfig.variants);
|
|
97
|
+
}
|
|
98
|
+
else {
|
|
99
|
+
const discovered = discoverVariants(skillDir);
|
|
100
|
+
const hint = discovered.length > 0 ? `\n skill-dir (${skillDir}) 下发现的候选:${discovered.join(', ')}` : '';
|
|
101
|
+
throw new Error(`请通过 --control / --treatment 或 --config eval.yaml 声明 variant 角色。\n`
|
|
102
|
+
+ ` 示例:omk bench run --control baseline --treatment my-skill${hint}\n`
|
|
103
|
+
+ ` 术语见 docs/terminology-spec.md(v0.16 起废除 --variants,改用 experiment role 显式声明)`);
|
|
104
|
+
}
|
|
105
|
+
const seenNames = new Set();
|
|
106
|
+
for (const spec of variantSpecs) {
|
|
107
|
+
if (seenNames.has(spec.name)) {
|
|
108
|
+
throw new Error(`variant "${spec.name}" 重复出现——同一 variant 不能同时属于 --control 与 --treatment,也不能在 --treatment 中重复。`);
|
|
109
|
+
}
|
|
110
|
+
seenNames.add(spec.name);
|
|
111
|
+
}
|
|
112
|
+
// 4) Apply CLI > config > hard-coded default for all other fields.
|
|
113
|
+
const executorName = values.executor ?? evalConfig?.executor ?? 'claude';
|
|
114
|
+
const judgeExecutorName = values['judge-executor'] ?? evalConfig?.judgeExecutor ?? executorName;
|
|
115
|
+
const model = values.model ?? evalConfig?.model ?? 'sonnet';
|
|
116
|
+
const judgeModelRaw = values['judge-model'] !== undefined
|
|
117
|
+
? values['judge-model']
|
|
118
|
+
: evalConfig?.judgeModel ?? 'haiku';
|
|
119
|
+
const judgeModel = judgeModelRaw ?? 'haiku';
|
|
120
|
+
const outputDir = resolve(values['output-dir'] ?? DEFAULT_REPORTS_DIR);
|
|
121
|
+
const concurrencyRaw = values.concurrency !== undefined
|
|
122
|
+
? Number(values.concurrency)
|
|
123
|
+
: evalConfig?.concurrency ?? 1;
|
|
124
|
+
const concurrency = Math.max(1, Number(concurrencyRaw) || 1);
|
|
125
|
+
const timeoutSec = values.timeout !== undefined
|
|
126
|
+
? Number(values.timeout)
|
|
127
|
+
: evalConfig?.timeoutMs
|
|
128
|
+
? evalConfig.timeoutMs / 1000
|
|
129
|
+
: 120;
|
|
130
|
+
const timeoutMs = Math.max(1, Number(timeoutSec) || 120) * 1000;
|
|
131
|
+
const noJudge = values['no-judge'] ?? false;
|
|
132
|
+
const noCache = values['no-cache'] ?? evalConfig?.noCache ?? false;
|
|
133
|
+
const dryRun = values['dry-run'] ?? false;
|
|
134
|
+
const skipPreflight = values['skip-preflight'] ?? false;
|
|
135
|
+
const mcpConfig = values['mcp-config'] ?? evalConfig?.mcpConfig;
|
|
136
|
+
const verbose = values.verbose ?? false;
|
|
137
|
+
const retry = Math.max(0, Number(values.retry ?? 0) || 0);
|
|
138
|
+
const resume = values.resume;
|
|
139
|
+
const blind = values.blind ?? evalConfig?.blind ?? false;
|
|
140
|
+
const layeredStats = values['layered-stats'] ?? false;
|
|
141
|
+
return {
|
|
142
|
+
values,
|
|
143
|
+
config: {
|
|
144
|
+
samplesPath: resolve(samplesFile),
|
|
145
|
+
skillDir,
|
|
146
|
+
variantSpecs,
|
|
147
|
+
model,
|
|
148
|
+
judgeModel,
|
|
149
|
+
outputDir,
|
|
150
|
+
noJudge,
|
|
151
|
+
noCache,
|
|
152
|
+
dryRun,
|
|
153
|
+
concurrency,
|
|
154
|
+
timeoutMs,
|
|
155
|
+
executorName,
|
|
156
|
+
judgeExecutorName,
|
|
157
|
+
skipPreflight,
|
|
158
|
+
mcpConfig,
|
|
159
|
+
verbose,
|
|
160
|
+
retry,
|
|
161
|
+
resume,
|
|
162
|
+
blind,
|
|
163
|
+
layeredStats,
|
|
164
|
+
},
|
|
165
|
+
};
|
|
166
|
+
}
|
|
167
|
+
// ---------------------------------------------------------------------------
|
|
168
|
+
// Help text
|
|
169
|
+
// ---------------------------------------------------------------------------
|
|
170
|
+
const HELP = `
|
|
171
|
+
oh-my-knowledge — Knowledge artifact evaluation toolkit
|
|
172
|
+
|
|
173
|
+
Usage:
|
|
174
|
+
omk bench run [options] Run an evaluation
|
|
175
|
+
omk bench report [options] Start the report server
|
|
176
|
+
omk bench ci [options] Run evaluation and exit with pass/fail code
|
|
177
|
+
omk bench init [dir] Scaffold a new eval project
|
|
178
|
+
omk bench gen-samples [skill] Generate eval-samples from skill content
|
|
179
|
+
omk bench diff <id1> <id2> Compare two evaluation reports
|
|
180
|
+
omk bench evolve <skill> Self-improve a skill through iterative evaluation
|
|
181
|
+
|
|
182
|
+
omk analyze <dir> Analyze cc session trace(s), produce skill 健康度日报 (v0.18)
|
|
183
|
+
|
|
184
|
+
Options for "bench run":
|
|
185
|
+
|
|
186
|
+
--samples <path> Sample file (default: eval-samples.json)
|
|
187
|
+
--skill-dir <path> Skill definitions directory (default: skills)
|
|
188
|
+
--control <expr> Control-group variant expression (experiment role = control)
|
|
189
|
+
--treatment <v1,v2> Treatment-group variant expressions (comma-separated; role = treatment)
|
|
190
|
+
Each variant expression resolves to an artifact and optional runtime context:
|
|
191
|
+
"baseline" — bare model, no artifact injected
|
|
192
|
+
"git:name" — artifact from last commit
|
|
193
|
+
"git:ref:name" — artifact from specific commit
|
|
194
|
+
path with "/" — artifact from file directly (e.g. ./v1.md)
|
|
195
|
+
"name@/cwd" — attach runtime context / cwd
|
|
196
|
+
At least one of --control / --treatment must be provided.
|
|
197
|
+
--config <path> YAML/JSON config file (evaluation-as-code).
|
|
198
|
+
Declares samples + variants + model + executor in one file.
|
|
199
|
+
CLI flags override config fields when both are provided.
|
|
200
|
+
Relative paths inside the config are resolved against its directory.
|
|
201
|
+
--model <name> Model under test (default: sonnet)
|
|
202
|
+
--judge-model <name> Judge model (default: haiku)
|
|
203
|
+
--output-dir <path> Report output directory (default: ~/.oh-my-knowledge/reports/)
|
|
204
|
+
--no-judge Skip LLM judging
|
|
205
|
+
--no-cache Disable result caching
|
|
206
|
+
--dry-run Preview tasks without executing
|
|
207
|
+
--blind Blind A/B mode: hide variant names in report
|
|
208
|
+
--concurrency <n> Number of parallel tasks (default: 1)
|
|
209
|
+
--timeout <seconds> Executor timeout per task in seconds (default: 120)
|
|
210
|
+
--repeat <n> Run evaluation N times for variance analysis (default: 1)
|
|
211
|
+
--retry <n> Retry failed tasks up to N times with exponential backoff (default: 0)
|
|
212
|
+
--resume <report-id> Resume from a previous report, skipping completed tasks
|
|
213
|
+
--executor <name> Executor: claude, openai, gemini, anthropic-api, openai-api,
|
|
214
|
+
or any shell command (e.g. "python my_provider.py")
|
|
215
|
+
--judge-executor <name> Executor for LLM judge (default: same as --executor)
|
|
216
|
+
--each Evaluate each skill independently against baseline
|
|
217
|
+
Requires {name}.eval-samples.json paired with each skill
|
|
218
|
+
--skip-preflight Skip model connectivity check before evaluation
|
|
219
|
+
--mcp-config <path> MCP config file for URL fetching via MCP servers
|
|
220
|
+
(default: .mcp.json in current directory)
|
|
221
|
+
--no-serve Skip auto-starting report server after evaluation
|
|
222
|
+
--verbose Print detailed progress for each sample (exec result, grading phases)
|
|
223
|
+
--layered-stats Expand the three-layer (fact/behavior/judge) independent
|
|
224
|
+
significance breakdown in the HTML report by default.
|
|
225
|
+
Without this flag, the breakdown is collapsed behind a
|
|
226
|
+
click-to-expand summary under each comparison.
|
|
227
|
+
|
|
228
|
+
Options for "bench ci":
|
|
229
|
+
(same as "bench run", plus:)
|
|
230
|
+
--threshold <number> Minimum score to pass, applied INDEPENDENTLY to each of
|
|
231
|
+
the three layers (fact / behavior / LLM judge). ANY
|
|
232
|
+
layer below threshold fails the gate — this prevents
|
|
233
|
+
composite averaging from masking a single-layer collapse.
|
|
234
|
+
Default: 3.5. If all three layers are absent (no
|
|
235
|
+
assertions and no rubric defined in eval-samples), the
|
|
236
|
+
gate FAILS with a configuration hint — no composite fallback.
|
|
237
|
+
|
|
238
|
+
Options for "bench report":
|
|
239
|
+
--port <number> Server port (default: 7799)
|
|
240
|
+
--reports-dir <path> Reports directory (default: ~/.oh-my-knowledge/reports/)
|
|
241
|
+
--export <id> Export report as standalone HTML file
|
|
242
|
+
--dev Dev mode: auto-restart on lib/ file changes
|
|
243
|
+
|
|
244
|
+
Options for "bench gen-samples":
|
|
245
|
+
--each Generate for all skills missing eval-samples
|
|
246
|
+
--count <n> Number of samples to generate per skill (default: 5)
|
|
247
|
+
--model <name> Model for generation (default: sonnet)
|
|
248
|
+
--skill-dir <path> Skill directory (default: skills), used with --each
|
|
249
|
+
|
|
250
|
+
Options for "analyze":
|
|
251
|
+
<dir> Input: cc session JSONL file / dir (e.g. ~/.claude/projects/<slug>)
|
|
252
|
+
--kb <path> Knowledge base root (default: auto-infer from trace cwd)
|
|
253
|
+
--last <duration> Time window like "7d" / "30d" (default: all)
|
|
254
|
+
--from <iso> Window start (ISO8601), takes precedence over --last
|
|
255
|
+
--to <iso> Window end (ISO8601), takes precedence over --last
|
|
256
|
+
--skills <n1,n2,...> Whitelist skills to analyze (default: all)
|
|
257
|
+
--output-dir <path> Output dir (default: ~/.oh-my-knowledge/analyses/)
|
|
258
|
+
|
|
259
|
+
Options for "bench evolve":
|
|
260
|
+
--rounds <n> Maximum evolution rounds (default: 5)
|
|
261
|
+
--target <score> Stop early when score reaches this threshold
|
|
262
|
+
--samples <path> Sample file (default: eval-samples.json)
|
|
263
|
+
--model <name> Model under test (default: sonnet)
|
|
264
|
+
--judge-model <name> Judge model (default: haiku)
|
|
265
|
+
--improve-model <name> Model for generating improvements (default: sonnet)
|
|
266
|
+
--concurrency <n> Parallel eval tasks (default: 1)
|
|
267
|
+
--timeout <seconds> Executor timeout per task in seconds (default: 120)
|
|
268
|
+
--executor <name> Executor to use (default: claude)
|
|
269
|
+
|
|
270
|
+
Examples:
|
|
271
|
+
omk bench run --control v1 --treatment v2
|
|
272
|
+
omk bench run --control baseline --treatment my-skill
|
|
273
|
+
omk bench run --control git:my-skill --treatment my-skill
|
|
274
|
+
omk bench run --control ./old-skill.md --treatment ./new-skill.md
|
|
275
|
+
omk bench run --control baseline --treatment v1,v2,v3
|
|
276
|
+
omk bench run --config eval.yaml
|
|
277
|
+
omk bench run --config eval.yaml --model sonnet-4.6 # CLI overrides config
|
|
278
|
+
omk bench run --each
|
|
279
|
+
omk bench run --dry-run
|
|
280
|
+
omk bench report --port 8080
|
|
281
|
+
omk bench report --export v1-vs-v2-20260326-1832
|
|
282
|
+
omk bench init my-eval
|
|
283
|
+
omk bench gen-samples skills/my-skill.md
|
|
284
|
+
omk bench gen-samples --each
|
|
285
|
+
omk bench diff <report-id-1> <report-id-2>
|
|
286
|
+
omk bench evolve skills/my-skill.md --rounds 5
|
|
287
|
+
omk analyze ~/.claude/projects/-Users-lizhiyao-Documents-oh-my-knowledge
|
|
288
|
+
omk analyze ~/.claude/projects/my-project --last 7d --kb /path/to/project
|
|
289
|
+
omk analyze ~/.claude/projects/my-project --skills audit,polish
|
|
290
|
+
`.trim();
|
|
291
|
+
// ---------------------------------------------------------------------------
|
|
292
|
+
// Update check
|
|
293
|
+
// ---------------------------------------------------------------------------
|
|
294
|
+
async function checkUpdate() {
|
|
295
|
+
try {
|
|
296
|
+
const { readFileSync } = await import('node:fs');
|
|
297
|
+
const { fileURLToPath } = await import('node:url');
|
|
298
|
+
const { dirname, join } = await import('node:path');
|
|
299
|
+
const __dirname = dirname(fileURLToPath(import.meta.url));
|
|
300
|
+
const pkg = JSON.parse(readFileSync(join(__dirname, 'package.json'), 'utf-8'));
|
|
301
|
+
const registry = pkg.publishConfig?.registry || 'https://registry.npmjs.org';
|
|
302
|
+
const res = await fetch(`${registry}/${pkg.name}/latest`, { signal: AbortSignal.timeout(3000) });
|
|
303
|
+
if (!res.ok)
|
|
304
|
+
return;
|
|
305
|
+
const data = await res.json();
|
|
306
|
+
if (data.version && data.version !== pkg.version) {
|
|
307
|
+
process.stderr.write(`\n💡 新版本可用: ${pkg.version} → ${data.version},运行 npm update ${pkg.name} -g 更新\n\n`);
|
|
308
|
+
}
|
|
309
|
+
}
|
|
310
|
+
catch { /* 静默失败,不影响正常使用 */ }
|
|
311
|
+
}
|
|
312
|
+
// ---------------------------------------------------------------------------
|
|
313
|
+
// Main
|
|
314
|
+
// ---------------------------------------------------------------------------
|
|
315
|
+
async function main() {
|
|
316
|
+
checkUpdate();
|
|
317
|
+
const [domain, command, ...rest] = process.argv.slice(2);
|
|
318
|
+
if (!domain || domain === '--help' || domain === '-h') {
|
|
319
|
+
console.log(HELP);
|
|
320
|
+
process.exit(0);
|
|
321
|
+
}
|
|
322
|
+
if (domain === 'analyze') {
|
|
323
|
+
const args = command ? [command, ...rest] : [];
|
|
324
|
+
await handleAnalyze(args);
|
|
325
|
+
return;
|
|
326
|
+
}
|
|
327
|
+
if (domain !== 'bench') {
|
|
328
|
+
console.error(`Unknown domain: ${domain}. Use "omk bench <command>" or "omk analyze <dir>".`);
|
|
329
|
+
process.exit(1);
|
|
330
|
+
}
|
|
331
|
+
if (!command || command === '--help' || command === '-h') {
|
|
332
|
+
console.log(HELP);
|
|
333
|
+
process.exit(0);
|
|
334
|
+
}
|
|
335
|
+
switch (command) {
|
|
336
|
+
case 'run':
|
|
337
|
+
await handleRun(rest);
|
|
338
|
+
break;
|
|
339
|
+
case 'report':
|
|
340
|
+
await handleReport(rest);
|
|
341
|
+
break;
|
|
342
|
+
case 'init':
|
|
343
|
+
await handleInit(rest);
|
|
344
|
+
break;
|
|
345
|
+
case 'ci':
|
|
346
|
+
await handleCi(rest);
|
|
347
|
+
break;
|
|
348
|
+
case 'gen-samples':
|
|
349
|
+
await handleGenSamples(rest);
|
|
350
|
+
break;
|
|
351
|
+
case 'evolve':
|
|
352
|
+
await handleEvolve(rest);
|
|
353
|
+
break;
|
|
354
|
+
case 'diff':
|
|
355
|
+
await handleDiff(rest);
|
|
356
|
+
break;
|
|
357
|
+
default:
|
|
358
|
+
console.error(`Unknown command: bench ${command}. Use "run", "report", "ci", "init", "gen-samples", or "evolve".`);
|
|
359
|
+
process.exit(1);
|
|
360
|
+
}
|
|
361
|
+
}
|
|
362
|
+
// ---------------------------------------------------------------------------
|
|
363
|
+
// Progress callback
|
|
364
|
+
// ---------------------------------------------------------------------------
|
|
365
|
+
function defaultOnProgress({ phase, completed, total, sample_id, variant, durationMs, inputTokens, outputTokens, costUSD, score, outputPreview, judgePhase: _judgePhase, judgeDim, skipped, attempt, maxAttempts, error, }) {
|
|
366
|
+
if (phase === 'preflight') {
|
|
367
|
+
process.stderr.write('⏳ 预检模型连通性...\n');
|
|
368
|
+
return;
|
|
369
|
+
}
|
|
370
|
+
if (phase === 'retry') {
|
|
371
|
+
process.stderr.write(`[${completed}/${total}] ${sample_id}/${variant} 🔄 重试 ${attempt}/${maxAttempts}...\n`);
|
|
372
|
+
return;
|
|
373
|
+
}
|
|
374
|
+
if (phase === 'error') {
|
|
375
|
+
process.stderr.write(`[${completed}/${total}] ${sample_id}/${variant} ❌ ${error}\n`);
|
|
376
|
+
return;
|
|
377
|
+
}
|
|
378
|
+
if (phase === 'start') {
|
|
379
|
+
process.stderr.write(`[${completed}/${total}] ${sample_id}/${variant} ⏳ 执行中...\n`);
|
|
380
|
+
}
|
|
381
|
+
else if (phase === 'exec_done') {
|
|
382
|
+
const costInfo = costUSD != null && costUSD > 0 ? ` $${costUSD.toFixed(4)}` : '';
|
|
383
|
+
process.stderr.write(`[${completed}/${total}] ${sample_id}/${variant} 执行完成 ${durationMs}ms ${inputTokens}+${outputTokens} tokens${costInfo}\n`);
|
|
384
|
+
if (outputPreview) {
|
|
385
|
+
process.stderr.write(` 输出预览: ${outputPreview.slice(0, 150).replace(/\n/g, ' ')}\n`);
|
|
386
|
+
}
|
|
387
|
+
}
|
|
388
|
+
else if (phase === 'grading') {
|
|
389
|
+
const dimInfo = judgeDim ? ` [${judgeDim}]` : '';
|
|
390
|
+
process.stderr.write(`[${completed}/${total}] ${sample_id}/${variant} 评审中${dimInfo}...\n`);
|
|
391
|
+
}
|
|
392
|
+
else if (phase === 'judge_done') {
|
|
393
|
+
const dimInfo = judgeDim ? ` [${judgeDim}]` : '';
|
|
394
|
+
process.stderr.write(`[${completed}/${total}] ${sample_id}/${variant} 评审完成${dimInfo} score=${score}\n`);
|
|
395
|
+
}
|
|
396
|
+
else if (phase === 'done' && skipped) {
|
|
397
|
+
if (sample_id)
|
|
398
|
+
process.stderr.write(`[${completed}/${total}] ${sample_id}/${variant} ⏭ 已跳过(已有结果)\n`);
|
|
399
|
+
}
|
|
400
|
+
else {
|
|
401
|
+
const costInfo = costUSD != null && costUSD > 0 ? ` $${costUSD.toFixed(4)}` : '';
|
|
402
|
+
const scoreInfo = typeof score === 'number' ? ` score=${score}` : '';
|
|
403
|
+
process.stderr.write(`[${completed}/${total}] ${sample_id}/${variant} ✓ ${durationMs}ms ${inputTokens}+${outputTokens} tokens${costInfo}${scoreInfo}\n`);
|
|
404
|
+
}
|
|
405
|
+
}
|
|
406
|
+
// ---------------------------------------------------------------------------
|
|
407
|
+
// handleRun
|
|
408
|
+
// ---------------------------------------------------------------------------
|
|
409
|
+
async function handleRun(argv) {
|
|
410
|
+
const { values, config } = parseRunConfig(argv, {
|
|
411
|
+
blind: { type: 'boolean', default: false },
|
|
412
|
+
repeat: { type: 'string', default: '1' },
|
|
413
|
+
});
|
|
414
|
+
const { runEvaluation, runMultiple, runEachEvaluation } = await import('./eval-workflows/run-evaluation.js');
|
|
415
|
+
config.blind = values.blind;
|
|
416
|
+
config.onProgress = defaultOnProgress;
|
|
417
|
+
try {
|
|
418
|
+
// --each mode: evaluate each skill independently
|
|
419
|
+
if (values.each) {
|
|
420
|
+
const { report, filePath } = await runEachEvaluation({
|
|
421
|
+
...config,
|
|
422
|
+
onSkillProgress({ phase, skill, current, total }) {
|
|
423
|
+
if (phase === 'start') {
|
|
424
|
+
process.stderr.write(`\n=== [${current}/${total}] Skill: ${skill} ===\n`);
|
|
425
|
+
}
|
|
426
|
+
},
|
|
427
|
+
});
|
|
428
|
+
console.log(JSON.stringify(report, null, 2));
|
|
429
|
+
if (filePath) {
|
|
430
|
+
process.stderr.write('\n✅ 批量评测完成\n');
|
|
431
|
+
process.stderr.write(`📄 Report saved to: ${filePath}\n`);
|
|
432
|
+
if (!values['no-serve'] && process.stdout.isTTY) {
|
|
433
|
+
const { createReportServer } = await import('./server/report-server.js');
|
|
434
|
+
const server = createReportServer({ reportsDir: config.outputDir });
|
|
435
|
+
const serverUrl = await server.start();
|
|
436
|
+
const reportUrl = `${serverUrl}/run/${report.id}`;
|
|
437
|
+
process.stderr.write(`\n📊 Report server running at ${serverUrl}\n`);
|
|
438
|
+
process.stderr.write(`👉 View report: ${reportUrl}\n`);
|
|
439
|
+
process.stderr.write('\nPress Ctrl+C to stop the server\n');
|
|
440
|
+
const { platform } = await import('node:os');
|
|
441
|
+
const openCmd = platform() === 'darwin' ? 'open' : platform() === 'win32' ? 'start' : 'xdg-open';
|
|
442
|
+
const { execFile: execFileCb } = await import('node:child_process');
|
|
443
|
+
execFileCb(openCmd, [reportUrl], () => { });
|
|
444
|
+
}
|
|
445
|
+
else if (!values['no-serve']) {
|
|
446
|
+
process.stderr.write('\n💡 非交互环境,已跳过 report server\n');
|
|
447
|
+
process.stderr.write(` 查看报告: omk bench report --reports-dir ${config.outputDir}\n`);
|
|
448
|
+
}
|
|
449
|
+
}
|
|
450
|
+
return;
|
|
451
|
+
}
|
|
452
|
+
// --repeat 诚实输入校验:非 ≥1 整数时提示并钳到 1,不静默掩盖用户错字/极端输入
|
|
453
|
+
const repeatRaw = values.repeat;
|
|
454
|
+
const parsedRepeat = repeatRaw !== undefined ? Number(repeatRaw) : 1;
|
|
455
|
+
if (repeatRaw !== undefined && (!Number.isFinite(parsedRepeat) || parsedRepeat < 1)) {
|
|
456
|
+
process.stderr.write(`⚠ --repeat "${repeatRaw}" 无效(期望 ≥ 1 的整数),已按 1 次评测执行\n`);
|
|
457
|
+
}
|
|
458
|
+
const repeatCount = Math.max(1, Math.floor(parsedRepeat) || 1);
|
|
459
|
+
let report;
|
|
460
|
+
let filePath;
|
|
461
|
+
if (repeatCount > 1) {
|
|
462
|
+
const result = await runMultiple({
|
|
463
|
+
...config,
|
|
464
|
+
repeat: repeatCount,
|
|
465
|
+
onRepeatProgress({ run, total }) {
|
|
466
|
+
process.stderr.write(`\n=== Run ${run}/${total} ===\n`);
|
|
467
|
+
},
|
|
468
|
+
});
|
|
469
|
+
report = result.report;
|
|
470
|
+
filePath = null;
|
|
471
|
+
}
|
|
472
|
+
else {
|
|
473
|
+
const result = (await runEvaluation(config));
|
|
474
|
+
report = result.report;
|
|
475
|
+
filePath = result.filePath;
|
|
476
|
+
}
|
|
477
|
+
console.log(JSON.stringify(report, null, 2));
|
|
478
|
+
if (filePath) {
|
|
479
|
+
process.stderr.write('\n✅ 评测完成\n');
|
|
480
|
+
process.stderr.write(`📄 Report saved to: ${filePath}\n`);
|
|
481
|
+
if (!values['no-serve'] && process.stdout.isTTY) {
|
|
482
|
+
// Auto-start report server
|
|
483
|
+
const { createReportServer } = await import('./server/report-server.js');
|
|
484
|
+
const server = createReportServer({
|
|
485
|
+
reportsDir: config.outputDir,
|
|
486
|
+
});
|
|
487
|
+
const serverUrl = await server.start();
|
|
488
|
+
const reportUrl = `${serverUrl}/run/${report.id}`;
|
|
489
|
+
process.stderr.write(`\n📊 Report server running at ${serverUrl}\n`);
|
|
490
|
+
process.stderr.write(`👉 View report: ${reportUrl}\n`);
|
|
491
|
+
process.stderr.write('\nPress Ctrl+C to stop the server\n');
|
|
492
|
+
// Auto-open report in browser
|
|
493
|
+
const { platform } = await import('node:os');
|
|
494
|
+
const openCmd = platform() === 'darwin' ? 'open' : platform() === 'win32' ? 'start' : 'xdg-open';
|
|
495
|
+
const { execFile: execFileCb } = await import('node:child_process');
|
|
496
|
+
execFileCb(openCmd, [reportUrl], () => { });
|
|
497
|
+
}
|
|
498
|
+
else if (!values['no-serve']) {
|
|
499
|
+
process.stderr.write('\n💡 非交互环境,已跳过 report server\n');
|
|
500
|
+
process.stderr.write(` 查看报告: omk bench report --reports-dir ${config.outputDir}\n`);
|
|
501
|
+
}
|
|
502
|
+
}
|
|
503
|
+
}
|
|
504
|
+
catch (err) {
|
|
505
|
+
console.error(`Error: ${err.message}`);
|
|
506
|
+
process.exit(1);
|
|
507
|
+
}
|
|
508
|
+
}
|
|
509
|
+
// ---------------------------------------------------------------------------
|
|
510
|
+
// handleReport
|
|
511
|
+
// ---------------------------------------------------------------------------
|
|
512
|
+
async function handleReport(argv) {
|
|
513
|
+
const { values } = parseArgs({
|
|
514
|
+
args: argv,
|
|
515
|
+
options: {
|
|
516
|
+
port: { type: 'string', default: '7799' },
|
|
517
|
+
'reports-dir': { type: 'string', default: DEFAULT_REPORTS_DIR },
|
|
518
|
+
export: { type: 'string' },
|
|
519
|
+
dev: { type: 'boolean', default: false },
|
|
520
|
+
},
|
|
521
|
+
strict: false,
|
|
522
|
+
});
|
|
523
|
+
// Dev mode: restart server on file changes via node --watch
|
|
524
|
+
if (values.dev && !process.env.__OMK_DEV_CHILD) {
|
|
525
|
+
const { spawn } = await import('node:child_process');
|
|
526
|
+
const { fileURLToPath } = await import('node:url');
|
|
527
|
+
const cliPath = fileURLToPath(import.meta.url);
|
|
528
|
+
const libDir = resolve(cliPath, '..', 'lib');
|
|
529
|
+
const args = [
|
|
530
|
+
'--watch-path', libDir, cliPath, 'bench', 'report',
|
|
531
|
+
'--port', values.port,
|
|
532
|
+
'--reports-dir', values['reports-dir'],
|
|
533
|
+
];
|
|
534
|
+
const child = spawn(process.execPath, args, {
|
|
535
|
+
stdio: 'inherit',
|
|
536
|
+
env: { ...process.env, __OMK_DEV_CHILD: '1' },
|
|
537
|
+
});
|
|
538
|
+
child.on('exit', (code) => process.exit(code || 0));
|
|
539
|
+
return;
|
|
540
|
+
}
|
|
541
|
+
if (values.export) {
|
|
542
|
+
const { createFileStore } = await import('./server/report-store.js');
|
|
543
|
+
const { renderRunDetail, renderEachRunDetail } = await import('./renderer/html-renderer.js');
|
|
544
|
+
const { writeFileSync } = await import('node:fs');
|
|
545
|
+
const store = createFileStore(resolve(values['reports-dir']));
|
|
546
|
+
const report = await store.get(values.export);
|
|
547
|
+
if (!report) {
|
|
548
|
+
console.error(`Report not found: ${values.export}`);
|
|
549
|
+
process.exit(1);
|
|
550
|
+
}
|
|
551
|
+
const html = report.each ? renderEachRunDetail(report) : renderRunDetail(report);
|
|
552
|
+
const outPath = resolve(`${values.export}.html`);
|
|
553
|
+
writeFileSync(outPath, html);
|
|
554
|
+
console.log(`Exported to: ${outPath}`);
|
|
555
|
+
console.log('Open in browser, or Ctrl+P to save as PDF');
|
|
556
|
+
return;
|
|
557
|
+
}
|
|
558
|
+
const { createReportServer } = await import('./server/report-server.js');
|
|
559
|
+
const server = createReportServer({
|
|
560
|
+
port: Number(values.port),
|
|
561
|
+
reportsDir: resolve(values['reports-dir']),
|
|
562
|
+
});
|
|
563
|
+
const url = await server.start();
|
|
564
|
+
console.log(`Report server running at ${url}`);
|
|
565
|
+
console.log('Press Ctrl+C to stop');
|
|
566
|
+
}
|
|
567
|
+
// ---------------------------------------------------------------------------
|
|
568
|
+
// handleInit
|
|
569
|
+
// ---------------------------------------------------------------------------
|
|
570
|
+
const INIT_SAMPLES = `[
|
|
571
|
+
{
|
|
572
|
+
"sample_id": "s001",
|
|
573
|
+
"prompt": "审查以下代码",
|
|
574
|
+
"context": "function authenticate(username, password) {\\n const query = \`SELECT * FROM users WHERE name='\${username}' AND pass='\${password}'\`;\\n return db.execute(query);\\n}",
|
|
575
|
+
"rubric": "应识别 SQL 注入风险,建议使用参数化查询",
|
|
576
|
+
"assertions": [
|
|
577
|
+
{ "type": "contains", "value": "SQL", "weight": 1 },
|
|
578
|
+
{ "type": "contains", "value": "注入", "weight": 1 },
|
|
579
|
+
{ "type": "contains", "value": "参数化", "weight": 0.5 },
|
|
580
|
+
{ "type": "not_contains", "value": "没有问题", "weight": 0.5 }
|
|
581
|
+
],
|
|
582
|
+
"dimensions": {
|
|
583
|
+
"security": "是否准确识别出 SQL 注入漏洞并说明其危害",
|
|
584
|
+
"actionability": "是否给出可直接使用的参数化查询修复代码"
|
|
585
|
+
}
|
|
586
|
+
},
|
|
587
|
+
{
|
|
588
|
+
"sample_id": "s002",
|
|
589
|
+
"prompt": "审查以下代码",
|
|
590
|
+
"context": "async function fetchData(url) {\\n const res = await fetch(url);\\n const data = await res.json();\\n return data;\\n}",
|
|
591
|
+
"rubric": "应指出缺少错误处理(网络异常、非 JSON 响应、HTTP 错误状态码)",
|
|
592
|
+
"assertions": [
|
|
593
|
+
{ "type": "contains", "value": "错误处理", "weight": 1 },
|
|
594
|
+
{ "type": "regex", "pattern": "try[\\\\s\\\\S]*catch|错误|异常|error", "flags": "i", "weight": 1 },
|
|
595
|
+
{ "type": "contains", "value": "status", "weight": 0.5 }
|
|
596
|
+
],
|
|
597
|
+
"dimensions": {
|
|
598
|
+
"robustness": "是否指出了所有缺失的错误处理场景",
|
|
599
|
+
"actionability": "是否给出了完整的 try-catch 修复代码"
|
|
600
|
+
}
|
|
601
|
+
},
|
|
602
|
+
{
|
|
603
|
+
"sample_id": "s003",
|
|
604
|
+
"prompt": "审查以下代码",
|
|
605
|
+
"context": "function renderComment(comment) {\\n document.getElementById('output').innerHTML = '<p>' + comment + '</p>';\\n}",
|
|
606
|
+
"rubric": "应识别 XSS 风险,建议使用 textContent 或转义 HTML",
|
|
607
|
+
"assertions": [
|
|
608
|
+
{ "type": "contains", "value": "XSS", "weight": 1 },
|
|
609
|
+
{ "type": "regex", "pattern": "textContent|转义|escape|sanitize", "flags": "i", "weight": 1 },
|
|
610
|
+
{ "type": "contains", "value": "innerHTML", "weight": 0.5 }
|
|
611
|
+
],
|
|
612
|
+
"dimensions": {
|
|
613
|
+
"security": "是否准确识别出 XSS 漏洞并说明攻击方式",
|
|
614
|
+
"actionability": "是否给出使用 textContent 或转义的修复代码"
|
|
615
|
+
}
|
|
616
|
+
}
|
|
617
|
+
]
|
|
618
|
+
`;
|
|
619
|
+
const INIT_SKILL_V1 = '你是一个代码审查助手。请审查用户提供的代码,指出潜在问题。';
|
|
620
|
+
const INIT_SKILL_V2 = `你是一个高级代码审查专家。请从以下维度审查用户提供的代码:
|
|
621
|
+
|
|
622
|
+
1. 安全性:是否存在注入、XSS、敏感信息泄露等风险
|
|
623
|
+
2. 健壮性:是否有适当的错误处理和边界检查
|
|
624
|
+
3. 可维护性:命名是否清晰、结构是否合理
|
|
625
|
+
4. 性能:是否存在明显的性能瓶颈
|
|
626
|
+
|
|
627
|
+
对每个维度给出具体的改进建议,并标注严重程度(高/中/低)。
|
|
628
|
+
`;
|
|
629
|
+
// ---------------------------------------------------------------------------
|
|
630
|
+
// handleAnalyze (v0.18 skill 健康度日报)
|
|
631
|
+
// ---------------------------------------------------------------------------
|
|
632
|
+
function parseLastWindow(spec) {
|
|
633
|
+
// "7d" / "24h" / "30m" → ISO timestamp (from = now - spec)
|
|
634
|
+
const m = /^(\d+)([dhm])$/.exec(spec);
|
|
635
|
+
if (!m)
|
|
636
|
+
return null;
|
|
637
|
+
const n = Number(m[1]);
|
|
638
|
+
const unit = m[2];
|
|
639
|
+
const ms = unit === 'd' ? n * 86400_000 : unit === 'h' ? n * 3600_000 : n * 60_000;
|
|
640
|
+
return new Date(Date.now() - ms).toISOString();
|
|
641
|
+
}
|
|
642
|
+
async function handleAnalyze(argv) {
|
|
643
|
+
const { values, positionals } = parseArgs({
|
|
644
|
+
args: argv,
|
|
645
|
+
allowPositionals: true,
|
|
646
|
+
options: {
|
|
647
|
+
kb: { type: 'string' },
|
|
648
|
+
last: { type: 'string' },
|
|
649
|
+
from: { type: 'string' },
|
|
650
|
+
to: { type: 'string' },
|
|
651
|
+
skills: { type: 'string' },
|
|
652
|
+
'output-dir': { type: 'string' },
|
|
653
|
+
},
|
|
654
|
+
});
|
|
655
|
+
const dir = positionals[0];
|
|
656
|
+
if (!dir) {
|
|
657
|
+
console.error('Usage: omk analyze <dir> [--kb <path>] [--last 7d] [--from ISO] [--to ISO] [--skills name1,name2]');
|
|
658
|
+
process.exit(1);
|
|
659
|
+
}
|
|
660
|
+
const tracePath = resolve(dir);
|
|
661
|
+
const { existsSync, mkdirSync, writeFileSync } = await import('node:fs');
|
|
662
|
+
if (!existsSync(tracePath)) {
|
|
663
|
+
console.error(`Trace path does not exist: ${tracePath}`);
|
|
664
|
+
process.exit(1);
|
|
665
|
+
}
|
|
666
|
+
// 时间窗: --from/--to 优先, --last fallback
|
|
667
|
+
let from = values.from;
|
|
668
|
+
if (!from && values.last) {
|
|
669
|
+
const inferred = parseLastWindow(values.last);
|
|
670
|
+
if (!inferred) {
|
|
671
|
+
console.error(`Invalid --last format: "${values.last}". Expected e.g. "7d" / "24h" / "30m".`);
|
|
672
|
+
process.exit(1);
|
|
673
|
+
}
|
|
674
|
+
from = inferred;
|
|
675
|
+
}
|
|
676
|
+
const to = values.to;
|
|
677
|
+
const skills = values.skills ? values.skills.split(',').map((s) => s.trim()).filter(Boolean) : undefined;
|
|
678
|
+
console.log(`[omk] analyzing ${tracePath}...`);
|
|
679
|
+
const { computeSkillHealthReport } = await import('./observability/production-analyzer.js');
|
|
680
|
+
const report = computeSkillHealthReport(tracePath, {
|
|
681
|
+
kbRoot: values.kb ? resolve(values.kb) : undefined,
|
|
682
|
+
from,
|
|
683
|
+
to,
|
|
684
|
+
skills,
|
|
685
|
+
});
|
|
686
|
+
const { renderSkillHealthReport } = await import('./renderer/skill-health-renderer.js');
|
|
687
|
+
const html = renderSkillHealthReport(report);
|
|
688
|
+
const outDir = resolve(values['output-dir'] || join(process.env.HOME || '.', '.oh-my-knowledge', 'analyses'));
|
|
689
|
+
mkdirSync(outDir, { recursive: true });
|
|
690
|
+
const timestamp = new Date().toISOString().replace(/[:.]/g, '-').slice(0, 19);
|
|
691
|
+
const outPath = join(outDir, `${timestamp}-skill-health.html`);
|
|
692
|
+
writeFileSync(outPath, html);
|
|
693
|
+
// 控制台摘要
|
|
694
|
+
const { sessionCount, segmentCount, toolCallCount, toolFailureRate } = report.meta;
|
|
695
|
+
console.log('');
|
|
696
|
+
console.log(`sessions: ${sessionCount} · segments: ${segmentCount} · tool calls: ${toolCallCount} · fail rate: ${(toolFailureRate * 100).toFixed(1)}%`);
|
|
697
|
+
console.log(`overall: gapRate ${(report.overall.gapRate * 100).toFixed(1)}% · weightedGapRate ${(report.overall.weightedGapRate * 100).toFixed(1)}% · health: ${report.overall.healthBand}`);
|
|
698
|
+
console.log('');
|
|
699
|
+
const skillRows = Object.values(report.bySkill)
|
|
700
|
+
.sort((a, b) => b.segmentCount - a.segmentCount)
|
|
701
|
+
.slice(0, 10)
|
|
702
|
+
.map((s) => ` ${s.skillName.padEnd(24)} segs=${String(s.segmentCount).padStart(4)} gapRate=${String(Math.round(s.gap.gapRate * 100) + '%').padStart(4)} weighted=${String(Math.round(s.gap.weightedGapRate * 100) + '%').padStart(4)}${s.coverage ? ` cov=${Math.round(s.coverage.fileCoverageRate * 100)}%` : ''}`);
|
|
703
|
+
console.log('top skills:');
|
|
704
|
+
console.log(skillRows.join('\n'));
|
|
705
|
+
console.log('');
|
|
706
|
+
console.log(`report written to: ${outPath}`);
|
|
707
|
+
}
|
|
708
|
+
async function handleInit(argv) {
|
|
709
|
+
const targetDir = resolve(argv[0] || '.');
|
|
710
|
+
const { writeFileSync, mkdirSync } = await import('node:fs');
|
|
711
|
+
mkdirSync(join(targetDir, 'skills'), { recursive: true });
|
|
712
|
+
writeFileSync(join(targetDir, 'eval-samples.json'), INIT_SAMPLES);
|
|
713
|
+
writeFileSync(join(targetDir, 'skills', 'v1.md'), INIT_SKILL_V1);
|
|
714
|
+
writeFileSync(join(targetDir, 'skills', 'v2.md'), INIT_SKILL_V2);
|
|
715
|
+
console.log(`Eval project scaffolded at: ${targetDir}`);
|
|
716
|
+
console.log('');
|
|
717
|
+
console.log('Next steps:');
|
|
718
|
+
console.log(' 1. Edit eval-samples.json to add your test cases');
|
|
719
|
+
console.log(' 2. Edit skills/v1.md and skills/v2.md with your skill versions');
|
|
720
|
+
console.log(' 3. Run: omk bench run --control v1 --treatment v2');
|
|
721
|
+
}
|
|
722
|
+
// ---------------------------------------------------------------------------
|
|
723
|
+
// handleGenSamples
|
|
724
|
+
// ---------------------------------------------------------------------------
|
|
725
|
+
async function handleGenSamples(argv) {
|
|
726
|
+
const { values } = parseArgs({
|
|
727
|
+
args: argv,
|
|
728
|
+
options: {
|
|
729
|
+
each: { type: 'boolean', default: false },
|
|
730
|
+
count: { type: 'string', default: '5' },
|
|
731
|
+
model: { type: 'string', default: 'sonnet' },
|
|
732
|
+
'skill-dir': { type: 'string', default: 'skills' },
|
|
733
|
+
},
|
|
734
|
+
strict: false,
|
|
735
|
+
allowPositionals: true,
|
|
736
|
+
});
|
|
737
|
+
const { generateSamples } = await import('./authoring/generator.js');
|
|
738
|
+
const { readFileSync, writeFileSync } = await import('node:fs');
|
|
739
|
+
const count = Math.max(1, Number(values.count) || 5);
|
|
740
|
+
const model = values.model;
|
|
741
|
+
if (values.each) {
|
|
742
|
+
// Batch mode: generate for all skills missing eval-samples
|
|
743
|
+
const skillDir = resolve(values['skill-dir']);
|
|
744
|
+
if (!existsSync(skillDir)) {
|
|
745
|
+
console.error(`Skill directory not found: ${skillDir}`);
|
|
746
|
+
process.exit(1);
|
|
747
|
+
}
|
|
748
|
+
const { readdirSync, statSync } = await import('node:fs');
|
|
749
|
+
const entries = readdirSync(skillDir);
|
|
750
|
+
let generated = 0;
|
|
751
|
+
for (const entry of entries) {
|
|
752
|
+
let name;
|
|
753
|
+
let skillPath;
|
|
754
|
+
let samplesPath;
|
|
755
|
+
const fullPath = join(skillDir, entry);
|
|
756
|
+
if (entry.endsWith('.md') && !entry.endsWith('.eval-samples.json')) {
|
|
757
|
+
name = entry.slice(0, -3);
|
|
758
|
+
skillPath = fullPath;
|
|
759
|
+
samplesPath = join(skillDir, `${name}.eval-samples.json`);
|
|
760
|
+
}
|
|
761
|
+
else if (statSync(fullPath).isDirectory()) {
|
|
762
|
+
const skillMd = join(fullPath, 'SKILL.md');
|
|
763
|
+
if (!existsSync(skillMd))
|
|
764
|
+
continue;
|
|
765
|
+
name = entry;
|
|
766
|
+
skillPath = skillMd;
|
|
767
|
+
samplesPath = join(fullPath, 'eval-samples.json');
|
|
768
|
+
}
|
|
769
|
+
else {
|
|
770
|
+
continue;
|
|
771
|
+
}
|
|
772
|
+
if (existsSync(samplesPath)) {
|
|
773
|
+
process.stderr.write(`⏭️ ${name}: eval-samples 已存在,跳过\n`);
|
|
774
|
+
continue;
|
|
775
|
+
}
|
|
776
|
+
process.stderr.write(`🔄 ${name}: 正在生成 ${count} 个测试样本...\n`);
|
|
777
|
+
try {
|
|
778
|
+
const skillContent = readFileSync(skillPath, 'utf-8');
|
|
779
|
+
const { samples, costUSD } = await generateSamples({ skillContent, count, model });
|
|
780
|
+
writeFileSync(samplesPath, JSON.stringify(samples, null, 2));
|
|
781
|
+
process.stderr.write(`✅ ${name}: 已生成 ${samples.length} 个样本 → ${samplesPath} (${costUSD > 0 ? `$${costUSD.toFixed(4)}` : ''})\n`);
|
|
782
|
+
generated++;
|
|
783
|
+
}
|
|
784
|
+
catch (err) {
|
|
785
|
+
process.stderr.write(`❌ ${name}: ${err.message}\n`);
|
|
786
|
+
}
|
|
787
|
+
}
|
|
788
|
+
if (generated === 0) {
|
|
789
|
+
console.log('没有需要生成的 eval-samples(所有 skill 已有配对文件)');
|
|
790
|
+
}
|
|
791
|
+
else {
|
|
792
|
+
console.log(`\n共生成 ${generated} 份 eval-samples,请审查后运行: omk bench run --each`);
|
|
793
|
+
}
|
|
794
|
+
}
|
|
795
|
+
else {
|
|
796
|
+
// Single skill mode
|
|
797
|
+
const skillPath = argv.find((a) => !a.startsWith('-'));
|
|
798
|
+
if (!skillPath) {
|
|
799
|
+
console.error('请指定 skill 文件路径,例如: omk bench gen-samples skills/my-skill.md');
|
|
800
|
+
process.exit(1);
|
|
801
|
+
}
|
|
802
|
+
const resolvedPath = resolve(skillPath);
|
|
803
|
+
if (!existsSync(resolvedPath)) {
|
|
804
|
+
console.error(`Skill file not found: ${resolvedPath}`);
|
|
805
|
+
process.exit(1);
|
|
806
|
+
}
|
|
807
|
+
const skillContent = readFileSync(resolvedPath, 'utf-8');
|
|
808
|
+
const outputPath = resolve('eval-samples.json');
|
|
809
|
+
if (existsSync(outputPath)) {
|
|
810
|
+
console.error(`eval-samples.json 已存在。如需覆盖请先删除。`);
|
|
811
|
+
process.exit(1);
|
|
812
|
+
}
|
|
813
|
+
process.stderr.write(`🔄 正在生成 ${count} 个测试样本...\n`);
|
|
814
|
+
try {
|
|
815
|
+
const { samples, costUSD } = await generateSamples({ skillContent, count, model });
|
|
816
|
+
writeFileSync(outputPath, JSON.stringify(samples, null, 2));
|
|
817
|
+
process.stderr.write(`✅ 已生成 ${samples.length} 个样本 → ${outputPath} (${costUSD > 0 ? `$${costUSD.toFixed(4)}` : ''})\n`);
|
|
818
|
+
console.log('\n请审查生成的测试样本后运行: omk bench run');
|
|
819
|
+
}
|
|
820
|
+
catch (err) {
|
|
821
|
+
console.error(`生成失败: ${err.message}`);
|
|
822
|
+
process.exit(1);
|
|
823
|
+
}
|
|
824
|
+
}
|
|
825
|
+
}
|
|
826
|
+
// ---------------------------------------------------------------------------
|
|
827
|
+
// handleEvolve
|
|
828
|
+
// ---------------------------------------------------------------------------
|
|
829
|
+
async function handleEvolve(argv) {
|
|
830
|
+
const { values } = parseArgs({
|
|
831
|
+
args: argv,
|
|
832
|
+
options: {
|
|
833
|
+
rounds: { type: 'string', default: '5' },
|
|
834
|
+
target: { type: 'string' },
|
|
835
|
+
samples: { type: 'string', default: 'eval-samples.json' },
|
|
836
|
+
model: { type: 'string', default: 'sonnet' },
|
|
837
|
+
'judge-model': { type: 'string', default: 'haiku' },
|
|
838
|
+
'improve-model': { type: 'string', default: 'sonnet' },
|
|
839
|
+
concurrency: { type: 'string', default: '1' },
|
|
840
|
+
timeout: { type: 'string', default: '120' },
|
|
841
|
+
executor: { type: 'string', default: 'claude' },
|
|
842
|
+
'skip-preflight': { type: 'boolean', default: false },
|
|
843
|
+
},
|
|
844
|
+
strict: false,
|
|
845
|
+
allowPositionals: true,
|
|
846
|
+
});
|
|
847
|
+
const skillPath = argv.find((a) => !a.startsWith('-'));
|
|
848
|
+
if (!skillPath) {
|
|
849
|
+
console.error('请指定 skill 文件路径,例如: omk bench evolve skills/my-skill.md');
|
|
850
|
+
process.exit(1);
|
|
851
|
+
}
|
|
852
|
+
let samplesFile = values.samples ?? 'eval-samples.json';
|
|
853
|
+
if (samplesFile === 'eval-samples.json' && !existsSync(resolve(samplesFile))) {
|
|
854
|
+
if (existsSync(resolve('eval-samples.yaml')))
|
|
855
|
+
samplesFile = 'eval-samples.yaml';
|
|
856
|
+
else if (existsSync(resolve('eval-samples.yml')))
|
|
857
|
+
samplesFile = 'eval-samples.yml';
|
|
858
|
+
}
|
|
859
|
+
const { evolveSkill } = await import('./authoring/evolver.js');
|
|
860
|
+
process.stderr.write(`\n=== Evolution: ${skillPath} ===\n`);
|
|
861
|
+
try {
|
|
862
|
+
const result = await evolveSkill({
|
|
863
|
+
skillPath: resolve(skillPath),
|
|
864
|
+
samplesPath: resolve(samplesFile),
|
|
865
|
+
rounds: Math.max(1, Number(values.rounds) || 5),
|
|
866
|
+
target: values.target ? Number(values.target) : null,
|
|
867
|
+
model: values.model,
|
|
868
|
+
judgeModel: values['judge-model'],
|
|
869
|
+
improveModel: values['improve-model'],
|
|
870
|
+
executorName: values.executor,
|
|
871
|
+
concurrency: Math.max(1, Number(values.concurrency) || 1),
|
|
872
|
+
timeoutMs: Math.max(1, Number(values.timeout) || 120) * 1000,
|
|
873
|
+
skipPreflight: values['skip-preflight'],
|
|
874
|
+
onProgress: defaultOnProgress,
|
|
875
|
+
onRoundProgress({ round, totalRounds: _totalRounds, phase, score, delta, accepted, costUSD, error }) {
|
|
876
|
+
if (phase === 'baseline') {
|
|
877
|
+
process.stderr.write(`Round 0 (baseline): score=${score.toFixed(2)} ($${costUSD.toFixed(4)})\n`);
|
|
878
|
+
}
|
|
879
|
+
else if (phase === 'error') {
|
|
880
|
+
process.stderr.write(`Round ${round}: ✗ 改进生成失败: ${error}\n`);
|
|
881
|
+
}
|
|
882
|
+
else if (phase === 'done') {
|
|
883
|
+
const deltaStr = delta >= 0 ? `+${delta.toFixed(2)}` : delta.toFixed(2);
|
|
884
|
+
const status = accepted ? '✓ ACCEPT' : '✗ REJECT';
|
|
885
|
+
process.stderr.write(`Round ${round}: score=${score.toFixed(2)} (${deltaStr}) ${status} ($${costUSD.toFixed(4)})\n`);
|
|
886
|
+
}
|
|
887
|
+
},
|
|
888
|
+
});
|
|
889
|
+
const improvement = result.startScore > 0
|
|
890
|
+
? ((result.finalScore - result.startScore) / result.startScore * 100).toFixed(1)
|
|
891
|
+
: '0';
|
|
892
|
+
process.stderr.write(`\n✅ ${result.startScore.toFixed(2)} → ${result.finalScore.toFixed(2)} (+${improvement}%) | ${result.totalRounds} 轮 | $${result.totalCostUSD.toFixed(4)}\n`);
|
|
893
|
+
process.stderr.write(`Best: ${result.bestSkillPath} → ${resolve(skillPath)}\n`);
|
|
894
|
+
process.stderr.write(`所有版本保存在: ${join(resolve(skillPath, '..'), 'evolve')}/\n`);
|
|
895
|
+
if (result.reportId) {
|
|
896
|
+
process.stderr.write(`📊 评测报告: omk bench report (ID: ${result.reportId})\n`);
|
|
897
|
+
}
|
|
898
|
+
console.log(JSON.stringify(result, null, 2));
|
|
899
|
+
}
|
|
900
|
+
catch (err) {
|
|
901
|
+
console.error(`Error: ${err.message}`);
|
|
902
|
+
process.exit(1);
|
|
903
|
+
}
|
|
904
|
+
}
|
|
905
|
+
// ---------------------------------------------------------------------------
|
|
906
|
+
// handleCi
|
|
907
|
+
// ---------------------------------------------------------------------------
|
|
908
|
+
async function handleCi(argv) {
|
|
909
|
+
const { values, config } = parseRunConfig(argv, {
|
|
910
|
+
threshold: { type: 'string', default: '3.5' },
|
|
911
|
+
});
|
|
912
|
+
const { runEvaluation } = await import('./eval-workflows/run-evaluation.js');
|
|
913
|
+
config.onProgress = defaultOnProgress;
|
|
914
|
+
try {
|
|
915
|
+
const { report } = (await runEvaluation(config));
|
|
916
|
+
const threshold = Number(values.threshold);
|
|
917
|
+
if (report.dryRun) {
|
|
918
|
+
console.log('CI dry-run: no scores to check');
|
|
919
|
+
process.exit(0);
|
|
920
|
+
}
|
|
921
|
+
// three-gate 逻辑抽到 src/eval-core/ci-gates.ts 作纯函数,便于测试;此处只做 IO。
|
|
922
|
+
const { evaluateCiGates } = await import('./eval-core/ci-gates.js');
|
|
923
|
+
const { allPass, lines } = evaluateCiGates(report.summary || {}, threshold);
|
|
924
|
+
for (const line of lines)
|
|
925
|
+
console.log(line);
|
|
926
|
+
process.exit(allPass ? 0 : 1);
|
|
927
|
+
}
|
|
928
|
+
catch (err) {
|
|
929
|
+
console.error(`Error: ${err.message}`);
|
|
930
|
+
process.exit(1);
|
|
931
|
+
}
|
|
932
|
+
}
|
|
933
|
+
// ---------------------------------------------------------------------------
|
|
934
|
+
// handleDiff
|
|
935
|
+
// ---------------------------------------------------------------------------
|
|
936
|
+
async function handleDiff(argv) {
|
|
937
|
+
if (argv.length < 2) {
|
|
938
|
+
console.error('Usage: omk bench diff <report-id-1> <report-id-2>');
|
|
939
|
+
process.exit(1);
|
|
940
|
+
}
|
|
941
|
+
const { createFileStore } = await import('./server/report-store.js');
|
|
942
|
+
const store = createFileStore(resolve(DEFAULT_REPORTS_DIR));
|
|
943
|
+
const [id1, id2] = argv;
|
|
944
|
+
const r1 = await store.get(id1);
|
|
945
|
+
const r2 = await store.get(id2);
|
|
946
|
+
if (!r1) {
|
|
947
|
+
console.error(`Report not found: ${id1}`);
|
|
948
|
+
process.exit(1);
|
|
949
|
+
}
|
|
950
|
+
if (!r2) {
|
|
951
|
+
console.error(`Report not found: ${id2}`);
|
|
952
|
+
process.exit(1);
|
|
953
|
+
}
|
|
954
|
+
console.log(`\n Diff: ${id1} → ${id2}\n`);
|
|
955
|
+
// Git info — r1/r2 are guaranteed non-null after process.exit() guards above
|
|
956
|
+
const g1 = r1.meta?.gitInfo;
|
|
957
|
+
const g2 = r2.meta?.gitInfo;
|
|
958
|
+
if (g1 || g2) {
|
|
959
|
+
console.log(` Git: ${g1?.commitShort || '?'}${g1?.dirty ? '*' : ''} (${g1?.branch || '?'}) → ${g2?.commitShort || '?'}${g2?.dirty ? '*' : ''} (${g2?.branch || '?'})`);
|
|
960
|
+
}
|
|
961
|
+
// Per-variant comparison
|
|
962
|
+
const variants = [...new Set([...(r1.meta?.variants || []), ...(r2.meta?.variants || [])])];
|
|
963
|
+
for (const v of variants) {
|
|
964
|
+
const s1 = r1.summary?.[v];
|
|
965
|
+
const s2 = r2.summary?.[v];
|
|
966
|
+
if (!s1 && !s2)
|
|
967
|
+
continue;
|
|
968
|
+
console.log(`\n [${v}]`);
|
|
969
|
+
const score1 = s1?.avgCompositeScore ?? '-';
|
|
970
|
+
const score2 = s2?.avgCompositeScore ?? '-';
|
|
971
|
+
const scoreDelta = typeof score1 === 'number' && typeof score2 === 'number'
|
|
972
|
+
? ` (${score2 > score1 ? '+' : ''}${(score2 - score1).toFixed(2)})`
|
|
973
|
+
: '';
|
|
974
|
+
console.log(` Score: ${score1} → ${score2}${scoreDelta}`);
|
|
975
|
+
const turns1 = s1?.avgNumTurns ?? '-';
|
|
976
|
+
const turns2 = s2?.avgNumTurns ?? '-';
|
|
977
|
+
console.log(` Turns: ${turns1} → ${turns2}`);
|
|
978
|
+
// Tool calls comparison (agent metrics)
|
|
979
|
+
if (s1?.avgToolCalls != null || s2?.avgToolCalls != null) {
|
|
980
|
+
const tc1 = s1?.avgToolCalls ?? '-';
|
|
981
|
+
const tc2 = s2?.avgToolCalls ?? '-';
|
|
982
|
+
console.log(` Tools: ${tc1} → ${tc2}`);
|
|
983
|
+
const sr1 = s1?.toolSuccessRate != null ? `${(s1.toolSuccessRate * 100).toFixed(0)}%` : '-';
|
|
984
|
+
const sr2 = s2?.toolSuccessRate != null ? `${(s2.toolSuccessRate * 100).toFixed(0)}%` : '-';
|
|
985
|
+
console.log(` ToolOK: ${sr1} → ${sr2}`);
|
|
986
|
+
}
|
|
987
|
+
const cost1 = s1?.avgCostPerSample ?? 0;
|
|
988
|
+
const cost2 = s2?.avgCostPerSample ?? 0;
|
|
989
|
+
const costPct = cost1 > 0 ? ` (${cost2 > cost1 ? '+' : ''}${(((cost2 - cost1) / cost1) * 100).toFixed(0)}%)` : '';
|
|
990
|
+
console.log(` Cost: $${cost1.toFixed(4)} → $${cost2.toFixed(4)}${costPct}`);
|
|
991
|
+
// Skill hash change
|
|
992
|
+
const h1 = r1.meta?.artifactHashes?.[v];
|
|
993
|
+
const h2 = r2.meta?.artifactHashes?.[v];
|
|
994
|
+
if (h1 && h2 && h1 !== h2) {
|
|
995
|
+
console.log(` Skill: ${h1.slice(0, 8)} → ${h2.slice(0, 8)} (changed)`);
|
|
996
|
+
}
|
|
997
|
+
}
|
|
998
|
+
console.log('');
|
|
999
|
+
}
|
|
1000
|
+
// ---------------------------------------------------------------------------
|
|
1001
|
+
// Entry
|
|
1002
|
+
// ---------------------------------------------------------------------------
|
|
1003
|
+
main();
|
|
1004
|
+
//# sourceMappingURL=cli.js.map
|