oh-my-knowledge 0.23.0 → 0.24.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +63 -64
- package/README.zh.md +64 -63
- package/dist/src/analysis/report-diagnostics.d.ts +3 -2
- package/dist/src/analysis/report-diagnostics.d.ts.map +1 -1
- package/dist/src/analysis/report-diagnostics.js +107 -80
- package/dist/src/analysis/report-diagnostics.js.map +1 -1
- package/dist/src/analysis/sample-diagnostics.d.ts +4 -1
- package/dist/src/analysis/sample-diagnostics.d.ts.map +1 -1
- package/dist/src/analysis/sample-diagnostics.js +123 -29
- package/dist/src/analysis/sample-diagnostics.js.map +1 -1
- package/dist/src/authoring/evolver.d.ts +5 -0
- package/dist/src/authoring/evolver.d.ts.map +1 -1
- package/dist/src/authoring/evolver.js +20 -3
- package/dist/src/authoring/evolver.js.map +1 -1
- package/dist/src/cli/i18n-dict.d.ts +1 -1
- package/dist/src/cli/i18n-dict.d.ts.map +1 -1
- package/dist/src/cli/i18n-dict.js +28 -24
- package/dist/src/cli/i18n-dict.js.map +1 -1
- package/dist/src/cli/index.js +80 -66
- package/dist/src/cli/index.js.map +1 -1
- package/dist/src/cli/parse-run-config.d.ts.map +1 -1
- package/dist/src/cli/parse-run-config.js +4 -4
- package/dist/src/cli/parse-run-config.js.map +1 -1
- package/dist/src/cli/progress.d.ts +1 -0
- package/dist/src/cli/progress.d.ts.map +1 -1
- package/dist/src/cli/progress.js +8 -1
- package/dist/src/cli/progress.js.map +1 -1
- package/dist/src/eval-core/cache.d.ts +7 -5
- package/dist/src/eval-core/cache.d.ts.map +1 -1
- package/dist/src/eval-core/cache.js +11 -7
- package/dist/src/eval-core/cache.js.map +1 -1
- package/dist/src/eval-core/comparability.d.ts +11 -0
- package/dist/src/eval-core/comparability.d.ts.map +1 -0
- package/dist/src/eval-core/comparability.js +271 -0
- package/dist/src/eval-core/comparability.js.map +1 -0
- package/dist/src/eval-core/evaluation-execution.d.ts +4 -1
- package/dist/src/eval-core/evaluation-execution.d.ts.map +1 -1
- package/dist/src/eval-core/evaluation-execution.js +25 -17
- package/dist/src/eval-core/evaluation-execution.js.map +1 -1
- package/dist/src/eval-core/evaluation-job.d.ts +2 -2
- package/dist/src/eval-core/evaluation-job.d.ts.map +1 -1
- package/dist/src/eval-core/evaluation-job.js +2 -2
- package/dist/src/eval-core/evaluation-job.js.map +1 -1
- package/dist/src/eval-core/evaluation-reporting.d.ts +3 -1
- package/dist/src/eval-core/evaluation-reporting.d.ts.map +1 -1
- package/dist/src/eval-core/evaluation-reporting.js +64 -4
- package/dist/src/eval-core/evaluation-reporting.js.map +1 -1
- package/dist/src/eval-core/execution-strategy.js +2 -2
- package/dist/src/eval-core/execution-strategy.js.map +1 -1
- package/dist/src/eval-core/schema.d.ts.map +1 -1
- package/dist/src/eval-core/schema.js +13 -0
- package/dist/src/eval-core/schema.js.map +1 -1
- package/dist/src/eval-workflows/batch-evaluation-workflow.d.ts +111 -0
- package/dist/src/eval-workflows/batch-evaluation-workflow.d.ts.map +1 -0
- package/dist/src/eval-workflows/batch-evaluation-workflow.js +215 -0
- package/dist/src/eval-workflows/batch-evaluation-workflow.js.map +1 -0
- package/dist/src/eval-workflows/evaluation-pipeline.d.ts +5 -3
- package/dist/src/eval-workflows/evaluation-pipeline.d.ts.map +1 -1
- package/dist/src/eval-workflows/evaluation-pipeline.js +6 -5
- package/dist/src/eval-workflows/evaluation-pipeline.js.map +1 -1
- package/dist/src/eval-workflows/evaluation-preparation.d.ts +8 -6
- package/dist/src/eval-workflows/evaluation-preparation.d.ts.map +1 -1
- package/dist/src/eval-workflows/evaluation-preparation.js +8 -4
- package/dist/src/eval-workflows/evaluation-preparation.js.map +1 -1
- package/dist/src/eval-workflows/run-evaluation.d.ts +20 -17
- package/dist/src/eval-workflows/run-evaluation.d.ts.map +1 -1
- package/dist/src/eval-workflows/run-evaluation.js +25 -15
- package/dist/src/eval-workflows/run-evaluation.js.map +1 -1
- package/dist/src/executors/claude-cli.d.ts.map +1 -1
- package/dist/src/executors/claude-cli.js +11 -6
- package/dist/src/executors/claude-cli.js.map +1 -1
- package/dist/src/executors/codex-cli-trace.d.ts +10 -0
- package/dist/src/executors/codex-cli-trace.d.ts.map +1 -0
- package/dist/src/executors/codex-cli-trace.js +123 -0
- package/dist/src/executors/codex-cli-trace.js.map +1 -0
- package/dist/src/executors/codex-cli.d.ts +18 -0
- package/dist/src/executors/codex-cli.d.ts.map +1 -0
- package/dist/src/executors/codex-cli.js +254 -0
- package/dist/src/executors/codex-cli.js.map +1 -0
- package/dist/src/executors/codex-sdk.d.ts +18 -0
- package/dist/src/executors/codex-sdk.d.ts.map +1 -0
- package/dist/src/executors/codex-sdk.js +214 -0
- package/dist/src/executors/codex-sdk.js.map +1 -0
- package/dist/src/executors/gemini.d.ts.map +1 -1
- package/dist/src/executors/gemini.js +28 -24
- package/dist/src/executors/gemini.js.map +1 -1
- package/dist/src/executors/index.d.ts.map +1 -1
- package/dist/src/executors/index.js +7 -2
- package/dist/src/executors/index.js.map +1 -1
- package/dist/src/executors/runtime-fingerprint.d.ts +7 -0
- package/dist/src/executors/runtime-fingerprint.d.ts.map +1 -0
- package/dist/src/executors/runtime-fingerprint.js +277 -0
- package/dist/src/executors/runtime-fingerprint.js.map +1 -0
- package/dist/src/executors/script.d.ts.map +1 -1
- package/dist/src/executors/script.js +47 -55
- package/dist/src/executors/script.js.map +1 -1
- package/dist/src/executors/shared.d.ts +78 -1
- package/dist/src/executors/shared.d.ts.map +1 -1
- package/dist/src/executors/shared.js +203 -1
- package/dist/src/executors/shared.js.map +1 -1
- package/dist/src/grading/assertions.d.ts.map +1 -1
- package/dist/src/grading/assertions.js +18 -3
- package/dist/src/grading/assertions.js.map +1 -1
- package/dist/src/grading/index.d.ts.map +1 -1
- package/dist/src/grading/index.js +11 -0
- package/dist/src/grading/index.js.map +1 -1
- package/dist/src/grading/judge.d.ts.map +1 -1
- package/dist/src/grading/judge.js +75 -6
- package/dist/src/grading/judge.js.map +1 -1
- package/dist/src/inputs/skill-loader.d.ts +1 -1
- package/dist/src/inputs/skill-loader.d.ts.map +1 -1
- package/dist/src/inputs/skill-loader.js +1 -1
- package/dist/src/inputs/skill-loader.js.map +1 -1
- package/dist/src/renderer/html-renderer.d.ts +5 -4
- package/dist/src/renderer/html-renderer.d.ts.map +1 -1
- package/dist/src/renderer/html-renderer.js +217 -93
- package/dist/src/renderer/html-renderer.js.map +1 -1
- package/dist/src/renderer/layout.d.ts +2 -1
- package/dist/src/renderer/layout.d.ts.map +1 -1
- package/dist/src/renderer/layout.js +30 -42
- package/dist/src/renderer/layout.js.map +1 -1
- package/dist/src/renderer/summary.d.ts +3 -3
- package/dist/src/renderer/summary.d.ts.map +1 -1
- package/dist/src/renderer/summary.js +232 -42
- package/dist/src/renderer/summary.js.map +1 -1
- package/dist/src/renderer/trends.d.ts.map +1 -1
- package/dist/src/renderer/trends.js +5 -3
- package/dist/src/renderer/trends.js.map +1 -1
- package/dist/src/server/report-server.js +4 -4
- package/dist/src/server/report-server.js.map +1 -1
- package/dist/src/server/report-store.d.ts +7 -5
- package/dist/src/server/report-store.d.ts.map +1 -1
- package/dist/src/server/report-store.js +39 -11
- package/dist/src/server/report-store.js.map +1 -1
- package/dist/src/types/eval.d.ts +4 -2
- package/dist/src/types/eval.d.ts.map +1 -1
- package/dist/src/types/executor.d.ts +5 -0
- package/dist/src/types/executor.d.ts.map +1 -1
- package/dist/src/types/judge.d.ts +10 -0
- package/dist/src/types/judge.d.ts.map +1 -1
- package/dist/src/types/report.d.ts +120 -24
- package/dist/src/types/report.d.ts.map +1 -1
- package/dist/src/types/storage.d.ts +7 -7
- package/dist/src/types/storage.d.ts.map +1 -1
- package/package.json +2 -1
- package/dist/src/eval-workflows/each-evaluation-workflow.d.ts +0 -153
- package/dist/src/eval-workflows/each-evaluation-workflow.d.ts.map +0 -1
- package/dist/src/eval-workflows/each-evaluation-workflow.js +0 -178
- package/dist/src/eval-workflows/each-evaluation-workflow.js.map +0 -1
- package/dist/src/executors/openai-cli.d.ts +0 -3
- package/dist/src/executors/openai-cli.d.ts.map +0 -1
- package/dist/src/executors/openai-cli.js +0 -60
- package/dist/src/executors/openai-cli.js.map +0 -1
|
@@ -128,6 +128,10 @@ export const CLI_DICT = {
|
|
|
128
128
|
zh: '[{i}/{n}] {sample}/{variant} ✓ {ms}ms {input}+{output} tokens{cost}{score}\n',
|
|
129
129
|
en: '[{i}/{n}] {sample}/{variant} ✓ {ms}ms {input}+{output} tokens{cost}{score}\n',
|
|
130
130
|
},
|
|
131
|
+
'cli.progress.sample_failed_done': {
|
|
132
|
+
zh: '[{i}/{n}] {sample}/{variant} ✗ {ms}ms {input}+{output} tokens{cost} error={error}\n',
|
|
133
|
+
en: '[{i}/{n}] {sample}/{variant} ✗ {ms}ms {input}+{output} tokens{cost} error={error}\n',
|
|
134
|
+
},
|
|
131
135
|
'cli.run.invalid_repeat': {
|
|
132
136
|
zh: '⚠ --repeat "{value}" 无效 (期望 ≥ 1 的整数), 已按 1 次评测执行\n',
|
|
133
137
|
en: '⚠ --repeat "{value}" is invalid (expected an integer ≥ 1), falling back to 1 run\n',
|
|
@@ -261,8 +265,8 @@ export const CLI_DICT = {
|
|
|
261
265
|
en: 'No eval-samples need generating (all skills already have paired files)',
|
|
262
266
|
},
|
|
263
267
|
'cli.gen.batch_summary': {
|
|
264
|
-
zh: '\n共生成 {n} 份 eval-samples, 请审查后运行: omk bench run --
|
|
265
|
-
en: '\nGenerated {n} eval-samples files. Review them, then run: omk bench run --
|
|
268
|
+
zh: '\n共生成 {n} 份 eval-samples, 请审查后运行: omk bench run --batch',
|
|
269
|
+
en: '\nGenerated {n} eval-samples files. Review them, then run: omk bench run --batch',
|
|
266
270
|
},
|
|
267
271
|
'cli.gen.specify_skill_path': {
|
|
268
272
|
zh: '请指定 skill 文件路径, 例如: omk bench gen-samples skills/my-skill.md',
|
|
@@ -297,20 +301,20 @@ export const CLI_DICT = {
|
|
|
297
301
|
en: '\n=== Evolution: {path} ===\n',
|
|
298
302
|
},
|
|
299
303
|
'cli.evolve.round_baseline': {
|
|
300
|
-
zh: '第 0 轮 (基线): score={score} (
|
|
301
|
-
en: 'Round 0 (baseline): score={score} (
|
|
304
|
+
zh: '第 0 轮 (基线): score={score} ({cost})\n',
|
|
305
|
+
en: 'Round 0 (baseline): score={score} ({cost})\n',
|
|
302
306
|
},
|
|
303
307
|
'cli.evolve.round_error': {
|
|
304
308
|
zh: '第 {round} 轮: ✗ 改进生成失败: {error}\n',
|
|
305
309
|
en: 'Round {round}: ✗ improvement generation failed: {error}\n',
|
|
306
310
|
},
|
|
307
311
|
'cli.evolve.round_done': {
|
|
308
|
-
zh: '第 {round} 轮: score={score} ({delta}) {status} (
|
|
309
|
-
en: 'Round {round}: score={score} ({delta}) {status} (
|
|
312
|
+
zh: '第 {round} 轮: score={score} ({delta}) {status} ({cost})\n',
|
|
313
|
+
en: 'Round {round}: score={score} ({delta}) {status} ({cost})\n',
|
|
310
314
|
},
|
|
311
315
|
'cli.evolve.summary': {
|
|
312
|
-
zh: '\n✅ {start} → {final} (+{percent}%) | 共 {rounds} 轮 |
|
|
313
|
-
en: '\n✅ {start} → {final} (+{percent}%) | {rounds} rounds |
|
|
316
|
+
zh: '\n✅ {start} → {final} (+{percent}%) | 共 {rounds} 轮 | {cost}\n',
|
|
317
|
+
en: '\n✅ {start} → {final} (+{percent}%) | {rounds} rounds | {cost}\n',
|
|
314
318
|
},
|
|
315
319
|
'cli.evolve.best_path': {
|
|
316
320
|
zh: '最优版本: {best} → {target}\n',
|
|
@@ -412,7 +416,7 @@ bench run 选项:
|
|
|
412
416
|
在一个文件里声明 samples + variants + model + executor。
|
|
413
417
|
CLI flag 会覆盖配置文件中的同名字段。
|
|
414
418
|
配置中的相对路径相对于配置文件所在目录解析。
|
|
415
|
-
--model <name>
|
|
419
|
+
--model <name> 任务执行模型 (默认: sonnet)
|
|
416
420
|
--judge-model <name> 评委模型 (默认: haiku)
|
|
417
421
|
--output-dir <path> 报告输出目录 (默认: ~/.oh-my-knowledge/reports/)
|
|
418
422
|
--no-judge 跳过 LLM 评委
|
|
@@ -439,10 +443,10 @@ bench run 选项:
|
|
|
439
443
|
stderr 警告提示耗时。
|
|
440
444
|
--retry <n> 失败任务最多重试 N 次, 指数退避 (默认: 0)
|
|
441
445
|
--resume <report-id> 从历史报告恢复, 跳过已完成任务
|
|
442
|
-
--executor <name> 执行器: claude /
|
|
443
|
-
openai-api, 或任意 shell 命令 (例如 "python my_provider.py")
|
|
446
|
+
--executor <name> 执行器: claude / claude-sdk / codex / openai / gemini /
|
|
447
|
+
anthropic-api / openai-api, 或任意 shell 命令 (例如 "python my_provider.py")
|
|
444
448
|
--judge-executor <name> 评委执行器 (默认: 同 --executor)
|
|
445
|
-
--
|
|
449
|
+
--batch 批量评测:每个 skill 独立 vs baseline
|
|
446
450
|
需要每个 skill 有配对的 {name}.eval-samples.json
|
|
447
451
|
--skip-preflight 评测前跳过模型连通性预检
|
|
448
452
|
--mcp-config <path> 通过 MCP server 抓 URL 用的 MCP 配置文件
|
|
@@ -483,10 +487,10 @@ bench report 选项:
|
|
|
483
487
|
--dev 开发模式: lib/ 文件改动时自动重启
|
|
484
488
|
|
|
485
489
|
bench gen-samples 选项:
|
|
486
|
-
--
|
|
490
|
+
--batch 为所有还没 eval-samples 的 skill 生成
|
|
487
491
|
--count <n> 每个 skill 生成多少条用例 (默认: 5)
|
|
488
492
|
--model <name> 生成用的模型 (默认: sonnet)
|
|
489
|
-
--skill-dir <path> skill 目录 (默认: skills), 配合 --
|
|
493
|
+
--skill-dir <path> skill 目录 (默认: skills), 配合 --batch 用
|
|
490
494
|
|
|
491
495
|
analyze 选项:
|
|
492
496
|
<dir> 输入: cc session JSONL 文件 / 目录
|
|
@@ -502,7 +506,7 @@ bench evolve 选项:
|
|
|
502
506
|
--rounds <n> 最大演化轮数 (默认: 5)
|
|
503
507
|
--target <score> 达到该分数即提前停止
|
|
504
508
|
--samples <path> 用例文件 (默认: eval-samples.json)
|
|
505
|
-
--model <name>
|
|
509
|
+
--model <name> 任务执行模型 (默认: sonnet)
|
|
506
510
|
--judge-model <name> 评委模型 (默认: haiku)
|
|
507
511
|
--improve-model <name> 生成改进版的模型 (默认: sonnet)
|
|
508
512
|
--concurrency <n> 并发评测任务数 (默认: 1)
|
|
@@ -520,7 +524,7 @@ bench evolve 选项:
|
|
|
520
524
|
omk bench run --control baseline --treatment v1,v2,v3
|
|
521
525
|
omk bench run --config eval.yaml
|
|
522
526
|
omk bench run --config eval.yaml --model sonnet-4.6 # CLI 覆盖配置
|
|
523
|
-
omk bench run --
|
|
527
|
+
omk bench run --batch
|
|
524
528
|
omk bench run --dry-run
|
|
525
529
|
omk bench report --port 8080
|
|
526
530
|
omk bench report --export v1-vs-v2-20260326-1832
|
|
@@ -558,7 +562,7 @@ Options for "bench run":
|
|
|
558
562
|
Declares samples + variants + model + executor in one file.
|
|
559
563
|
CLI flags override config fields when both are provided.
|
|
560
564
|
Relative paths inside the config are resolved against its directory.
|
|
561
|
-
--model <name>
|
|
565
|
+
--model <name> Task execution model (default: sonnet)
|
|
562
566
|
--judge-model <name> Judge model (default: haiku)
|
|
563
567
|
--output-dir <path> Report output directory (default: ~/.oh-my-knowledge/reports/)
|
|
564
568
|
--no-judge Skip LLM judging
|
|
@@ -586,10 +590,10 @@ Options for "bench run":
|
|
|
586
590
|
triggers a stderr warning about runtime cost.
|
|
587
591
|
--retry <n> Retry failed tasks up to N times with exponential backoff (default: 0)
|
|
588
592
|
--resume <report-id> Resume from a previous report, skipping completed tasks
|
|
589
|
-
--executor <name> Executor: claude,
|
|
590
|
-
or any shell command (e.g. "python my_provider.py")
|
|
593
|
+
--executor <name> Executor: claude, claude-sdk, codex, openai, gemini,
|
|
594
|
+
anthropic-api, openai-api, or any shell command (e.g. "python my_provider.py")
|
|
591
595
|
--judge-executor <name> Executor for LLM judge (default: same as --executor)
|
|
592
|
-
--
|
|
596
|
+
--batch Batch evaluation: each skill independently against baseline
|
|
593
597
|
Requires {name}.eval-samples.json paired with each skill
|
|
594
598
|
--skip-preflight Skip model connectivity check before evaluation
|
|
595
599
|
--mcp-config <path> MCP config file for URL fetching via MCP servers
|
|
@@ -637,10 +641,10 @@ Options for "bench report":
|
|
|
637
641
|
--dev Dev mode: auto-restart on lib/ file changes
|
|
638
642
|
|
|
639
643
|
Options for "bench gen-samples":
|
|
640
|
-
--
|
|
644
|
+
--batch Generate for all skills missing eval-samples
|
|
641
645
|
--count <n> Number of samples to generate per skill (default: 5)
|
|
642
646
|
--model <name> Model for generation (default: sonnet)
|
|
643
|
-
--skill-dir <path> Skill directory (default: skills), used with --
|
|
647
|
+
--skill-dir <path> Skill directory (default: skills), used with --batch
|
|
644
648
|
|
|
645
649
|
Options for "analyze":
|
|
646
650
|
<dir> Input: cc session JSONL file / dir (e.g. ~/.claude/projects/<slug>)
|
|
@@ -655,7 +659,7 @@ Options for "bench evolve":
|
|
|
655
659
|
--rounds <n> Maximum evolution rounds (default: 5)
|
|
656
660
|
--target <score> Stop early when score reaches this threshold
|
|
657
661
|
--samples <path> Sample file (default: eval-samples.json)
|
|
658
|
-
--model <name>
|
|
662
|
+
--model <name> Task execution model (default: sonnet)
|
|
659
663
|
--judge-model <name> Judge model (default: haiku)
|
|
660
664
|
--improve-model <name> Model for generating improvements (default: sonnet)
|
|
661
665
|
--concurrency <n> Parallel eval tasks (default: 1)
|
|
@@ -673,7 +677,7 @@ Examples:
|
|
|
673
677
|
omk bench run --control baseline --treatment v1,v2,v3
|
|
674
678
|
omk bench run --config eval.yaml
|
|
675
679
|
omk bench run --config eval.yaml --model sonnet-4.6 # CLI overrides config
|
|
676
|
-
omk bench run --
|
|
680
|
+
omk bench run --batch
|
|
677
681
|
omk bench run --dry-run
|
|
678
682
|
omk bench report --port 8080
|
|
679
683
|
omk bench report --export v1-vs-v2-20260326-1832
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"i18n-dict.js","sourceRoot":"","sources":["../../../src/cli/i18n-dict.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;GAgDG;
|
|
1
|
+
{"version":3,"file":"i18n-dict.js","sourceRoot":"","sources":["../../../src/cli/i18n-dict.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;GAgDG;AA0HH,MAAM,CAAC,MAAM,QAAQ,GAAsC;IACzD,gCAAgC,EAAE;QAChC,EAAE,EAAE,0CAA0C;QAC9C,EAAE,EAAE,uEAAuE;KAC5E;IACD,sBAAsB,EAAE;QACtB,EAAE,EAAE,sBAAsB;QAC1B,EAAE,EAAE,+BAA+B;KACpC;IACD,2BAA2B,EAAE;QAC3B,EAAE,EAAE,mEAAmE;QACvE,EAAE,EAAE,6EAA6E;KAClF;IACD,kCAAkC,EAAE;QAClC,EAAE,EAAE,iDAAiD;QACrD,EAAE,EAAE,yEAAyE;KAC9E;IACD,qBAAqB,EAAE;QACrB,EAAE,EAAE,iBAAiB;QACrB,EAAE,EAAE,mCAAmC;KACxC;IACD,2BAA2B,EAAE;QAC3B,EAAE,EAAE,MAAM;QACV,EAAE,EAAE,aAAa;KAClB;IACD,iCAAiC,EAAE;QACjC,EAAE,EAAE,uCAAuC;QAC3C,EAAE,EAAE,oDAAoD;KACzD;IACD,gCAAgC,EAAE;QAChC,EAAE,EAAE,yDAAyD;QAC7D,EAAE,EAAE,kEAAkE;KACvE;IACD,wBAAwB,EAAE;QACxB,EAAE,EAAE,oDAAoD;QACxD,EAAE,EAAE,qDAAqD;KAC1D;IACD,kCAAkC,EAAE;QAClC,EAAE,EAAE,0DAA0D;QAC9D,EAAE,EAAE,mFAAmF;KACxF;IACD,iCAAiC,EAAE;QACjC,EAAE,EAAE,kBAAkB;QACtB,EAAE,EAAE,+CAA+C;KACpD;IACD,2BAA2B,EAAE;QAC3B,EAAE,EAAE,yDAAyD;QAC7D,EAAE,EAAE,4DAA4D;KACjE;IACD,2BAA2B,EAAE;QAC3B,EAAE,EAAE,0CAA0C;QAC9C,EAAE,EAAE,0CAA0C;KAC/C;IACD,+BAA+B,EAAE;QAC/B,EAAE,EAAE,yCAAyC;QAC7C,EAAE,EAAE,6CAA6C;KAClD;IACD,+BAA+B,EAAE;QAC/B,EAAE,EAAE,0EAA0E;QAC9E,EAAE,EAAE,0EAA0E;KAC/E;IACD,6BAA6B,EAAE;QAC7B,EAAE,EAAE,qBAAqB;QACzB,EAAE,EAAE,+BAA+B;KACpC;IACD,sBAAsB,EAAE;QACtB,EAAE,EAAE,8CAA8C;QAClD,EAAE,EAAE,gDAAgD;KACrD;IACD,qBAAqB,EAAE;QACrB,EAAE,EAAE,0DAA0D;QAC9D,EAAE,EAAE,0DAA0D;KAC/D;IACD,sBAAsB,EAAE;QACtB,EAAE,EAAE,6CAA6C;QACjD,EAAE,EAAE,mDAAmD;KACxD;IACD,0BAA0B,EAAE;QAC1B,EAAE,EAAE,8EAA8E;QAClF,EAAE,EAAE,8EAA8E;KACnF;IACD,iCAAiC,EAAE;QACjC,EAAE,EAAE,qFAAqF;QACzF,EAAE,EAAE,qFAAqF;KAC1F;IACD,wBAAwB,EAAE;QACxB,EAAE,EAAE,oDAAoD;QACxD,EAAE,EAAE,oFAAoF;KACzF;IACD,8BAA8B,EAAE;QAC9B,EAAE,EAAE,+DAA+D;QACnE,EAAE,EAAE,iGAAiG;KACtG;IACD,qCAAqC,EAAE;QACrC,EAAE,EAAE,qEAAqE;QACzE,EAAE,EAAE,qFAAqF;KAC1F;IACD,qCAAqC,EAAE;QACrC,EAAE,EAAE,gGAAgG;QACpG,EAAE,EAAE,iIAAiI;KACtI;IACD,iCAAiC,EAAE;QACjC,EAAE,EAAE,2EAA2E;QAC/E,EAAE,EAAE,mGAAmG;KACxG;IACD,mCAAmC,EAAE;QACnC,EAAE,EAAE,iEAAiE;QACrE,EAAE,EAAE,gGAAgG;KACrG;IACD,qCAAqC,EAAE;QACrC,EAAE,EAAE,2DAA2D;QAC/D,EAAE,EAAE,0HAA0H;KAC/H;IACD,uBAAuB,EAAE;QACvB,EAAE,EAAE,sCAAsC;QAC1C,EAAE,EAAE,sCAAsC;KAC3C;IACD,qBAAqB,EAAE;QACrB,EAAE,EAAE,yBAAyB;QAC7B,EAAE,EAAE,yBAAyB;KAC9B;IACD,wBAAwB,EAAE;QACxB,EAAE,EAAE,cAAc;QAClB,EAAE,EAAE,6BAA6B;KAClC;IACD,uBAAuB,EAAE;QACvB,EAAE,EAAE,YAAY;QAChB,EAAE,EAAE,uBAAuB;KAC5B;IACD,sBAAsB,EAAE;QACtB,EAAE,EAAE,qBAAqB;QACzB,EAAE,EAAE,8BAA8B;KACnC;IACD,+BAA+B,EAAE;QAC/B,EAAE,EAAE,uBAAuB;QAC3B,EAAE,EAAE,uCAAuC;KAC5C;IACD,4BAA4B,EAAE;QAC5B,EAAE,EAAE,kBAAkB;QACtB,EAAE,EAAE,yBAAyB;KAC9B;IACD,4BAA4B,EAAE;QAC5B,EAAE,EAAE,mBAAmB;QACvB,EAAE,EAAE,qCAAqC;KAC1C;IACD,6BAA6B,EAAE;QAC7B,EAAE,EAAE,iCAAiC;QACrC,EAAE,EAAE,4DAA4D;KACjE;IACD,4BAA4B,EAAE;QAC5B,EAAE,EAAE,iDAAiD;QACrD,EAAE,EAAE,wDAAwD;KAC7D;IACD,0BAA0B,EAAE;QAC1B,EAAE,EAAE,kCAAkC;QACtC,EAAE,EAAE,4CAA4C;KACjD;IACD,yBAAyB,EAAE;QACzB,EAAE,EAAE,iBAAiB;QACrB,EAAE,EAAE,iBAAiB;KACtB;IACD,+BAA+B,EAAE;QAC/B,EAAE,EAAE,iBAAiB;QACrB,EAAE,EAAE,iBAAiB;KACtB;IACD,yBAAyB,EAAE;QACzB,EAAE,EAAE,eAAe;QACnB,EAAE,EAAE,kBAAkB;KACvB;IACD,6BAA6B,EAAE;QAC7B,EAAE,EAAE,wDAAwD;QAC5D,EAAE,EAAE,6FAA6F;KAClG;IACD,gCAAgC,EAAE;QAChC,EAAE,EAAE,sBAAsB;QAC1B,EAAE,EAAE,mCAAmC;KACxC;IACD,iCAAiC,EAAE;QACjC,EAAE,EAAE,sBAAsB;QAC1B,EAAE,EAAE,8BAA8B;KACnC;IACD,6BAA6B,EAAE;QAC7B,EAAE,EAAE,kBAAkB;QACtB,EAAE,EAAE,wBAAwB;KAC7B;IACD,2BAA2B,EAAE;QAC3B,EAAE,EAAE,+DAA+D;QACnE,EAAE,EAAE,+EAA+E;KACpF;IACD,gCAAgC,EAAE;QAChC,EAAE,EAAE,mCAAmC;QACvC,EAAE,EAAE,sCAAsC;KAC3C;IACD,qCAAqC,EAAE;QACrC,EAAE,EAAE,yCAAyC;QAC7C,EAAE,EAAE,qDAAqD;KAC1D;IACD,gCAAgC,EAAE;QAChC,EAAE,EAAE,oCAAoC;QACxC,EAAE,EAAE,qDAAqD;KAC1D;IACD,0BAA0B,EAAE;QAC1B,EAAE,EAAE,oCAAoC;QACxC,EAAE,EAAE,+CAA+C;KACpD;IACD,oBAAoB,EAAE;QACpB,EAAE,EAAE,wCAAwC;QAC5C,EAAE,EAAE,kDAAkD;KACvD;IACD,sBAAsB,EAAE;QACtB,EAAE,EAAE,uBAAuB;QAC3B,EAAE,EAAE,uBAAuB;KAC5B;IACD,2BAA2B,EAAE;QAC3B,EAAE,EAAE,yCAAyC;QAC7C,EAAE,EAAE,wEAAwE;KAC7E;IACD,uBAAuB,EAAE;QACvB,EAAE,EAAE,yDAAyD;QAC7D,EAAE,EAAE,kFAAkF;KACvF;IACD,4BAA4B,EAAE;QAC5B,EAAE,EAAE,8DAA8D;QAClE,EAAE,EAAE,kFAAkF;KACvF;IACD,gCAAgC,EAAE;QAChC,EAAE,EAAE,oCAAoC;QACxC,EAAE,EAAE,6EAA6E;KAClF;IACD,2BAA2B,EAAE;QAC3B,EAAE,EAAE,4BAA4B;QAChC,EAAE,EAAE,uCAAuC;KAC5C;IACD,qBAAqB,EAAE;QACrB,EAAE,EAAE,gCAAgC;QACpC,EAAE,EAAE,0CAA0C;KAC/C;IACD,qBAAqB,EAAE;QACrB,EAAE,EAAE,gCAAgC;QACpC,EAAE,EAAE,4DAA4D;KACjE;IACD,gBAAgB,EAAE;QAChB,EAAE,EAAE,iBAAiB;QACrB,EAAE,EAAE,8BAA8B;KACnC;IACD,+BAA+B,EAAE;QAC/B,EAAE,EAAE,yDAAyD;QAC7D,EAAE,EAAE,6EAA6E;KAClF;IACD,2BAA2B,EAAE;QAC3B,EAAE,EAAE,+BAA+B;QACnC,EAAE,EAAE,+BAA+B;KACpC;IACD,2BAA2B,EAAE;QAC3B,EAAE,EAAE,sCAAsC;QAC1C,EAAE,EAAE,8CAA8C;KACnD;IACD,wBAAwB,EAAE;QACxB,EAAE,EAAE,kCAAkC;QACtC,EAAE,EAAE,2DAA2D;KAChE;IACD,uBAAuB,EAAE;QACvB,EAAE,EAAE,0DAA0D;QAC9D,EAAE,EAAE,4DAA4D;KACjE;IACD,oBAAoB,EAAE;QACpB,EAAE,EAAE,+DAA+D;QACnE,EAAE,EAAE,kEAAkE;KACvE;IACD,sBAAsB,EAAE;QACtB,EAAE,EAAE,2BAA2B;QAC/B,EAAE,EAAE,2BAA2B;KAChC;IACD,2BAA2B,EAAE;QAC3B,EAAE,EAAE,oBAAoB;QACxB,EAAE,EAAE,iCAAiC;KACtC;IACD,wBAAwB,EAAE;QACxB,EAAE,EAAE,wCAAwC;QAC5C,EAAE,EAAE,0CAA0C;KAC/C;IACD,wBAAwB,EAAE;QACxB,EAAE,EAAE,sBAAsB;QAC1B,EAAE,EAAE,6BAA6B;KAClC;IACD,qCAAqC,EAAE;QACrC,EAAE,EAAE,+DAA+D;QACnE,EAAE,EAAE,wFAAwF;KAC7F;IACD,sBAAsB,EAAE;QACtB,EAAE,EAAE,+BAA+B;QACnC,EAAE,EAAE,qCAAqC;KAC1C;IACD,8BAA8B,EAAE;QAC9B,EAAE,EAAE,+DAA+D;QACnE,EAAE,EAAE,mGAAmG;KACxG;IACD,wBAAwB,EAAE;QACxB,EAAE,EAAE,mDAAmD;QACvD,EAAE,EAAE,uEAAuE;KAC5E;IACD,+BAA+B,EAAE;QAC/B,EAAE,EAAE,oCAAoC;QACxC,EAAE,EAAE,uDAAuD;KAC5D;IACD,iCAAiC,EAAE;QACjC,EAAE,EAAE,0BAA0B;QAC9B,EAAE,EAAE,4BAA4B;KACjC;IACD,8BAA8B,EAAE;QAC9B,EAAE,EAAE,cAAc;QAClB,EAAE,EAAE,cAAc;KACnB;IACD,4BAA4B,EAAE;QAC5B,EAAE,EAAE,yBAAyB;QAC7B,EAAE,EAAE,iCAAiC;KACtC;IACD,2BAA2B,EAAE;QAC3B,EAAE,EAAE,uCAAuC;QAC3C,EAAE,EAAE,6CAA6C;KAClD;IACD,kCAAkC,EAAE;QAClC,EAAE,EAAE,2CAA2C;QAC/C,EAAE,EAAE,uDAAuD;KAC5D;IACD,4CAA4C,EAAE;QAC5C,EAAE,EAAE,WAAW;QACf,EAAE,EAAE,iBAAiB;KACtB;IACD,8CAA8C,EAAE;QAC9C,EAAE,EAAE,KAAK;QACT,EAAE,EAAE,eAAe;KACpB;IACD,uCAAuC,EAAE;QACvC,EAAE,EAAE,kDAAkD;QACtD,EAAE,EAAE,6EAA6E;KAClF;IACD,eAAe,EAAE;QACf,EAAE,EAAE;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;CAiJP;QACG,EAAE,EAAE;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;CAwJP;KACE;IACD,qBAAqB,EAAE;QACrB,EAAE,EAAE;YACF,KAAK;YACL,0EAA0E;YAC1E,yEAAyE;YACzE,EAAE;YACF,KAAK;YACL,0DAA0D;YAC1D,0DAA0D;YAC1D,iFAAiF;YACjF,8CAA8C;SAC/C,CAAC,IAAI,CAAC,IAAI,CAAC;QACZ,EAAE,EAAE;YACF,QAAQ;YACR,+EAA+E;YAC/E,iFAAiF;YACjF,EAAE;YACF,UAAU;YACV,2EAA2E;YAC3E,uFAAuF;YACvF,iGAAiG;YACjG,wEAAwE;SACzE,CAAC,IAAI,CAAC,IAAI,CAAC;KACb;IACD,wBAAwB,EAAE;QACxB,EAAE,EAAE,gGAAgG;QACpG,EAAE,EAAE,mGAAmG;KACxG;IACD,eAAe,EAAE;QACf,EAAE,EAAE;YACF,EAAE;YACF,iCAAiC;YACjC,EAAE;YACF,MAAM;YACN,iEAAiE;YACjE,oDAAoD;YACpD,0EAA0E;YAC1E,4CAA4C;YAC5C,wCAAwC;YACxC,EAAE;SACH,CAAC,IAAI,CAAC,IAAI,CAAC;QACZ,EAAE,EAAE;YACF,EAAE;YACF,oCAAoC;YACpC,EAAE;YACF,cAAc;YACd,iFAAiF;YACjF,uEAAuE;YACvE,+FAA+F;YAC/F,4CAA4C;YAC5C,wCAAwC;YACxC,EAAE;SACH,CAAC,IAAI,CAAC,IAAI,CAAC;KACb;IACD,0BAA0B,EAAE;QAC1B,EAAE,EAAE;YACF,EAAE;YACF,2DAA2D;YAC3D,EAAE;YACF,KAAK;YACL,kDAAkD;YAClD,+BAA+B;YAC/B,EAAE;YACF,KAAK;YACL,wEAAwE;YACxE,qEAAqE;YACrE,uDAAuD;YACvD,qDAAqD;YACrD,wDAAwD;YACxD,yDAAyD;YACzD,2CAA2C;YAC3C,EAAE;SACH,CAAC,IAAI,CAAC,IAAI,CAAC;QACZ,EAAE,EAAE;YACF,EAAE;YACF,8DAA8D;YAC9D,EAAE;YACF,QAAQ;YACR,+EAA+E;YAC/E,0EAA0E;YAC1E,EAAE;YACF,UAAU;YACV,uFAAuF;YACvF,0FAA0F;YAC1F,2EAA2E;YAC3E,2EAA2E;YAC3E,sEAAsE;YACtE,oEAAoE;YACpE,sDAAsD;YACtD,EAAE;SACH,CAAC,IAAI,CAAC,IAAI,CAAC;KACb;IACD,qBAAqB,EAAE;QACrB,EAAE,EAAE;YACF,EAAE;YACF,+CAA+C;YAC/C,EAAE;YACF,sCAAsC;YACtC,EAAE;YACF,4DAA4D;YAC5D,kDAAkD;YAClD,wDAAwD;YACxD,cAAc;YACd,EAAE;YACF,KAAK;YACL,iEAAiE;YACjE,+CAA+C;YAC/C,EAAE;SACH,CAAC,IAAI,CAAC,IAAI,CAAC;QACZ,EAAE,EAAE;YACF,EAAE;YACF,kDAAkD;YAClD,EAAE;YACF,qEAAqE;YACrE,kCAAkC;YAClC,EAAE;YACF,uEAAuE;YACvE,uEAAuE;YACvE,qEAAqE;YACrE,wEAAwE;YACxE,8DAA8D;YAC9D,EAAE;YACF,UAAU;YACV,gFAAgF;YAChF,8DAA8D;YAC9D,EAAE;SACH,CAAC,IAAI,CAAC,IAAI,CAAC;KACb;IACD,kBAAkB,EAAE;QAClB,EAAE,EAAE;YACF,EAAE;YACF,4CAA4C;YAC5C,EAAE;YACF,8DAA8D;YAC9D,EAAE;YACF,aAAa;YACb,6BAA6B;YAC7B,mDAAmD;YACnD,gCAAgC;YAChC,8BAA8B;YAC9B,gCAAgC;YAChC,sCAAsC;YACtC,EAAE;YACF,KAAK;YACL,oEAAoE;YACpE,qEAAqE;YACrE,+CAA+C;YAC/C,0CAA0C;YAC1C,EAAE;SACH,CAAC,IAAI,CAAC,IAAI,CAAC;QACZ,EAAE,EAAE;YACF,EAAE;YACF,+CAA+C;YAC/C,EAAE;YACF,2FAA2F;YAC3F,EAAE;YACF,iBAAiB;YACjB,6DAA6D;YAC7D,sGAAsG;YACtG,sDAAsD;YACtD,0CAA0C;YAC1C,yDAAyD;YACzD,mEAAmE;YACnE,EAAE;YACF,UAAU;YACV,mFAAmF;YACnF,yFAAyF;YACzF,qEAAqE;YACrE,oDAAoD;YACpD,EAAE;SACH,CAAC,IAAI,CAAC,IAAI,CAAC;KACb;IACD,mBAAmB,EAAE;QACnB,EAAE,EAAE;YACF,EAAE;YACF,6CAA6C;YAC7C,EAAE;YACF,+CAA+C;YAC/C,6CAA6C;YAC7C,EAAE;YACF,KAAK;YACL,mCAAmC;YACnC,qFAAqF;YACrF,qDAAqD;YACrD,+DAA+D;YAC/D,kFAAkF;YAClF,iDAAiD;YACjD,iDAAiD;YACjD,sDAAsD;YACtD,EAAE;SACH,CAAC,IAAI,CAAC,IAAI,CAAC;QACZ,EAAE,EAAE;YACF,EAAE;YACF,gDAAgD;YAChD,EAAE;YACF,wEAAwE;YACxE,6EAA6E;YAC7E,2DAA2D;YAC3D,EAAE;YACF,UAAU;YACV,6CAA6C;YAC7C,6GAA6G;YAC7G,mEAAmE;YACnE,2EAA2E;YAC3E,yGAAyG;YACzG,0EAA0E;YAC1E,6EAA6E;YAC7E,uEAAuE;YACvE,EAAE;SACH,CAAC,IAAI,CAAC,IAAI,CAAC;KACb;IACD,mBAAmB,EAAE;QACnB,EAAE,EAAE;YACF,EAAE;YACF,6CAA6C;YAC7C,EAAE;YACF,2CAA2C;YAC3C,8CAA8C;YAC9C,EAAE;YACF,KAAK;YACL,mCAAmC;YACnC,6CAA6C;YAC7C,mEAAmE;YACnE,0CAA0C;YAC1C,kEAAkE;YAClE,wDAAwD;YACxD,EAAE;SACH,CAAC,IAAI,CAAC,IAAI,CAAC;QACZ,EAAE,EAAE;YACF,EAAE;YACF,gDAAgD;YAChD,EAAE;YACF,iFAAiF;YACjF,gDAAgD;YAChD,6DAA6D;YAC7D,EAAE;YACF,UAAU;YACV,6CAA6C;YAC7C,uDAAuD;YACvD,wFAAwF;YACxF,+DAA+D;YAC/D,qFAAqF;YACrF,sFAAsF;YACtF,EAAE;SACH,CAAC,IAAI,CAAC,IAAI,CAAC;KACb;IACD,uCAAuC;IACvC,8BAA8B,EAAE;QAC9B,EAAE,EAAE,mCAAmC;QACvC,EAAE,EAAE,yBAAyB;KAC9B;IACD,mCAAmC,EAAE;QACnC,EAAE,EAAE,OAAO;QACX,EAAE,EAAE,eAAe;KACpB;IACD,6BAA6B,EAAE;QAC7B,EAAE,EAAE,IAAI;QACR,EAAE,EAAE,OAAO;KACZ;IACD,kCAAkC,EAAE;QAClC,EAAE,EAAE,gGAAgG;QACpG,EAAE,EAAE,4HAA4H;KACjI;IACD,gCAAgC,EAAE;QAChC,EAAE,EAAE,IAAI;QACR,EAAE,EAAE,UAAU;KACf;CACF,CAAC"}
|
package/dist/src/cli/index.js
CHANGED
|
@@ -7,6 +7,19 @@ import { tCli, getCliLang, parseLangFromArgv, langFromArgv } from './i18n.js';
|
|
|
7
7
|
import { parseRunConfig, DEFAULT_REPORTS_DIR, COMMON_OPTIONS, } from './parse-run-config.js';
|
|
8
8
|
import { makeOnProgress } from './progress.js';
|
|
9
9
|
import { checkUpdate } from './update-check.js';
|
|
10
|
+
function requireEvaluationReport(report, id, lang) {
|
|
11
|
+
if (!report) {
|
|
12
|
+
console.error(tCli('cli.common.report_not_found', lang, { id }));
|
|
13
|
+
process.exit(1);
|
|
14
|
+
}
|
|
15
|
+
if (report.kind === 'batch-evaluation') {
|
|
16
|
+
console.error(lang === 'zh'
|
|
17
|
+
? `报告 ${id} 是 BatchEvaluationReport。该命令需要单次 EvaluationReport;请使用其中的 child reportId。`
|
|
18
|
+
: `Report ${id} is a BatchEvaluationReport. This command requires an EvaluationReport; use a child reportId from the batch.`);
|
|
19
|
+
process.exit(1);
|
|
20
|
+
}
|
|
21
|
+
return report;
|
|
22
|
+
}
|
|
10
23
|
// ---------------------------------------------------------------------------
|
|
11
24
|
// Main
|
|
12
25
|
// ---------------------------------------------------------------------------
|
|
@@ -88,6 +101,10 @@ async function main() {
|
|
|
88
101
|
// ---------------------------------------------------------------------------
|
|
89
102
|
async function handleRun(argv) {
|
|
90
103
|
const lang = langFromArgv(argv);
|
|
104
|
+
if (argv.includes('--help') || argv.includes('-h')) {
|
|
105
|
+
console.log(tCli('cli.help.main', lang).trim());
|
|
106
|
+
process.exit(0);
|
|
107
|
+
}
|
|
91
108
|
const { values, config } = parseRunConfig(argv, {
|
|
92
109
|
blind: { type: 'boolean' },
|
|
93
110
|
repeat: { type: 'string', default: '1' },
|
|
@@ -101,13 +118,13 @@ async function handleRun(argv) {
|
|
|
101
118
|
'budget-per-sample-usd': { type: 'string' },
|
|
102
119
|
'budget-per-sample-ms': { type: 'string' },
|
|
103
120
|
});
|
|
104
|
-
const { runEvaluation, runMultiple,
|
|
121
|
+
const { runEvaluation, runMultiple, runBatchEvaluation } = await import('../eval-workflows/run-evaluation.js');
|
|
105
122
|
if (values.blind !== undefined) {
|
|
106
123
|
config.blind = values.blind;
|
|
107
124
|
}
|
|
108
125
|
config.onProgress = makeOnProgress(lang);
|
|
109
126
|
// --repeat 输入校验: 非 ≥1 整数时提示并钳到 1, 不静默掩盖用户错字 / 极端输入。
|
|
110
|
-
// 提前到 --
|
|
127
|
+
// 提前到 --batch 分支之前, 保证 batch 模式也能读到 repeat。
|
|
111
128
|
const repeatRaw = values.repeat;
|
|
112
129
|
const parsedRepeat = repeatRaw !== undefined ? Number(repeatRaw) : 1;
|
|
113
130
|
if (repeatRaw !== undefined && (!Number.isFinite(parsedRepeat) || parsedRepeat < 1)) {
|
|
@@ -182,9 +199,9 @@ async function handleRun(argv) {
|
|
|
182
199
|
config.bootstrapSamples = bsCount;
|
|
183
200
|
}
|
|
184
201
|
try {
|
|
185
|
-
// --
|
|
186
|
-
if (values.
|
|
187
|
-
const { report, filePath } = await
|
|
202
|
+
// --batch mode: evaluate each skill independently
|
|
203
|
+
if (values.batch) {
|
|
204
|
+
const { report, filePath } = await runBatchEvaluation({
|
|
188
205
|
...config,
|
|
189
206
|
repeat: repeatCount,
|
|
190
207
|
onSkillProgress({ phase, skill, current, total }) {
|
|
@@ -299,6 +316,10 @@ async function handleRun(argv) {
|
|
|
299
316
|
// ---------------------------------------------------------------------------
|
|
300
317
|
async function handleReport(argv) {
|
|
301
318
|
const lang = langFromArgv(argv);
|
|
319
|
+
if (argv.includes('--help') || argv.includes('-h')) {
|
|
320
|
+
console.log(tCli('cli.help.main', lang).trim());
|
|
321
|
+
process.exit(0);
|
|
322
|
+
}
|
|
302
323
|
const { values } = parseArgs({
|
|
303
324
|
args: argv,
|
|
304
325
|
options: {
|
|
@@ -330,7 +351,7 @@ async function handleReport(argv) {
|
|
|
330
351
|
}
|
|
331
352
|
if (values.export) {
|
|
332
353
|
const { createFileStore } = await import('../server/report-store.js');
|
|
333
|
-
const {
|
|
354
|
+
const { renderReportDocumentDetail } = await import('../renderer/html-renderer.js');
|
|
334
355
|
const { writeFileSync } = await import('node:fs');
|
|
335
356
|
const store = createFileStore(resolve(values['reports-dir']));
|
|
336
357
|
const report = await store.get(values.export);
|
|
@@ -338,7 +359,7 @@ async function handleReport(argv) {
|
|
|
338
359
|
console.error(tCli('cli.common.report_not_found', lang, { id: values.export }));
|
|
339
360
|
process.exit(1);
|
|
340
361
|
}
|
|
341
|
-
const html =
|
|
362
|
+
const html = renderReportDocumentDetail(report);
|
|
342
363
|
const outPath = resolve(`${values.export}.html`);
|
|
343
364
|
writeFileSync(outPath, html);
|
|
344
365
|
console.log(`Exported to: ${outPath}`);
|
|
@@ -517,11 +538,15 @@ async function handleInit(argv) {
|
|
|
517
538
|
// ---------------------------------------------------------------------------
|
|
518
539
|
async function handleGenSamples(argv) {
|
|
519
540
|
const lang = langFromArgv(argv);
|
|
541
|
+
if (argv.includes('--help') || argv.includes('-h')) {
|
|
542
|
+
console.log(tCli('cli.help.main', lang).trim());
|
|
543
|
+
process.exit(0);
|
|
544
|
+
}
|
|
520
545
|
const { values } = parseArgs({
|
|
521
546
|
args: argv,
|
|
522
547
|
options: {
|
|
523
548
|
...COMMON_OPTIONS,
|
|
524
|
-
|
|
549
|
+
batch: { type: 'boolean', default: false },
|
|
525
550
|
count: { type: 'string', default: '5' },
|
|
526
551
|
model: { type: 'string', default: 'sonnet' },
|
|
527
552
|
'skill-dir': { type: 'string', default: 'skills' },
|
|
@@ -533,7 +558,7 @@ async function handleGenSamples(argv) {
|
|
|
533
558
|
const { readFileSync, writeFileSync } = await import('node:fs');
|
|
534
559
|
const count = Math.max(1, Number(values.count) || 5);
|
|
535
560
|
const model = values.model;
|
|
536
|
-
if (values.
|
|
561
|
+
if (values.batch) {
|
|
537
562
|
// Batch mode: generate for all skills missing eval-samples
|
|
538
563
|
const skillDir = resolve(values['skill-dir']);
|
|
539
564
|
if (!existsSync(skillDir)) {
|
|
@@ -677,10 +702,13 @@ async function handleEvolve(argv) {
|
|
|
677
702
|
timeoutMs: Math.max(1, Number(values.timeout) || 120) * 1000,
|
|
678
703
|
skipPreflight: values['skip-preflight'],
|
|
679
704
|
onProgress: makeOnProgress(lang),
|
|
680
|
-
onRoundProgress({ round, totalRounds: _totalRounds, phase, score, delta, accepted, costUSD, error }) {
|
|
705
|
+
onRoundProgress({ round, totalRounds: _totalRounds, phase, score, delta, accepted, costUSD, costReported, error }) {
|
|
706
|
+
// costReported=false 时显示「—」而不是 $0.0000(executor 不报 cost,如 codex)。
|
|
707
|
+
// 缺位 / true 当 reported 走旧格式。
|
|
708
|
+
const fmtRoundCost = (c, r) => r ? `$${c.toFixed(4)}` : '—';
|
|
681
709
|
if (phase === 'baseline') {
|
|
682
710
|
process.stderr.write(tCli('cli.evolve.round_baseline', lang, {
|
|
683
|
-
score: score.toFixed(2), cost: costUSD
|
|
711
|
+
score: score.toFixed(2), cost: fmtRoundCost(costUSD, costReported !== false),
|
|
684
712
|
}));
|
|
685
713
|
}
|
|
686
714
|
else if (phase === 'error') {
|
|
@@ -692,7 +720,7 @@ async function handleEvolve(argv) {
|
|
|
692
720
|
const delta_ = delta >= 0 ? `+${delta.toFixed(2)}` : delta.toFixed(2);
|
|
693
721
|
const status = accepted ? '✓ ACCEPT' : '✗ REJECT';
|
|
694
722
|
process.stderr.write(tCli('cli.evolve.round_done', lang, {
|
|
695
|
-
round, score: score.toFixed(2), delta: delta_, status, cost: costUSD
|
|
723
|
+
round, score: score.toFixed(2), delta: delta_, status, cost: fmtRoundCost(costUSD, costReported !== false),
|
|
696
724
|
}));
|
|
697
725
|
}
|
|
698
726
|
},
|
|
@@ -700,9 +728,12 @@ async function handleEvolve(argv) {
|
|
|
700
728
|
const improvement = result.startScore > 0
|
|
701
729
|
? ((result.finalScore - result.startScore) / result.startScore * 100).toFixed(1)
|
|
702
730
|
: '0';
|
|
731
|
+
const totalCostStr = result.costReported === false
|
|
732
|
+
? '—' // 任一轮的 executor 不报 cost → totalCostUSD 是 lower-bound
|
|
733
|
+
: `$${result.totalCostUSD.toFixed(4)}`;
|
|
703
734
|
process.stderr.write(tCli('cli.evolve.summary', lang, {
|
|
704
735
|
start: result.startScore.toFixed(2), final: result.finalScore.toFixed(2),
|
|
705
|
-
percent: improvement, rounds: result.totalRounds, cost:
|
|
736
|
+
percent: improvement, rounds: result.totalRounds, cost: totalCostStr,
|
|
706
737
|
}));
|
|
707
738
|
process.stderr.write(tCli('cli.evolve.best_path', lang, {
|
|
708
739
|
best: result.bestSkillPath, target: resolve(skillPath),
|
|
@@ -738,11 +769,12 @@ async function handleGate(argv) {
|
|
|
738
769
|
const { runEvaluation } = await import('../eval-workflows/run-evaluation.js');
|
|
739
770
|
config.onProgress = makeOnProgress(lang);
|
|
740
771
|
try {
|
|
741
|
-
const { report } = (await runEvaluation(config));
|
|
742
|
-
if (
|
|
772
|
+
const { report: document } = (await runEvaluation(config));
|
|
773
|
+
if (document.dryRun) {
|
|
743
774
|
console.log('Gate dry-run: no scores to check');
|
|
744
775
|
process.exit(0);
|
|
745
776
|
}
|
|
777
|
+
const report = requireEvaluationReport(document, 'current run', lang);
|
|
746
778
|
// gate 内核 = run + verdict, 自动覆盖 omk 全部决策维度(三层 layer-gate /
|
|
747
779
|
// bootstrap diff CI / saturation / Krippendorff α)。computeVerdict 是单一
|
|
748
780
|
// 决策源, exit code 跟 verdict.level 走 — 数据 underpowered 直接 FAIL,
|
|
@@ -816,16 +848,8 @@ async function handleDiff(argv) {
|
|
|
816
848
|
return;
|
|
817
849
|
}
|
|
818
850
|
const [id1, id2] = positional;
|
|
819
|
-
const r1 = await store.get(id1);
|
|
820
|
-
const r2 = await store.get(id2);
|
|
821
|
-
if (!r1) {
|
|
822
|
-
console.error(tCli('cli.common.report_not_found', lang, { id: id1 }));
|
|
823
|
-
process.exit(1);
|
|
824
|
-
}
|
|
825
|
-
if (!r2) {
|
|
826
|
-
console.error(tCli('cli.common.report_not_found', lang, { id: id2 }));
|
|
827
|
-
process.exit(1);
|
|
828
|
-
}
|
|
851
|
+
const r1 = requireEvaluationReport(await store.get(id1), id1, lang);
|
|
852
|
+
const r2 = requireEvaluationReport(await store.get(id2), id2, lang);
|
|
829
853
|
console.log(`\n Diff: ${id1} → ${id2}\n`);
|
|
830
854
|
// Git info — r1/r2 are guaranteed non-null after process.exit() guards above
|
|
831
855
|
const g1 = r1.meta?.gitInfo;
|
|
@@ -833,6 +857,10 @@ async function handleDiff(argv) {
|
|
|
833
857
|
if (g1 || g2) {
|
|
834
858
|
console.log(` Git: ${g1?.commitShort || '?'}${g1?.dirty ? '*' : ''} (${g1?.branch || '?'}) → ${g2?.commitShort || '?'}${g2?.dirty ? '*' : ''} (${g2?.branch || '?'})`);
|
|
835
859
|
}
|
|
860
|
+
const { crossReportComparabilityWarnings, formatComparabilityWarnings } = await import('../eval-core/comparability.js');
|
|
861
|
+
const comparability = formatComparabilityWarnings(crossReportComparabilityWarnings(r1, r2), lang);
|
|
862
|
+
if (comparability)
|
|
863
|
+
process.stderr.write(`\n${comparability}\n\n`);
|
|
836
864
|
// Per-variant comparison
|
|
837
865
|
const variants = [...new Set([...(r1.meta?.variants || []), ...(r2.meta?.variants || [])])];
|
|
838
866
|
for (const v of variants) {
|
|
@@ -861,8 +889,14 @@ async function handleDiff(argv) {
|
|
|
861
889
|
}
|
|
862
890
|
const cost1 = s1?.avgCostPerSample ?? 0;
|
|
863
891
|
const cost2 = s2?.avgCostPerSample ?? 0;
|
|
864
|
-
const
|
|
865
|
-
|
|
892
|
+
const reported1 = s1?.execCostReported !== false;
|
|
893
|
+
const reported2 = s2?.execCostReported !== false;
|
|
894
|
+
const fmt = (c, r) => r ? `$${c.toFixed(4)}` : '—';
|
|
895
|
+
// 任一边 not reported 就不报增减百分比(没意义)
|
|
896
|
+
const costPct = (reported1 && reported2 && cost1 > 0)
|
|
897
|
+
? ` (${cost2 > cost1 ? '+' : ''}${(((cost2 - cost1) / cost1) * 100).toFixed(0)}%)`
|
|
898
|
+
: '';
|
|
899
|
+
console.log(` Cost: ${fmt(cost1, reported1)} → ${fmt(cost2, reported2)}${costPct}`);
|
|
866
900
|
// Skill hash change
|
|
867
901
|
const h1 = r1.meta?.artifactHashes?.[v];
|
|
868
902
|
const h2 = r2.meta?.artifactHashes?.[v];
|
|
@@ -880,11 +914,11 @@ async function handleDiff(argv) {
|
|
|
880
914
|
* `--variant` overrides which variant is the "treatment" side.
|
|
881
915
|
*/
|
|
882
916
|
async function runSampleLevelDiff(reportId, store, flags, lang) {
|
|
883
|
-
const report = await store.get(reportId);
|
|
884
|
-
|
|
885
|
-
|
|
886
|
-
|
|
887
|
-
|
|
917
|
+
const report = requireEvaluationReport(await store.get(reportId), reportId, lang);
|
|
918
|
+
const { reportComparabilityWarnings, formatComparabilityWarnings } = await import('../eval-core/comparability.js');
|
|
919
|
+
const comparability = formatComparabilityWarnings(reportComparabilityWarnings(report), lang);
|
|
920
|
+
if (comparability)
|
|
921
|
+
process.stderr.write(`\n${comparability}\n\n`);
|
|
888
922
|
const variants = report.meta?.variants ?? [];
|
|
889
923
|
if (variants.length < 2) {
|
|
890
924
|
console.error('Sample-level diff needs at least 2 variants in the report.');
|
|
@@ -1050,15 +1084,11 @@ async function handleGold(argv) {
|
|
|
1050
1084
|
console.error(`warn: ${i.message}`);
|
|
1051
1085
|
}
|
|
1052
1086
|
const store = createFileStore(resolve(values['reports-dir']));
|
|
1053
|
-
const report = await store.get(reportId);
|
|
1054
|
-
if (!report) {
|
|
1055
|
-
console.error(tCli('cli.common.report_not_found', lang, { id: reportId }));
|
|
1056
|
-
process.exit(1);
|
|
1057
|
-
}
|
|
1087
|
+
const report = requireEvaluationReport(await store.get(reportId), reportId, lang);
|
|
1058
1088
|
const samples = Math.max(100, Number(values['bootstrap-samples']) || 1000);
|
|
1059
1089
|
const seedVal = values.seed != null ? Number(values.seed) : undefined;
|
|
1060
1090
|
const result = compareGoldToReport({
|
|
1061
|
-
report
|
|
1091
|
+
report,
|
|
1062
1092
|
gold: dataset,
|
|
1063
1093
|
variant: values.variant,
|
|
1064
1094
|
samples,
|
|
@@ -1106,11 +1136,7 @@ async function handleDebiasValidate(argv) {
|
|
|
1106
1136
|
});
|
|
1107
1137
|
const { createFileStore } = await import('../server/report-store.js');
|
|
1108
1138
|
const store = createFileStore(resolve(values['reports-dir']));
|
|
1109
|
-
const report = await store.get(reportId);
|
|
1110
|
-
if (!report) {
|
|
1111
|
-
console.error(tCli('cli.common.report_not_found', lang, { id: reportId }));
|
|
1112
|
-
process.exit(1);
|
|
1113
|
-
}
|
|
1139
|
+
const report = requireEvaluationReport(await store.get(reportId), reportId, lang);
|
|
1114
1140
|
// Resolve samples path: --samples overrides; otherwise read from report.meta.request.
|
|
1115
1141
|
const samplesPath = values.samples
|
|
1116
1142
|
?? report.meta?.request?.samplesPath;
|
|
@@ -1133,7 +1159,7 @@ async function handleDebiasValidate(argv) {
|
|
|
1133
1159
|
const seedVal = values.seed != null ? Number(values.seed) : undefined;
|
|
1134
1160
|
const bsRaw = Number(values['bootstrap-samples']) || 1000;
|
|
1135
1161
|
const result = await validateLengthDebias({
|
|
1136
|
-
report
|
|
1162
|
+
report,
|
|
1137
1163
|
samples,
|
|
1138
1164
|
judgeExecutor,
|
|
1139
1165
|
judgeModel,
|
|
@@ -1167,11 +1193,7 @@ async function handleSaturation(argv) {
|
|
|
1167
1193
|
});
|
|
1168
1194
|
const { createFileStore } = await import('../server/report-store.js');
|
|
1169
1195
|
const store = createFileStore(resolve(values['reports-dir']));
|
|
1170
|
-
const report = await store.get(reportId);
|
|
1171
|
-
if (!report) {
|
|
1172
|
-
console.error(tCli('cli.common.report_not_found', lang, { id: reportId }));
|
|
1173
|
-
process.exit(1);
|
|
1174
|
-
}
|
|
1196
|
+
const report = requireEvaluationReport(await store.get(reportId), reportId, lang);
|
|
1175
1197
|
const saturation = report.variance?.saturation;
|
|
1176
1198
|
if (!saturation) {
|
|
1177
1199
|
console.error(tCli('cli.saturation.no_data', lang));
|
|
@@ -1238,12 +1260,12 @@ async function handleVerdict(argv) {
|
|
|
1238
1260
|
});
|
|
1239
1261
|
const { createFileStore } = await import('../server/report-store.js');
|
|
1240
1262
|
const store = createFileStore(resolve(values['reports-dir']));
|
|
1241
|
-
const report = await store.get(reportId);
|
|
1242
|
-
if (!report) {
|
|
1243
|
-
console.error(tCli('cli.common.report_not_found', lang, { id: reportId }));
|
|
1244
|
-
process.exit(1);
|
|
1245
|
-
}
|
|
1263
|
+
const report = requireEvaluationReport(await store.get(reportId), reportId, lang);
|
|
1246
1264
|
const { computeVerdict, formatVerdictText } = await import('../eval-core/verdict.js');
|
|
1265
|
+
const { reportComparabilityWarnings, formatComparabilityWarnings } = await import('../eval-core/comparability.js');
|
|
1266
|
+
const comparability = formatComparabilityWarnings(reportComparabilityWarnings(report), lang);
|
|
1267
|
+
if (comparability)
|
|
1268
|
+
process.stderr.write(`${comparability}\n`);
|
|
1247
1269
|
const result = computeVerdict(report, {
|
|
1248
1270
|
gateThreshold: values.threshold != null ? Number(values.threshold) : undefined,
|
|
1249
1271
|
triviallySmallDiff: values['trivial-diff'] != null ? Number(values['trivial-diff']) : undefined,
|
|
@@ -1287,11 +1309,7 @@ async function handleDiagnose(argv) {
|
|
|
1287
1309
|
});
|
|
1288
1310
|
const { createFileStore } = await import('../server/report-store.js');
|
|
1289
1311
|
const store = createFileStore(resolve(values['reports-dir']));
|
|
1290
|
-
const report = await store.get(reportId);
|
|
1291
|
-
if (!report) {
|
|
1292
|
-
console.error(tCli('cli.common.report_not_found', lang, { id: reportId }));
|
|
1293
|
-
process.exit(1);
|
|
1294
|
-
}
|
|
1312
|
+
const report = requireEvaluationReport(await store.get(reportId), reportId, lang);
|
|
1295
1313
|
// Try to read the samples file for near-duplicate detection. Source order:
|
|
1296
1314
|
// 1. --samples <path> override
|
|
1297
1315
|
// 2. report.meta.request.samplesPath (recorded at run time)
|
|
@@ -1320,7 +1338,7 @@ async function handleDiagnose(argv) {
|
|
|
1320
1338
|
latencyOutlierK: values['latency-k'] != null ? Number(values['latency-k']) : undefined,
|
|
1321
1339
|
flatThreshold: values.flat != null ? Number(values.flat) : undefined,
|
|
1322
1340
|
});
|
|
1323
|
-
console.log(formatSampleDiagnostics(diag, { topN }));
|
|
1341
|
+
console.log(formatSampleDiagnostics(diag, { topN, lang }));
|
|
1324
1342
|
// Sample design science coverage block. Render after diagnose 主体,因为
|
|
1325
1343
|
// coverage 是声明式元数据(capability/difficulty/construct/provenance)的整体分布,
|
|
1326
1344
|
// 跟 issue list 是不同视角的两件事。优先从 samples (现场加载) 算,fallback 到
|
|
@@ -1360,11 +1378,7 @@ async function handleFailures(argv) {
|
|
|
1360
1378
|
});
|
|
1361
1379
|
const { createFileStore } = await import('../server/report-store.js');
|
|
1362
1380
|
const store = createFileStore(resolve(values['reports-dir']));
|
|
1363
|
-
const report = await store.get(reportId);
|
|
1364
|
-
if (!report) {
|
|
1365
|
-
console.error(tCli('cli.common.report_not_found', lang, { id: reportId }));
|
|
1366
|
-
process.exit(1);
|
|
1367
|
-
}
|
|
1381
|
+
const report = requireEvaluationReport(await store.get(reportId), reportId, lang);
|
|
1368
1382
|
const judgeModel = values['judge-model'] ?? report.meta?.judgeModel;
|
|
1369
1383
|
if (!judgeModel) {
|
|
1370
1384
|
console.error(tCli('cli.common.no_judge_model', lang));
|
|
@@ -1374,7 +1388,7 @@ async function handleFailures(argv) {
|
|
|
1374
1388
|
const executor = createExecutor(values['judge-executor']);
|
|
1375
1389
|
const { clusterFailures, formatFailureClusterReport } = await import('../analysis/failure-clusterer.js');
|
|
1376
1390
|
const out = await clusterFailures({
|
|
1377
|
-
report
|
|
1391
|
+
report,
|
|
1378
1392
|
executor,
|
|
1379
1393
|
judgeModel,
|
|
1380
1394
|
maxClusters: Number(values['max-clusters']) || 5,
|