oh-my-knowledge 0.40.0 → 0.42.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (71) hide show
  1. package/README.md +8 -4
  2. package/README.zh.md +8 -4
  3. package/dist/analysis/report-diagnostics.d.ts +8 -1
  4. package/dist/analysis/report-diagnostics.js +109 -1
  5. package/dist/assets/agent-skills/omk/references/commands.md +2 -2
  6. package/dist/authoring/evolver.d.ts +3 -14
  7. package/dist/authoring/evolver.js +1 -52
  8. package/dist/authoring/generator.d.ts +24 -0
  9. package/dist/authoring/generator.js +36 -8
  10. package/dist/cli/commands/eval/index.d.ts +1 -1
  11. package/dist/cli/commands/eval/index.js +50 -15
  12. package/dist/cli/commands/init.js +10 -7
  13. package/dist/cli/lib/cmd-flags.d.ts +0 -1
  14. package/dist/cli/lib/i18n-dict/init.d.ts +1 -1
  15. package/dist/cli/lib/i18n-dict/init.js +14 -11
  16. package/dist/cli/lib/i18n-dict/run.d.ts +1 -1
  17. package/dist/cli/lib/i18n-dict/run.js +6 -2
  18. package/dist/cli/lib/parse-run-config.d.ts +3 -1
  19. package/dist/cli/lib/parse-run-config.js +0 -2
  20. package/dist/eval-core/evaluation-job.d.ts +2 -2
  21. package/dist/eval-core/evaluation-job.js +2 -2
  22. package/dist/eval-core/evaluation-reporting.d.ts +0 -1
  23. package/dist/eval-core/evaluation-reporting.js +9 -41
  24. package/dist/eval-core/execution-strategy.js +3 -2
  25. package/dist/eval-core/holdout.d.ts +66 -0
  26. package/dist/eval-core/holdout.js +118 -0
  27. package/dist/eval-core/judge-independence.d.ts +28 -0
  28. package/dist/eval-core/judge-independence.js +29 -0
  29. package/dist/eval-core/verdict.d.ts +53 -2
  30. package/dist/eval-core/verdict.js +216 -16
  31. package/dist/eval-workflows/batch-evaluation-workflow.d.ts +1 -1
  32. package/dist/eval-workflows/batch-evaluation-workflow.js +1 -2
  33. package/dist/eval-workflows/evaluation-pipeline/report-finalize.d.ts +1 -5
  34. package/dist/eval-workflows/evaluation-pipeline/report-finalize.js +19 -8
  35. package/dist/eval-workflows/evaluation-pipeline/run-state.d.ts +2 -2
  36. package/dist/eval-workflows/evaluation-pipeline/run-state.js +2 -2
  37. package/dist/eval-workflows/evaluation-pipeline.d.ts +3 -2
  38. package/dist/eval-workflows/evaluation-pipeline.js +3 -4
  39. package/dist/eval-workflows/run-evaluation.d.ts +6 -4
  40. package/dist/eval-workflows/run-evaluation.js +8 -6
  41. package/dist/executors/claude-cli.js +5 -6
  42. package/dist/executors/claude-sdk.d.ts +5 -2
  43. package/dist/executors/claude-sdk.js +13 -8
  44. package/dist/executors/codex-cli.js +3 -4
  45. package/dist/executors/shared.d.ts +2 -0
  46. package/dist/executors/shared.js +15 -0
  47. package/dist/grading/assertions.js +6 -122
  48. package/dist/grading/gold-cli.js +1 -1
  49. package/dist/grading/human-gold.d.ts +5 -3
  50. package/dist/grading/human-gold.js +5 -3
  51. package/dist/grading/index.d.ts +4 -4
  52. package/dist/grading/judge.d.ts +6 -14
  53. package/dist/grading/judge.js +5 -88
  54. package/dist/inputs/eval-config.js +12 -2
  55. package/dist/managed/evidence.js +1 -2
  56. package/dist/managed/version-scores.js +1 -1
  57. package/dist/renderer/html-renderer.js +0 -9
  58. package/dist/renderer/layout.js +4 -4
  59. package/dist/renderer/summary.js +59 -4
  60. package/dist/shared/llm-prompts/debias-instructions.d.ts +4 -0
  61. package/dist/shared/llm-prompts/debias-instructions.js +44 -0
  62. package/dist/shared/llm-prompts/judge-prompts.d.ts +30 -0
  63. package/dist/shared/llm-prompts/judge-prompts.js +205 -0
  64. package/dist/shared/llm-prompts/registry.d.ts +27 -0
  65. package/dist/shared/llm-prompts/registry.js +69 -0
  66. package/dist/types/eval.d.ts +15 -8
  67. package/dist/types/judge.d.ts +1 -1
  68. package/dist/types/report.d.ts +50 -3
  69. package/package.json +1 -1
  70. package/dist/grading/debias-validate.d.ts +0 -83
  71. package/dist/grading/debias-validate.js +0 -176
@@ -8,6 +8,7 @@ import { parseRunConfig } from '../../lib/parse-run-config.js';
8
8
  import { makeOnProgress } from '../../lib/progress.js';
9
9
  import { computeRunTally } from '../../lib/run-tally.js';
10
10
  import { DEFAULT_BOOTSTRAP_SAMPLES } from '../../../eval-core/bootstrap.js';
11
+ import { DEFAULT_GATE_THRESHOLD } from '../../../eval-core/verdict.js';
11
12
  import { EVALUATION_REPORT_SCHEMA_VERSION } from '../../../eval-core/evaluation-reporting.js';
12
13
  function isDryRunReport(report) {
13
14
  return Boolean(report && typeof report === 'object' && report.dryRun === true);
@@ -17,9 +18,11 @@ function isDryRunBatchReport(report) {
17
18
  }
18
19
  function verdictOptions(values) {
19
20
  const rawThreshold = values.threshold;
21
+ // 不传 --threshold 时返回 undefined,由 computeVerdict 应用 DEFAULT_GATE_THRESHOLD ——
22
+ // 避免在此再硬编码一份 3.5(单一来源在 verdict.ts)。
20
23
  const gateThreshold = rawThreshold !== undefined && Number.isFinite(Number(rawThreshold))
21
24
  ? Number(rawThreshold)
22
- : 3.5;
25
+ : undefined;
23
26
  const rawTrivial = values['trivial-diff'];
24
27
  const triviallySmallDiff = rawTrivial !== undefined && Number.isFinite(Number(rawTrivial))
25
28
  ? Number(rawTrivial)
@@ -38,10 +41,30 @@ function applyGateExitCode(code, values, lang) {
38
41
  process.stderr.write(tCli('cli.run.report_only_gate_skipped', lang));
39
42
  return 0;
40
43
  }
44
+ /**
45
+ * 完整 report JSON 是**机器输出**:重定向 / 管道(`omk eval > r.json`、`| jq`)时吐到 stdout 供下游消费。
46
+ * 交互式 TTY 下报告已存盘、(默认)还起了 report server,再刷上千行 JSON 只会把 verdict 淹没在屏幕外 ——
47
+ * 故只在非 TTY(stdout 被重定向 / 管道)时 dump。dry-run 的 JSON 是用户显式索取的产物,不走此门控。
48
+ */
49
+ function emitReportJson(report) {
50
+ if (!process.stdout.isTTY) {
51
+ console.log(JSON.stringify(report, null, 2));
52
+ }
53
+ }
54
+ /**
55
+ * 给人读的 verdict 文案:stdout 是 TTY 时进 stdout(交互终端没有 JSON,verdict 就是答案),
56
+ * 否则进 stderr —— 与 emitReportJson 配对,保证非 TTY 的 stdout 是**纯 report JSON**,
57
+ * `omk eval | jq` / `> report.json` 不会被末尾拼上的人类文案噎住(否则 JSON.parse 直接失败)。
58
+ */
59
+ function emitVerdictText(text) {
60
+ // 与 console.log 等价(对单个字符串 = write(text + '\n')),只切换目标流,逐字节保留既有文案。
61
+ const stream = process.stdout.isTTY ? process.stdout : process.stderr;
62
+ stream.write(text + '\n');
63
+ }
41
64
  async function emitEvaluationVerdict(report, values, lang) {
42
65
  const { computeVerdict, formatVerdictText } = await import('../../../eval-core/verdict.js');
43
66
  const result = computeVerdict(report, verdictOptions(values));
44
- console.log(formatVerdictText(result, { verbose: true }));
67
+ emitVerdictText(formatVerdictText(result, { verbose: true, lang }));
45
68
  await recordEvidenceSafely(report, result.level, values, lang);
46
69
  return verdictPasses(result.level, result.headline) ? 0 : 1;
47
70
  }
@@ -120,13 +143,13 @@ async function emitBatchVerdict(report, reportsDir, values, lang) {
120
143
  const status = lang === 'zh'
121
144
  ? (failed === 0 ? '通过' : '未通过')
122
145
  : (failed === 0 ? 'PASS' : 'FAIL');
123
- console.log(tCli('cli.run.batch_verdict_header', lang, {
146
+ emitVerdictText(tCli('cli.run.batch_verdict_header', lang, {
124
147
  status,
125
148
  passed,
126
149
  total: results.length,
127
150
  }));
128
151
  for (const result of results) {
129
- console.log(` ${result.verdict.level}: ${result.treatment} — ${result.verdict.headline}`);
152
+ emitVerdictText(` ${result.verdict.level}: ${result.treatment} — ${result.verdict.headline}`);
130
153
  }
131
154
  return failed === 0 ? 0 : 1;
132
155
  }
@@ -158,9 +181,6 @@ async function announceSavedReport({ report, filePath, reportsDir, values, lang,
158
181
  async function runEval(_args, flags, lang) {
159
182
  const { values, config, evalConfig } = parseRunConfig({ ...flags });
160
183
  const { runEvaluation, runMultiple, runBatchEvaluation } = await import('../../../eval-workflows/run-evaluation.js');
161
- if (values.blind !== undefined) {
162
- config.blind = values.blind;
163
- }
164
184
  config.onProgress = makeOnProgress(lang);
165
185
  const repeatRaw = values.repeat;
166
186
  const parsedRepeat = repeatRaw !== undefined ? Number(repeatRaw) : (evalConfig?.repeat ?? 1);
@@ -168,6 +188,13 @@ async function runEval(_args, flags, lang) {
168
188
  process.stderr.write(tCli('cli.run.invalid_repeat', lang, { value: repeatRaw }));
169
189
  }
170
190
  const repeatCount = Math.max(1, Math.floor(parsedRepeat) || 1);
191
+ const holdoutRatioRaw = values['holdout-ratio'];
192
+ const parsedHoldoutRatio = holdoutRatioRaw !== undefined ? Number(holdoutRatioRaw) : (evalConfig?.holdoutRatio ?? 0);
193
+ if (holdoutRatioRaw !== undefined && (!Number.isFinite(parsedHoldoutRatio) || parsedHoldoutRatio <= 0 || parsedHoldoutRatio >= 1)) {
194
+ process.stderr.write(tCli('cli.run.invalid_holdout_ratio', lang, { value: holdoutRatioRaw }));
195
+ }
196
+ if (parsedHoldoutRatio > 0 && parsedHoldoutRatio < 1)
197
+ config.holdoutRatio = parsedHoldoutRatio;
171
198
  const judgeRepeatRaw = values['judge-repeat'];
172
199
  const parsedJudgeRepeat = judgeRepeatRaw !== undefined ? Number(judgeRepeatRaw) : (evalConfig?.judgeRepeat ?? 1);
173
200
  if (judgeRepeatRaw !== undefined && (!Number.isFinite(parsedJudgeRepeat) || parsedJudgeRepeat < 1)) {
@@ -224,11 +251,12 @@ async function runEval(_args, flags, lang) {
224
251
  }
225
252
  },
226
253
  });
227
- console.log(JSON.stringify(report, null, 2));
228
254
  if (isDryRunBatchReport(report)) {
255
+ console.log(JSON.stringify(report, null, 2));
229
256
  console.log(tCli('cli.run.dry_run_no_scores', lang));
230
257
  throw new CliExit(0);
231
258
  }
259
+ emitReportJson(report);
232
260
  if (filePath) {
233
261
  await announceSavedReport({ report, filePath, reportsDir: config.outputDir, values, lang });
234
262
  }
@@ -285,7 +313,7 @@ async function runEval(_args, flags, lang) {
285
313
  }
286
314
  }
287
315
  }
288
- console.log(JSON.stringify(report, null, 2));
316
+ emitReportJson(report);
289
317
  if (filePath) {
290
318
  await announceSavedReport({ report, filePath, reportsDir: config.outputDir, values, lang });
291
319
  }
@@ -368,8 +396,8 @@ export default class Eval extends BaseCommand {
368
396
  }),
369
397
  'judge-models': Flags.string({
370
398
  description: bilingual({
371
- zh: '评委配置,格式 executor:model[,...],例 claude:haiku 或 claude:opus,openai:gpt-4o(≥ 2 个 = ensemble)。默认 <executor>:haiku。',
372
- en: 'Judge config: executor:model[,...]. e.g. claude:haiku or claude:opus,openai:gpt-4o (≥ 2 = ensemble). Default <executor>:haiku.',
399
+ zh: '评委配置,格式 executor:model[,...],例 claude:haiku 或 claude:opus,openai-api:gpt-4o(≥ 2 个 = ensemble)。默认 <executor>:haiku。',
400
+ en: 'Judge config: executor:model[,...]. e.g. claude:haiku or claude:opus,openai-api:gpt-4o (≥ 2 = ensemble). Default <executor>:haiku.',
373
401
  }),
374
402
  }),
375
403
  'output-dir': Flags.string({
@@ -453,13 +481,17 @@ export default class Eval extends BaseCommand {
453
481
  }),
454
482
  }),
455
483
  // ── eval-runner extra ──
456
- blind: Flags.boolean({
457
- description: bilingual({ zh: 'judge blind 模式', en: 'Blind judge mode' }),
458
- }),
459
484
  repeat: Flags.string({
460
485
  description: bilingual({ zh: '每个 sample 重复跑 N 次', en: 'Repeat each sample N times' }),
461
486
  parse: integerStringParser('--repeat', { min: 1 }),
462
487
  }),
488
+ 'holdout-ratio': Flags.string({
489
+ description: bilingual({
490
+ zh: '留出比例 0-1(如 0.3);切出 holdout 子集,对比 train/holdout 综合分检测过拟合',
491
+ en: 'Holdout fraction 0-1 (e.g. 0.3); splits a holdout subset, compares train/holdout composite to flag overfitting',
492
+ }),
493
+ parse: numberStringParser('--holdout-ratio', { min: 0, max: 1 }),
494
+ }),
463
495
  'judge-repeat': Flags.string({
464
496
  description: bilingual({ zh: '每个 dim 评 N 次', en: 'Judge each dim N times' }),
465
497
  parse: integerStringParser('--judge-repeat', { min: 1 }),
@@ -490,7 +522,10 @@ export default class Eval extends BaseCommand {
490
522
  parse: numberStringParser('--budget-per-sample-ms', { minExclusive: 0 }),
491
523
  }),
492
524
  threshold: Flags.string({
493
- description: bilingual({ zh: 'verdict 阈值,默认 3.5', en: 'Verdict threshold, default 3.5' }),
525
+ description: bilingual({
526
+ zh: `verdict 阈值,默认 ${DEFAULT_GATE_THRESHOLD}`,
527
+ en: `Verdict threshold, default ${DEFAULT_GATE_THRESHOLD}`,
528
+ }),
494
529
  parse: numberStringParser('--threshold'),
495
530
  }),
496
531
  'trivial-diff': Flags.string({
@@ -12,17 +12,21 @@ const INIT_OMK_GITIGNORE = `# omk 测量 bulk + doctor --fix 备份(项目本地
12
12
  /reports/
13
13
  /backups/
14
14
  `;
15
+ // 脚手架用例必须过 omk 自身的断言合规校验(load-samples.ts Rule A),否则新用户照
16
+ // 快速开始跑的第一条 omk eval 会直接硬报错。约束:contains / not_contains 的 value
17
+ // 只能是单个 ASCII token(长度 [2,40]、无内部空白、无 CJK);多词 / 中文语义匹配一律
18
+ // 走 rubric 交评委判;regex pattern 不能含 CJK。改这里前先跑 `omk eval --dry-run`
19
+ // (非 lenient 合规 oracle)与 test/cli/init-scaffold-conformance 回归测试。
15
20
  const INIT_SAMPLES = `[
16
21
  {
17
22
  "sample_id": "s001",
18
23
  "prompt": "审查以下代码",
19
24
  "context": "function authenticate(username, password) {\\n const query = \`SELECT * FROM users WHERE name='\${username}' AND pass='\${password}'\`;\\n return db.execute(query);\\n}",
20
- "rubric": "应识别 SQL 注入风险,建议使用参数化查询",
25
+ "rubric": "应识别 SQL 注入风险,建议使用参数化查询;不应把这段代码判为安全无问题。",
21
26
  "assertions": [
22
27
  { "type": "contains", "value": "SQL", "weight": 1 },
23
28
  { "type": "contains", "value": "injection", "weight": 1 },
24
- { "type": "regex", "pattern": "parameterized|prepared|placeholder|bind", "flags": "i", "weight": 0.5 },
25
- { "type": "not_contains", "value": "looks good", "weight": 0.5 }
29
+ { "type": "regex", "pattern": "parameterized|prepared|placeholder|bind", "flags": "i", "weight": 0.5 }
26
30
  ],
27
31
  "dimensions": {
28
32
  "security": "是否准确识别出 SQL 注入漏洞并说明其危害",
@@ -35,8 +39,7 @@ const INIT_SAMPLES = `[
35
39
  "context": "async function fetchData(url) {\\n const res = await fetch(url);\\n const data = await res.json();\\n return data;\\n}",
36
40
  "rubric": "应指出缺少错误处理(网络异常、非 JSON 响应、HTTP 错误状态码)",
37
41
  "assertions": [
38
- { "type": "contains", "value": "error handling", "weight": 1 },
39
- { "type": "regex", "pattern": "try[\\\\s\\\\S]*catch|exception|error", "flags": "i", "weight": 1 },
42
+ { "type": "regex", "pattern": "try[\\\\s\\\\S]*catch|catch|exception|error", "flags": "i", "weight": 1 },
40
43
  { "type": "contains", "value": "status", "weight": 0.5 }
41
44
  ],
42
45
  "dimensions": {
@@ -154,9 +157,9 @@ export default class Init extends BaseCommand {
154
157
  console.log(tCli('cli.init.scaffolded', lang, { dir: targetDir }));
155
158
  console.log('');
156
159
  console.log(tCli('cli.init.next_steps_title', lang));
157
- console.log(tCli('cli.init.next_step_edit_samples', lang));
158
- console.log(tCli('cli.init.next_step_edit_skills', lang));
159
160
  console.log(tCli('cli.init.next_step_run', lang));
161
+ console.log(tCli('cli.init.next_step_executor', lang));
162
+ console.log(tCli('cli.init.next_step_customize', lang));
160
163
  console.log(tCli('cli.init.note_codex_executor', lang));
161
164
  });
162
165
  }
@@ -96,7 +96,6 @@ export interface EvalFlags {
96
96
  'no-strict-baseline'?: boolean;
97
97
  effort?: string;
98
98
  'no-diagnostic'?: boolean;
99
- blind?: boolean;
100
99
  repeat?: string;
101
100
  'judge-repeat'?: string;
102
101
  bootstrap?: boolean;
@@ -1,3 +1,3 @@
1
1
  import type { CliMessage } from './types.js';
2
- export type InitMessageKey = 'cli.init.scaffolded' | 'cli.init.next_steps_title' | 'cli.init.next_step_edit_samples' | 'cli.init.next_step_edit_skills' | 'cli.init.next_step_run' | 'cli.init.note_codex_executor';
2
+ export type InitMessageKey = 'cli.init.scaffolded' | 'cli.init.next_steps_title' | 'cli.init.next_step_run' | 'cli.init.next_step_executor' | 'cli.init.next_step_customize' | 'cli.init.note_codex_executor';
3
3
  export declare const initDict: Record<InitMessageKey, CliMessage>;
@@ -1,23 +1,26 @@
1
1
  export const initDict = {
2
2
  'cli.init.scaffolded': {
3
- zh: '已初始化 omk 项目: {dir}',
3
+ zh: '已初始化 omk 项目:{dir}',
4
4
  en: 'omk project initialized at: {dir}',
5
5
  },
6
6
  'cli.init.next_steps_title': {
7
- zh: '下一步:',
7
+ zh: '下一步:',
8
8
  en: 'Next steps:',
9
9
  },
10
- 'cli.init.next_step_edit_samples': {
11
- zh: ' 1. 编辑 eval-samples.json,加入你要测的评测用例',
12
- en: ' 1. Edit eval-samples.json to add your test cases',
10
+ // 先让用户「无需改任何文件直接跑通」——脚手架的用例与 skill 本身可跑(已过合规校验),
11
+ // 跑出第一份报告是冷启动最该先发生的事;「换成你自己的」放到跑通之后。这也消除了
12
+ // 主 README「不用改任何文件」与旧 init「先编辑」的矛盾。
13
+ 'cli.init.next_step_run': {
14
+ zh: ' 1. 直接跑通(无需先改任何文件):omk eval --control code-review-v1 --treatment code-review-v2',
15
+ en: ' 1. Run it as-is (no edits needed): omk eval --control code-review-v1 --treatment code-review-v2',
13
16
  },
14
- 'cli.init.next_step_edit_skills': {
15
- zh: ' 2. 编辑 skills/code-review-v1/SKILL.md 和 skills/code-review-v2/SKILL.md, 为两个 skill 版本填入实际内容',
16
- en: ' 2. Edit skills/code-review-v1/SKILL.md and skills/code-review-v2/SKILL.md with your skill versions',
17
+ 'cli.init.next_step_executor': {
18
+ zh: ' 默认执行器与评委用 claude CLI,需先装好并登录;想换别的模型或离线跑(无需 API key)见文档「执行器」。',
19
+ en: ' The default executor and judge use the claude CLI (install and log in first); to use another model or run offline (no API key) see the Executors docs.',
17
20
  },
18
- 'cli.init.next_step_run': {
19
- zh: ' 3. 运行: omk eval --control code-review-v1 --treatment code-review-v2',
20
- en: ' 3. Run: omk eval --control code-review-v1 --treatment code-review-v2',
21
+ 'cli.init.next_step_customize': {
22
+ zh: ' 2. 跑通后,把 skills/code-review-v1/SKILL.md 和 skills/code-review-v2/SKILL.md 与 eval-samples.json 换成你自己的 skill 和用例',
23
+ en: ' 2. Once it runs, replace skills/code-review-v1/SKILL.md and skills/code-review-v2/SKILL.md and eval-samples.json with your own skills and cases',
21
24
  },
22
25
  'cli.init.note_codex_executor': {
23
26
  zh: '\n注: omk 评测时把 SKILL.md 整文(含 frontmatter)作为 system prompt 注入——跨 executor 一致(claude / codex / openai-api / gemini 都走同一条路径,不依赖任何 executor 的 native skill auto-discovery 或 Skill 工具机制)。frontmatter 在 prompt 头部对 model 行为无显著影响。\n模板带 Claude Code 兼容的 frontmatter(name + description)是为了让同一份 directory-skill 也能 deploy 到 Claude Code:把整个目录复制到 ~/.claude/skills/code-review-v1/(整目录,不是单个 SKILL.md),Claude SDK 才能识别。这是 omk 评测之外的 bonus,一份文件双向 dogfood。',
@@ -1,3 +1,3 @@
1
1
  import type { CliMessage } from './types.js';
2
- export type RunMessageKey = 'cli.progress.preflight_starting' | 'cli.progress.sample_retry' | 'cli.progress.sample_error' | 'cli.progress.sample_executing' | 'cli.progress.sample_exec_done' | 'cli.progress.output_preview' | 'cli.progress.judging' | 'cli.progress.judged' | 'cli.progress.skipped' | 'cli.progress.sample_done' | 'cli.progress.sample_failed_done' | 'cli.run.invalid_repeat' | 'cli.run.invalid_judge_repeat' | 'cli.run.no_debias_length_active' | 'cli.run.invalid_bootstrap_samples' | 'cli.run.bootstrap_samples_too_large' | 'cli.run.dry_run_no_scores' | 'cli.run.skill_section' | 'cli.run.run_section' | 'cli.run.batch_complete' | 'cli.run.batch_verdict_header' | 'cli.run.batch_child_report_missing' | 'cli.run.eval_complete' | 'cli.run.tally' | 'cli.run.report_saved' | 'cli.run.evidence_recorded' | 'cli.run.evidence_recorded_unbound' | 'cli.run.report_only_gate_skipped' | 'cli.run.report_server_running' | 'cli.run.report_server_view' | 'cli.run.report_server_stop' | 'cli.run.no_serve_in_non_tty' | 'cli.run.no_serve_view_hint' | 'cli.run.gold_load_failed' | 'cli.run.gold_load_issue' | 'cli.run.contamination_warning' | 'cli.run.skip_connectivity_warning';
2
+ export type RunMessageKey = 'cli.progress.preflight_starting' | 'cli.progress.sample_retry' | 'cli.progress.sample_error' | 'cli.progress.sample_executing' | 'cli.progress.sample_exec_done' | 'cli.progress.output_preview' | 'cli.progress.judging' | 'cli.progress.judged' | 'cli.progress.skipped' | 'cli.progress.sample_done' | 'cli.progress.sample_failed_done' | 'cli.run.invalid_repeat' | 'cli.run.invalid_holdout_ratio' | 'cli.run.invalid_judge_repeat' | 'cli.run.no_debias_length_active' | 'cli.run.invalid_bootstrap_samples' | 'cli.run.bootstrap_samples_too_large' | 'cli.run.dry_run_no_scores' | 'cli.run.skill_section' | 'cli.run.run_section' | 'cli.run.batch_complete' | 'cli.run.batch_verdict_header' | 'cli.run.batch_child_report_missing' | 'cli.run.eval_complete' | 'cli.run.tally' | 'cli.run.report_saved' | 'cli.run.evidence_recorded' | 'cli.run.evidence_recorded_unbound' | 'cli.run.report_only_gate_skipped' | 'cli.run.report_server_running' | 'cli.run.report_server_view' | 'cli.run.report_server_stop' | 'cli.run.no_serve_in_non_tty' | 'cli.run.no_serve_view_hint' | 'cli.run.gold_load_failed' | 'cli.run.gold_load_issue' | 'cli.run.contamination_warning' | 'cli.run.skip_connectivity_warning';
3
3
  export declare const runDict: Record<RunMessageKey, CliMessage>;
@@ -51,9 +51,13 @@ export const runDict = {
51
51
  zh: '⚠ --judge-repeat "{value}" 无效 (期望 ≥ 1 的整数), 已按 1 次 judge 执行\n',
52
52
  en: '⚠ --judge-repeat "{value}" is invalid (expected an integer ≥ 1), falling back to 1 judge call\n',
53
53
  },
54
+ 'cli.run.invalid_holdout_ratio': {
55
+ zh: '⚠ --holdout-ratio "{value}" 无效 (期望 0 到 1 之间的小数), 已忽略、不做 holdout 切分\n',
56
+ en: '⚠ --holdout-ratio "{value}" is invalid (expected a fraction in (0, 1)), ignored — no holdout split\n',
57
+ },
54
58
  'cli.run.no_debias_length_active': {
55
- zh: 'ℹ --no-debias-length 已生效: judge prompt 退回 v2-cot, 与 < v0.21 报告 hash 一致。\n',
56
- en: 'ℹ --no-debias-length is active: judge prompt reverts to v2-cot, matching < v0.21 report hashes.\n',
59
+ zh: 'ℹ --no-debias-length 已生效:judge prompt 去掉长度去偏指令(debias-off 变体),hash 与默认开启时不同。\n',
60
+ en: 'ℹ --no-debias-length is active: the judge prompt drops the length-debias instruction (debias-off variant); its hash differs from the default.\n',
57
61
  },
58
62
  'cli.run.invalid_bootstrap_samples': {
59
63
  zh: '⚠ --bootstrap-samples "{value}" 无效 (期望 ≥ 100 的整数), 已按 1000 执行\n',
@@ -42,10 +42,12 @@ export interface RunConfig {
42
42
  lang: 'zh' | 'en' | undefined;
43
43
  mcpConfig: string | undefined;
44
44
  verbose: boolean | undefined;
45
- blind?: boolean | undefined;
46
45
  retry?: number;
47
46
  resume?: string;
48
47
  layeredStats?: boolean;
48
+ /** --holdout-ratio R (0 < R < 1). Hold out a deterministic sample slice; report-finalize
49
+ * computes train vs holdout composite (report.analysis.holdout) for the overfitting gate. */
50
+ holdoutRatio?: number;
49
51
  /** --judge-repeat N. Calls LLM judge N times per (sample × dimension). Default 1. */
50
52
  judgeRepeat?: number;
51
53
  /** Unified judge config. Always non-empty; 1 entry = single judge, ≥ 2 = ensemble.
@@ -94,7 +94,6 @@ export function parseRunConfig(values) {
94
94
  const verbose = values.verbose ?? false;
95
95
  const retry = Math.max(0, Number(values.retry ?? 0) || 0);
96
96
  const resume = values.resume;
97
- const blind = values.blind ?? evalConfig?.blind ?? false;
98
97
  const layeredStats = values['layered-stats'] ?? false;
99
98
  // strict-baseline default true. Reconcile both flag forms with eval.yaml fallback.
100
99
  // Priority: --no-strict-baseline > --strict-baseline > eval.yaml strictBaseline > true。
@@ -144,7 +143,6 @@ export function parseRunConfig(values) {
144
143
  verbose,
145
144
  retry,
146
145
  resume,
147
- blind,
148
146
  layeredStats,
149
147
  budget: evalConfig?.budget,
150
148
  strictBaseline,
@@ -1,5 +1,5 @@
1
1
  import type { Artifact, EvaluationErrorCategory, EvaluationJob, EvaluationRequest, EvaluationRun, JudgeConfig } from '../types/index.js';
2
- export declare function buildEvaluationRequest({ samplesPath, skillDir, artifacts, model, executor, noJudge, concurrency, timeoutMs, noCache, dryRun, blind, project, owner, tags, repeat, batch, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget, effort, }: {
2
+ export declare function buildEvaluationRequest({ samplesPath, skillDir, artifacts, model, executor, noJudge, concurrency, timeoutMs, noCache, dryRun, project, owner, tags, repeat, holdoutRatio, batch, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget, effort, }: {
3
3
  samplesPath: string;
4
4
  skillDir: string;
5
5
  artifacts: Artifact[];
@@ -10,11 +10,11 @@ export declare function buildEvaluationRequest({ samplesPath, skillDir, artifact
10
10
  timeoutMs?: number;
11
11
  noCache: boolean;
12
12
  dryRun: boolean;
13
- blind: boolean;
14
13
  project?: string;
15
14
  owner?: string;
16
15
  tags?: string[];
17
16
  repeat?: number;
17
+ holdoutRatio?: number;
18
18
  batch?: boolean;
19
19
  judgeRepeat?: number;
20
20
  judgeModels: JudgeConfig[];
@@ -1,7 +1,7 @@
1
1
  function nowIso() {
2
2
  return new Date().toISOString();
3
3
  }
4
- export function buildEvaluationRequest({ samplesPath, skillDir, artifacts, model, executor, noJudge, concurrency, timeoutMs, noCache, dryRun, blind, project, owner, tags, repeat, batch, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget, effort, }) {
4
+ export function buildEvaluationRequest({ samplesPath, skillDir, artifacts, model, executor, noJudge, concurrency, timeoutMs, noCache, dryRun, project, owner, tags, repeat, holdoutRatio, batch, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget, effort, }) {
5
5
  return {
6
6
  samplesPath,
7
7
  skillDir,
@@ -13,11 +13,11 @@ export function buildEvaluationRequest({ samplesPath, skillDir, artifacts, model
13
13
  timeoutMs,
14
14
  noCache,
15
15
  dryRun,
16
- blind,
17
16
  project,
18
17
  owner,
19
18
  tags,
20
19
  repeat,
20
+ holdoutRatio,
21
21
  batch,
22
22
  judgeRepeat,
23
23
  judgeModels,
@@ -28,7 +28,6 @@ interface AggregateReportOptions {
28
28
  layeredStats?: boolean;
29
29
  }
30
30
  export declare function aggregateReport({ runId, variants, model, judgeModel, noJudge, executorName, samples, tasks, results, totalCostUSD, artifacts, request, run, job, layeredStats, }: AggregateReportOptions): Report;
31
- export declare function applyBlindMode(report: Report, variants: string[], blindSeed: string): void;
32
31
  export interface PersistableReport {
33
32
  id: string;
34
33
  }
@@ -61,10 +61,14 @@ export function getCliVersion() {
61
61
  return PKG.version;
62
62
  }
63
63
  export function getGitInfo() {
64
+ // stdio 静默 stderr:在非 git 目录(如 omk init 出来的 demo)里 rev-parse 会打印
65
+ // `fatal: not a git repository` 到终端。catch 已把失败兜成 null(报告省略 git 信息),
66
+ // 这条 fatal 对用户是纯噪声,吞掉它。与 skill-loader 的 GIT_PROBE_STDIO 同口径。
67
+ const gitProbeStdio = ['ignore', 'pipe', 'ignore'];
64
68
  try {
65
- const commit = execFileSync('git', ['rev-parse', 'HEAD'], { encoding: 'utf-8' }).trim();
66
- const branch = execFileSync('git', ['rev-parse', '--abbrev-ref', 'HEAD'], { encoding: 'utf-8' }).trim();
67
- const dirty = execFileSync('git', ['status', '--porcelain'], { encoding: 'utf-8' }).trim().length > 0;
69
+ const commit = execFileSync('git', ['rev-parse', 'HEAD'], { encoding: 'utf-8', stdio: gitProbeStdio }).trim();
70
+ const branch = execFileSync('git', ['rev-parse', '--abbrev-ref', 'HEAD'], { encoding: 'utf-8', stdio: gitProbeStdio }).trim();
71
+ const dirty = execFileSync('git', ['status', '--porcelain'], { encoding: 'utf-8', stdio: gitProbeStdio }).trim().length > 0;
68
72
  return { commit, commitShort: commit.slice(0, 7), branch, dirty };
69
73
  }
70
74
  catch {
@@ -199,8 +203,8 @@ export function aggregateReport({ runId, variants, model, judgeModel, noJudge, e
199
203
  ...(noJudge ? {} : { runtime: getExecutorRuntimeFingerprint(jc.executor, jc.model, runtimeOptions) }),
200
204
  }));
201
205
  // length-debias is on by default; the request only sets it
202
- // false when the user passed --no-debias-length. The hash differs between
203
- // v3-cot-length (on) and v2-cot (off) so readers can detect the divergence.
206
+ // false when the user passed --no-debias-length. The judgePromptHash differs between
207
+ // the length-debias-on and -off prompt variants so readers can detect the divergence.
204
208
  const lengthDebiasOn = request?.lengthDebias !== false;
205
209
  const debiasModeList = [];
206
210
  if (lengthDebiasOn)
@@ -271,42 +275,6 @@ export function aggregateReport({ runId, variants, model, judgeModel, noJudge, e
271
275
  }])),
272
276
  };
273
277
  }
274
- export function applyBlindMode(report, variants, blindSeed) {
275
- const labels = variants.map((_, i) => String.fromCharCode(65 + i));
276
- let seed = parseInt(hashString(blindSeed).slice(0, 8), 16) | 0;
277
- const seededRandom = () => {
278
- seed |= 0;
279
- seed = seed + 0x6D2B79F5 | 0;
280
- let value = Math.imul(seed ^ seed >>> 15, 1 | seed);
281
- value ^= value + Math.imul(value ^ value >>> 7, 61 | value);
282
- return ((value ^ value >>> 14) >>> 0) / 4294967296;
283
- };
284
- const shuffled = [...variants];
285
- for (let i = shuffled.length - 1; i > 0; i--) {
286
- const j = Math.floor(seededRandom() * (i + 1));
287
- [shuffled[i], shuffled[j]] = [shuffled[j], shuffled[i]];
288
- }
289
- const blindMap = Object.fromEntries(shuffled.map((variant, i) => [labels[i], variant]));
290
- const reverseMap = Object.fromEntries(Object.entries(blindMap).map(([label, variant]) => [variant, label]));
291
- report.meta.blind = true;
292
- report.meta.blindMap = blindMap;
293
- report.meta.variants = labels;
294
- if (report.meta.executorRuntimes) {
295
- report.meta.executorRuntimes = Object.fromEntries(Object.entries(report.meta.executorRuntimes).map(([variant, runtime]) => [reverseMap[variant] ?? variant, runtime]));
296
- }
297
- const newSummary = {};
298
- for (const [variant, stats] of Object.entries(report.summary)) {
299
- newSummary[reverseMap[variant]] = stats;
300
- }
301
- report.summary = newSummary;
302
- for (const result of report.results) {
303
- const newVariants = {};
304
- for (const [variant, data] of Object.entries(result.variants)) {
305
- newVariants[reverseMap[variant]] = data;
306
- }
307
- result.variants = newVariants;
308
- }
309
- }
310
278
  export function persistReport(report, outputDir) {
311
279
  if (!outputDir)
312
280
  return null;
@@ -101,8 +101,9 @@ export function resolveExecutionStrategy(task, model, timeoutMs, verbose, effort
101
101
  verbose,
102
102
  ...(effort && { effort }),
103
103
  // pass skill-isolation declaration to executors. undefined keeps
104
- // SDK default; [] = strict isolation (skills:[] + disallowedTools:['Skill']);
105
- // [...] = whitelist (skills:[...] only).
104
+ // SDK default; [] = strict isolation (skills:[] + disallowedTools:['Skill']).
105
+ // non-empty allowedSkills is rejected upstream (validateEvalConfig) and by every
106
+ // executor — a skill whitelist could not be fully isolated, so it was removed.
106
107
  ...(task.artifact.allowedSkills !== undefined && { allowedSkills: task.artifact.allowedSkills }),
107
108
  // Sample.mocks 透传到 executor。executor(claude-sdk / claude-cli)
108
109
  // 自决定怎么落地(in-process hook vs 临时 CLAUDE_CONFIG_DIR + on-disk hook)。
@@ -0,0 +1,66 @@
1
+ /**
2
+ * Holdout split + train/holdout composite breakdown.
3
+ *
4
+ * Deterministic (no RNG) train/holdout partitioning of a sample set, plus the
5
+ * subset-composite recompute that lets `omk eval --holdout-ratio` and `omk evolve`
6
+ * score a variant on a withheld slice using the *same* aggregation as the headline
7
+ * composite (`buildVariantSummary`). Lives in eval-core so both the eval pipeline
8
+ * and the authoring/evolve loop depend *down* into it (authoring → eval-core is the
9
+ * established direction).
10
+ *
11
+ * A large train − holdout composite gap is the generalization / sample-set-overfitting
12
+ * signal the verdict's overfitting gate reads (`src/eval-core/verdict.ts`).
13
+ */
14
+ import type { Report, HoldoutBreakdown } from '../types/index.js';
15
+ /** A train / holdout partition of a sample set. */
16
+ export interface HoldoutSplit {
17
+ trainIds: Set<string>;
18
+ holdoutIds: Set<string>;
19
+ }
20
+ /** Below this many samples on any side, a split is too small to be meaningful —
21
+ * callers fall back to full-set scoring and mark the breakdown `disabled`. */
22
+ export declare const MIN_HOLDOUT_SUBSET = 3;
23
+ /** Pick `count` ids at an even stride across `ids` (deterministic, no RNG) so the
24
+ * picked subset is representative of the ordering and stable across rounds/runs. */
25
+ export declare function pickByStride(ids: string[], count: number): Set<string>;
26
+ /**
27
+ * Deterministically split sample ids into train / holdout by `ratio` (fraction
28
+ * held out). Holdout members are picked at an even stride so the partition is
29
+ * representative of the ordering, and the split is stable across rounds and runs
30
+ * (no RNG). Returns null when ratio ≤ 0 or either side would drop below
31
+ * MIN_HOLDOUT_SUBSET — the caller then scores on the full set.
32
+ */
33
+ export declare function splitHoldout(sampleIds: string[], ratio: number): HoldoutSplit | null;
34
+ /**
35
+ * Mean composite over the subset of a report's results whose sample_id is in
36
+ * `ids`, using the same aggregation as the full-run summary
37
+ * (`buildVariantSummary`) so train / holdout scores stay comparable to the
38
+ * headline composite. Returns 0 when the subset has no scorable entries.
39
+ */
40
+ export declare function subsetCompositeScore(report: Report, variantKey: string, ids: Set<string>): number;
41
+ /**
42
+ * How many subset results actually produced a usable composite (> 0) for a variant.
43
+ * `buildVariantSummary` averages only `compositeScore > 0` entries (schema.ts), so
44
+ * the mean can rest on far fewer samples than the authored split size when runs
45
+ * flake / partial-error / budget-abort. The overfitting gate must trust THIS count,
46
+ * not the authored `trainCount` / `holdoutCount`, or a 1-of-3 holdout gets dressed
47
+ * up as a 3-sample-backed conclusion.
48
+ */
49
+ export declare function subsetScorableCount(report: Report, variantKey: string, ids: Set<string>): number;
50
+ /**
51
+ * Train vs holdout composite breakdown per variant for `omk eval --holdout-ratio`.
52
+ * Post-hoc — never perturbs the headline aggregation or bootstrap CI.
53
+ *
54
+ * The split is taken over `sampleIdOrder` — the **stable authored sample order**
55
+ * (the loaded `samples` file order), NOT `report.results`, whose insertion order
56
+ * is the concurrent-completion order and drifts run-to-run. Binding the stride pick
57
+ * to the authored order is what makes the holdout (and the verdict overfitting gate
58
+ * it feeds) deterministic and reproducible. Subset scores are then read from
59
+ * `report.results` by id-set membership, which is order-independent.
60
+ *
61
+ * When the split is too small on either side (< MIN_HOLDOUT_SUBSET) it returns
62
+ * `{ disabled: true }` with an empty `perVariant`, so the verdict overfitting gate
63
+ * stays inert. The testSetHash watermark (gap-spec §7.1) is attached by the caller,
64
+ * shared with gapReports.
65
+ */
66
+ export declare function computeHoldoutBreakdown(report: Report, variantNames: string[], ratio: number, sampleIdOrder: string[]): HoldoutBreakdown;