oh-my-knowledge 0.41.0 → 0.42.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (36) hide show
  1. package/README.md +7 -2
  2. package/README.zh.md +7 -2
  3. package/dist/analysis/report-diagnostics.d.ts +8 -1
  4. package/dist/analysis/report-diagnostics.js +82 -1
  5. package/dist/assets/agent-skills/omk/references/commands.md +1 -0
  6. package/dist/authoring/evolver.d.ts +3 -14
  7. package/dist/authoring/evolver.js +1 -52
  8. package/dist/authoring/generator.d.ts +24 -0
  9. package/dist/authoring/generator.js +33 -6
  10. package/dist/cli/commands/eval/index.d.ts +1 -0
  11. package/dist/cli/commands/eval/index.js +40 -5
  12. package/dist/cli/commands/init.js +10 -7
  13. package/dist/cli/lib/i18n-dict/init.d.ts +1 -1
  14. package/dist/cli/lib/i18n-dict/init.js +14 -11
  15. package/dist/cli/lib/i18n-dict/run.d.ts +1 -1
  16. package/dist/cli/lib/i18n-dict/run.js +4 -0
  17. package/dist/cli/lib/parse-run-config.d.ts +3 -0
  18. package/dist/eval-core/evaluation-job.d.ts +2 -1
  19. package/dist/eval-core/evaluation-job.js +2 -1
  20. package/dist/eval-core/evaluation-reporting.js +7 -3
  21. package/dist/eval-core/holdout.d.ts +66 -0
  22. package/dist/eval-core/holdout.js +118 -0
  23. package/dist/eval-core/verdict.d.ts +44 -1
  24. package/dist/eval-core/verdict.js +175 -13
  25. package/dist/eval-workflows/evaluation-pipeline/report-finalize.js +19 -1
  26. package/dist/eval-workflows/evaluation-pipeline/run-state.d.ts +2 -1
  27. package/dist/eval-workflows/evaluation-pipeline/run-state.js +2 -1
  28. package/dist/eval-workflows/evaluation-pipeline.d.ts +3 -1
  29. package/dist/eval-workflows/evaluation-pipeline.js +2 -1
  30. package/dist/eval-workflows/run-evaluation.d.ts +5 -2
  31. package/dist/eval-workflows/run-evaluation.js +8 -5
  32. package/dist/inputs/eval-config.js +6 -0
  33. package/dist/renderer/summary.js +36 -3
  34. package/dist/types/eval.d.ts +7 -0
  35. package/dist/types/report.d.ts +49 -0
  36. package/package.json +1 -1
@@ -22,7 +22,7 @@ export interface EvaluationRunState {
22
22
  runningJob: EvaluationJob;
23
23
  resolvedJobStore: JobStore | null;
24
24
  }
25
- export declare function initializeEvaluationRunState({ samplesPath, skillDir, artifacts, model, judgeModel, noJudge, executorName, judgeExecutorName, concurrency, timeoutMs, noCache, project, owner, tags, runId, jobStore, persistJob, repeat, batch, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget, effort, }: {
25
+ export declare function initializeEvaluationRunState({ samplesPath, skillDir, artifacts, model, judgeModel, noJudge, executorName, judgeExecutorName, concurrency, timeoutMs, noCache, project, owner, tags, runId, jobStore, persistJob, repeat, holdoutRatio, batch, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget, effort, }: {
26
26
  samplesPath: string;
27
27
  skillDir: string;
28
28
  artifacts: Artifact[];
@@ -41,6 +41,7 @@ export declare function initializeEvaluationRunState({ samplesPath, skillDir, ar
41
41
  jobStore?: JobStore | null;
42
42
  persistJob?: boolean;
43
43
  repeat?: number;
44
+ holdoutRatio?: number;
44
45
  batch?: boolean;
45
46
  judgeRepeat?: number;
46
47
  judgeModels?: import('../../types/index.js').JudgeConfig[];
@@ -14,7 +14,7 @@
14
14
  import { buildEvaluationRequest, createFailedJob, createEvaluationRun, createQueuedJob, createSucceededJob, finalizeEvaluationRun, markJobRunning, failEvaluationRun, } from '../../eval-core/evaluation-job.js';
15
15
  import { createFileJobStore } from '../../server/job-store.js';
16
16
  import { DEFAULT_JOBS_DIR } from '../../eval-core/default-dirs.js';
17
- export async function initializeEvaluationRunState({ samplesPath, skillDir, artifacts, model, judgeModel, noJudge, executorName, judgeExecutorName, concurrency, timeoutMs, noCache, project, owner, tags, runId, jobStore, persistJob, repeat, batch, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget, effort, }) {
17
+ export async function initializeEvaluationRunState({ samplesPath, skillDir, artifacts, model, judgeModel, noJudge, executorName, judgeExecutorName, concurrency, timeoutMs, noCache, project, owner, tags, runId, jobStore, persistJob, repeat, holdoutRatio, batch, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget, effort, }) {
18
18
  const effectiveJudges = judgeModels && judgeModels.length > 0
19
19
  ? judgeModels
20
20
  : [{ executor: judgeExecutorName, model: judgeModel }];
@@ -33,6 +33,7 @@ export async function initializeEvaluationRunState({ samplesPath, skillDir, arti
33
33
  owner,
34
34
  tags,
35
35
  repeat,
36
+ holdoutRatio,
36
37
  batch,
37
38
  judgeRepeat,
38
39
  judgeModels: effectiveJudges,
@@ -61,6 +61,8 @@ export interface EvaluationPipelineOptions {
61
61
  layeredStats?: boolean;
62
62
  /** 透传到 meta.request.repeat */
63
63
  repeat?: number;
64
+ /** 透传到 meta.request.holdoutRatio;> 0 时 report-finalize 算 train/holdout 子集综合分。 */
65
+ holdoutRatio?: number;
64
66
  /** 透传到 meta.request.batch */
65
67
  batch?: boolean;
66
68
  /** 透传到 meta.request.judgeRepeat 与 grade(),每条 sample × dimension judge N 次 */
@@ -89,7 +91,7 @@ export interface EvaluationPipelineOptions {
89
91
  noDiagnostic?: boolean;
90
92
  }
91
93
  type VariantResult = import('../types/index.js').VariantResult;
92
- export declare function executeEvaluationPipeline({ samplesPath, samplesBaseDir, samplesSourceFiles, skillDir, samples, tasks, artifacts, model, judgeModel, noJudge, executorName, judgeExecutorName, executor, judgeExecutor, outputDir, project, owner, tags, concurrency, timeoutMs, noCache, jobStore, persistJob, onProgress, skipConnectivity, verbose, retry, existingResults, requires: _requires, layeredStats, repeat, batch, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget, strictBaseline, runId, lang, effort, noDiagnostic, }: EvaluationPipelineOptions): Promise<{
94
+ export declare function executeEvaluationPipeline({ samplesPath, samplesBaseDir, samplesSourceFiles, skillDir, samples, tasks, artifacts, model, judgeModel, noJudge, executorName, judgeExecutorName, executor, judgeExecutor, outputDir, project, owner, tags, concurrency, timeoutMs, noCache, jobStore, persistJob, onProgress, skipConnectivity, verbose, retry, existingResults, requires: _requires, layeredStats, repeat, holdoutRatio, batch, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget, strictBaseline, runId, lang, effort, noDiagnostic, }: EvaluationPipelineOptions): Promise<{
93
95
  report: Report;
94
96
  filePath: string | null;
95
97
  }>;
@@ -33,7 +33,7 @@ export { _computeTestSetHashForTest } from './evaluation-pipeline/test-set-hash.
33
33
  export async function executeEvaluationPipeline({ samplesPath, samplesBaseDir, samplesSourceFiles, skillDir, samples, tasks, artifacts, model, judgeModel, noJudge, executorName, judgeExecutorName, executor, judgeExecutor, outputDir = DEFAULT_OUTPUT_DIR, project, owner, tags, concurrency = 1, timeoutMs, noCache = false, jobStore = null, persistJob = true, onProgress = null, skipConnectivity = false, verbose = false, retry = 0, existingResults,
34
34
  // requires 现在由 runEvaluation 上游传给 doctor 处理; eval-pipeline 不再用
35
35
  // 但保留接口字段,避免破 programmatic API (类型层面接收, 内部忽略)
36
- requires: _requires, layeredStats = false, repeat, batch, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias = true, budget, strictBaseline, runId, lang = 'zh', effort, noDiagnostic, }) {
36
+ requires: _requires, layeredStats = false, repeat, holdoutRatio, batch, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias = true, budget, strictBaseline, runId, lang = 'zh', effort, noDiagnostic, }) {
37
37
  const variantNames = artifacts.map((artifact) => artifact.name);
38
38
  const runState = await initializeEvaluationRunState({
39
39
  samplesPath,
@@ -54,6 +54,7 @@ requires: _requires, layeredStats = false, repeat, batch, judgeRepeat, judgeMode
54
54
  jobStore,
55
55
  persistJob,
56
56
  repeat,
57
+ holdoutRatio,
57
58
  batch,
58
59
  judgeRepeat,
59
60
  judgeModels,
@@ -32,6 +32,9 @@ interface CommonEvaluationOptions {
32
32
  /** --repeat N. 1 表示单次(默认); > 1 时在 runMultiple 层聚合 variance。
33
33
  * 记入 report.meta.request.repeat 让 meta 如实反映用户输入。 */
34
34
  repeat?: number;
35
+ /** --holdout-ratio R. 0 / 缺省 = 不切分(默认)。> 0 时 report-finalize 算 train/holdout
36
+ * 子集综合分(report.analysis.holdout),供 verdict 过拟合门控读取。 */
37
+ holdoutRatio?: number;
35
38
  /** --batch 模式标记, true 表示当前评测是 skill batch 流程。
36
39
  * 记入 report.meta.request.batch。 */
37
40
  batch?: boolean;
@@ -117,7 +120,7 @@ export interface DryRunReport extends DryRunBase {
117
120
  samplesPath: string;
118
121
  tasks: DryRunTask[];
119
122
  }
120
- export declare function runEvaluation({ samplesPath, skillDir, variantSpecs, model, outputDir, project, owner, tags, noJudge, dryRun, concurrency, timeoutMs, noCache, executorName, jobStore, persistJob, onProgress, skipConnectivity, skipDoctor, lang, mcpConfig, verbose, retry, resume, runId, layeredStats, repeat, batch, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget, strictBaseline, effort, noDiagnostic, }: RunEvaluationOptions): Promise<{
123
+ export declare function runEvaluation({ samplesPath, skillDir, variantSpecs, model, outputDir, project, owner, tags, noJudge, dryRun, concurrency, timeoutMs, noCache, executorName, jobStore, persistJob, onProgress, skipConnectivity, skipDoctor, lang, mcpConfig, verbose, retry, resume, runId, layeredStats, repeat, holdoutRatio, batch, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget, strictBaseline, effort, noDiagnostic, }: RunEvaluationOptions): Promise<{
121
124
  report: Report | DryRunReport;
122
125
  filePath: string | null;
123
126
  }>;
@@ -133,7 +136,7 @@ export interface DryRunBatchReport extends DryRunBase {
133
136
  artifacts: DryRunBatchSkill[];
134
137
  }
135
138
  export declare function buildVarianceData(runs: Report[], bootstrapSamples?: number, seed?: number): VarianceData | null;
136
- export declare function runBatchEvaluation({ skillDir, model, outputDir, project, owner, tags, noJudge, dryRun, concurrency, timeoutMs, executorName, jobStore, persistJob, onProgress, onSkillProgress, skipConnectivity, skipDoctor, lang, mcpConfig, verbose, repeat, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget, noCache, strictBaseline, variantAllowedSkills, }: RunBatchEvaluationOptions): Promise<{
139
+ export declare function runBatchEvaluation({ skillDir, model, outputDir, project, owner, tags, noJudge, dryRun, concurrency, timeoutMs, executorName, jobStore, persistJob, onProgress, onSkillProgress, skipConnectivity, skipDoctor, lang, mcpConfig, verbose, repeat, holdoutRatio, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget, noCache, strictBaseline, variantAllowedSkills, }: RunBatchEvaluationOptions): Promise<{
137
140
  report: BatchEvaluationReport | DryRunBatchReport;
138
141
  filePath: string | null;
139
142
  }>;
@@ -8,7 +8,7 @@ import { buildDryRunTaskReport, prepareEvaluationRun, } from './evaluation-prepa
8
8
  import { executeEvaluationPipeline } from './evaluation-pipeline.js';
9
9
  import { findSaturationPoint } from '../analysis/saturation.js';
10
10
  import { bootstrapMeanCI, DEFAULT_BOOTSTRAP_ALPHA, DEFAULT_BOOTSTRAP_SAMPLES } from '../eval-core/bootstrap.js';
11
- export async function runEvaluation({ samplesPath, skillDir, variantSpecs = [], model = DEFAULT_MODEL, outputDir = DEFAULT_OUTPUT_DIR, project, owner, tags, noJudge = false, dryRun = false, concurrency = 1, timeoutMs, noCache = false, executorName = 'claude', jobStore = null, persistJob = true, onProgress = null, skipConnectivity = false, skipDoctor = false, lang = 'zh', mcpConfig, verbose = false, retry = 0, resume, runId, layeredStats = false, repeat, batch, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget, strictBaseline, effort, noDiagnostic, }) {
11
+ export async function runEvaluation({ samplesPath, skillDir, variantSpecs = [], model = DEFAULT_MODEL, outputDir = DEFAULT_OUTPUT_DIR, project, owner, tags, noJudge = false, dryRun = false, concurrency = 1, timeoutMs, noCache = false, executorName = 'claude', jobStore = null, persistJob = true, onProgress = null, skipConnectivity = false, skipDoctor = false, lang = 'zh', mcpConfig, verbose = false, retry = 0, resume, runId, layeredStats = false, repeat, holdoutRatio, batch, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget, strictBaseline, effort, noDiagnostic, }) {
12
12
  // Unified judgeModels → derive single-judge fields for downstream pipeline / grading
13
13
  // (which still operate on string `judgeModel` + `judgeExecutorName` fields per call).
14
14
  const effectiveJudgeModels = judgeModels && judgeModels.length > 0
@@ -165,6 +165,7 @@ export async function runEvaluation({ samplesPath, skillDir, variantSpecs = [],
165
165
  requires,
166
166
  layeredStats,
167
167
  repeat,
168
+ holdoutRatio,
168
169
  batch,
169
170
  judgeRepeat,
170
171
  judgeModels,
@@ -366,7 +367,7 @@ export function buildVarianceData(runs, bootstrapSamples = DEFAULT_BOOTSTRAP_SAM
366
367
  ...(saturation ? { saturation } : {}),
367
368
  };
368
369
  }
369
- export async function runBatchEvaluation({ skillDir, model = DEFAULT_MODEL, outputDir = DEFAULT_OUTPUT_DIR, project, owner, tags, noJudge = false, dryRun = false, concurrency = 1, timeoutMs, executorName = 'claude', jobStore = null, persistJob = true, onProgress = null, onSkillProgress = null, skipConnectivity = false, skipDoctor = false, lang = 'zh', mcpConfig, verbose = false, repeat, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget, noCache = false, strictBaseline, variantAllowedSkills, }) {
370
+ export async function runBatchEvaluation({ skillDir, model = DEFAULT_MODEL, outputDir = DEFAULT_OUTPUT_DIR, project, owner, tags, noJudge = false, dryRun = false, concurrency = 1, timeoutMs, executorName = 'claude', jobStore = null, persistJob = true, onProgress = null, onSkillProgress = null, skipConnectivity = false, skipDoctor = false, lang = 'zh', mcpConfig, verbose = false, repeat, holdoutRatio, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget, noCache = false, strictBaseline, variantAllowedSkills, }) {
370
371
  // Same unified judge derivation as runEvaluation (downstream pipeline / report build
371
372
  // still uses single judgeModel + judgeExecutorName per call).
372
373
  const effectiveJudgeModels = judgeModels && judgeModels.length > 0
@@ -485,12 +486,14 @@ export async function runBatchEvaluation({ skillDir, model = DEFAULT_MODEL, outp
485
486
  strictBaseline,
486
487
  variantAllowedSkills,
487
488
  runSingleEvaluation: async (options) => {
488
- // repeat > 1 时走 runMultiple 做 variance; batch=true 标记让 meta.request 如实反映
489
+ // repeat > 1 时走 runMultiple 做 variance; batch=true 标记让 meta.request 如实反映。
490
+ // holdoutRatio 显式转发(executeBatchEvaluationRuns 不一定把它塞进 options),否则
491
+ // batch 子报告 meta.request.holdoutRatio 丢失、过拟合门控对 batch 静默失效。
489
492
  if (repeat && repeat > 1) {
490
- const multi = await runMultiple({ ...options, repeat, batch: true, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget });
493
+ const multi = await runMultiple({ ...options, repeat, holdoutRatio, batch: true, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget });
491
494
  return { report: multi.report, filePath: multi.filePath };
492
495
  }
493
- const result = await runEvaluation({ ...options, batch: true, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget });
496
+ const result = await runEvaluation({ ...options, batch: true, holdoutRatio, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget });
494
497
  return { report: result.report, filePath: result.filePath };
495
498
  },
496
499
  });
@@ -171,6 +171,11 @@ function validateEvalConfig(parsed, configPath) {
171
171
  };
172
172
  assertPositiveIntOpt('repeat');
173
173
  assertPositiveIntOpt('judgeRepeat');
174
+ if (obj.holdoutRatio !== undefined) {
175
+ if (typeof obj.holdoutRatio !== 'number' || !Number.isFinite(obj.holdoutRatio) || obj.holdoutRatio <= 0 || obj.holdoutRatio >= 1) {
176
+ throw new Error(`${configPath}: holdoutRatio must be a number in (0, 1)`);
177
+ }
178
+ }
174
179
  if (obj.bootstrapSamples !== undefined) {
175
180
  if (typeof obj.bootstrapSamples !== 'number' || !Number.isFinite(obj.bootstrapSamples) || obj.bootstrapSamples < 100) {
176
181
  throw new Error(`${configPath}: bootstrapSamples must be a number ≥ 100`);
@@ -256,6 +261,7 @@ function validateEvalConfig(parsed, configPath) {
256
261
  variants,
257
262
  budget,
258
263
  repeat: obj.repeat,
264
+ holdoutRatio: obj.holdoutRatio,
259
265
  judgeRepeat: obj.judgeRepeat,
260
266
  bootstrap: obj.bootstrap,
261
267
  bootstrapSamples: obj.bootstrapSamples,
@@ -101,6 +101,30 @@ function computeMedianCVPercent(report) {
101
101
  const stab = medianStabilityCV(report);
102
102
  return stab ? stab.cv * 100 : null;
103
103
  }
104
+ // Verdict caveats(过拟合 / 知识缺口)渲染进 pill —— 让 HTML 报告和 CLI 说同一件事:
105
+ // CLI 在 verbose rationale 里给这两条,HTML 之前只剩一个泛化后的 level、看不到触发原因。
106
+ // 用 result.caveats 的结构化数据 i18n,而不是重解析 zh rationale 串。
107
+ function renderVerdictCaveats(caveats, lang) {
108
+ if (!caveats)
109
+ return '';
110
+ const lines = [];
111
+ if (caveats.overfitting) {
112
+ const c = caveats.overfitting;
113
+ lines.push(lang === 'zh'
114
+ ? `⚠ 过拟合敞口:${c.variant} 训练 ${c.trainScore.toFixed(2)} / 留出 ${c.holdoutScore.toFixed(2)}(差 ${c.gap.toFixed(2)}),提升可能不泛化`
115
+ : `⚠ Overfitting: ${c.variant} train ${c.trainScore.toFixed(2)} / holdout ${c.holdoutScore.toFixed(2)} (gap ${c.gap.toFixed(2)}) — gain may not generalize`);
116
+ }
117
+ if (caveats.gapSignal) {
118
+ const g = caveats.gapSignal;
119
+ const wm = g.testSetHash ? g.testSetHash.slice(0, 8) : (g.testSetPath ?? '');
120
+ lines.push(lang === 'zh'
121
+ ? `知识缺口率 ${g.gapRatePct}%(test set ${wm},informational)`
122
+ : `Knowledge gap ${g.gapRatePct}% (test set ${wm}, informational)`);
123
+ }
124
+ if (lines.length === 0)
125
+ return '';
126
+ return `<div class="page-verdict-caveats">${lines.map((l) => `<span class="page-verdict-caveat">${e(l)}</span>`).join('')}</div>`;
127
+ }
104
128
  export function renderVerdictPill(report, lang) {
105
129
  let result;
106
130
  try {
@@ -110,8 +134,16 @@ export function renderVerdictPill(report, lang) {
110
134
  return '';
111
135
  }
112
136
  const level = result.level;
113
- const pair = result.perPair?.[0];
137
+ // representative = top-level worst pair(与 CLI 同口径),不是第一对。多 treatment 报告里
138
+ // worst pair 不一定是 perPair[0],用它才不会把错的 treatment 名写进结论。fallback 兼容旧路径。
139
+ const pair = result.representative ?? result.perPair?.[0];
114
140
  const oneLine = verdictOneLine(level, lang, pair?.treatment, pair?.control);
141
+ // Δ/CI 证据必须跟文案指同一对:按 representative 匹配对应的 pairComparison(alpha 也走这对),
142
+ // 否则多 treatment 报告会出现「文案 t2、数字 t1」的混搭。匹配不到 / 无 representative 时 fallback [0]。
143
+ const pairComparisons = report.meta?.pairComparisons;
144
+ const activeComparison = (pair
145
+ ? pairComparisons?.find((p) => p.treatment === pair.treatment && p.control === pair.control)
146
+ : undefined) ?? pairComparisons?.[0];
115
147
  const tooltip = levelTooltip(level, lang);
116
148
  const prefix = lang === 'zh' ? '测评结论' : 'Verdict';
117
149
  // 机器可读 enum 永远是 level token; 显示给用户的文字按 lang i18n.
@@ -120,7 +152,7 @@ export function renderVerdictPill(report, lang) {
120
152
  // hero 只放「答案」: 分差是 verdict 的核心证据数字, 单独一枚 chip。
121
153
  // 评测规模 (用例数 × 轮次) 走「实验配置」section 的 subtitle 那条 canonical 路径,
122
154
  // 不在 hero 里重复; CV / CI 走 chip tooltip + 方法学审计 / 波动表。
123
- const ci = report.meta?.pairComparisons?.[0]?.diffBootstrapCI;
155
+ const ci = activeComparison?.diffBootstrapCI;
124
156
  const cvPct = computeMedianCVPercent(report);
125
157
  const metrics = [];
126
158
  if (ci) {
@@ -128,7 +160,7 @@ export function renderVerdictPill(report, lang) {
128
160
  const cvSuffix = cvPct != null
129
161
  ? (lang === 'zh' ? `;多轮稳定性 CV=${cvPct.toFixed(1)}% (${cvPct < 5 ? '稳' : cvPct <= STABILITY_UNSTABLE_CV * 100 ? '中' : '不稳'})` : `; CV=${cvPct.toFixed(1)}% (${cvPct < 5 ? 'stable' : cvPct <= STABILITY_UNSTABLE_CV * 100 ? 'moderate' : 'unstable'})`)
130
162
  : '';
131
- const pctLabel = ciLevelLabel(report.meta?.pairComparisons?.[0]?.alpha);
163
+ const pctLabel = ciLevelLabel(activeComparison?.alpha);
132
164
  const ciTipBase = lang === 'zh'
133
165
  ? `实验组与对照组综合分均值差(Δ)。bootstrap ${pctLabel} 可信区间 [${ci.low}, ${ci.high}],${ci.significant ? '不含 0 = 差异显著' : '跨过 0 = 差异不显著'}${cvSuffix}`
134
166
  : `Treatment minus control mean composite score (Δ). Bootstrap ${pctLabel} CI [${ci.low}, ${ci.high}], ${ci.significant ? 'excludes 0 ⇒ significant' : 'spans 0 ⇒ not significant'}${cvSuffix}`;
@@ -146,6 +178,7 @@ export function renderVerdictPill(report, lang) {
146
178
  <span class="page-verdict-badge"><span class="page-verdict-badge-dot" aria-hidden="true">●</span>${e(levelDisplay)}</span>
147
179
  <span class="page-verdict-text">${e(oneLine)}</span>
148
180
  </div>
181
+ ${renderVerdictCaveats(result.caveats, lang)}
149
182
  ${metricChips ? `<div class="page-verdict-metrics">${metricChips}</div>` : ''}
150
183
  </section>`;
151
184
  }
@@ -220,6 +220,9 @@ export interface EvalConfig {
220
220
  budget?: EvalBudget;
221
221
  /** --repeat N. Multi-run variance analysis. */
222
222
  repeat?: number;
223
+ /** --holdout-ratio R (0 < R < 1). Hold out a deterministic sample slice and
224
+ * report train vs holdout composite as a generalization / overfitting signal. */
225
+ holdoutRatio?: number;
223
226
  /** --judge-repeat N. Each (sample × dimension) judged N times for self-consistency stddev. */
224
227
  judgeRepeat?: number;
225
228
  /** --bootstrap. Distribution-free CI per variant + pairwise diff. */
@@ -257,6 +260,10 @@ export interface EvaluationRequest {
257
260
  dryRun: boolean;
258
261
  /** --repeat N; 1 表示单次跑,> 1 走 runMultiple 做 variance 分析 */
259
262
  repeat?: number;
263
+ /** --holdout-ratio R; 0 / 缺省表示不切分(默认)。> 0 时 report-finalize 在结果上
264
+ * post-hoc 切出 train / holdout 子集算综合分(`report.analysis.holdout`),供 verdict
265
+ * 的过拟合门控读取。see src/eval-core/holdout.ts */
266
+ holdoutRatio?: number;
260
267
  /** --batch; default absent/false. True means skill-batch mode. */
261
268
  batch?: boolean;
262
269
  /** --judge-repeat N; 每条 sample × dimension 用 LLM judge 跑 N 次, 输出 stddev. 默认 1 (单次). */
@@ -480,6 +480,34 @@ export interface AnalysisResult {
480
480
  * (capability / difficulty / construct / provenance); persisted on report
481
481
  * for studio to surface coverage gaps. See docs/specs/sample-design-spec.md. */
482
482
  sampleQuality?: SampleQualityAggregate;
483
+ /** Opt-in train/holdout generalization breakdown (`omk eval --holdout-ratio`).
484
+ * Absent on default runs; present only when a holdout ratio was requested. */
485
+ holdout?: HoldoutBreakdown;
486
+ }
487
+ /** Train vs holdout composite breakdown for `omk eval --holdout-ratio`.
488
+ * Computed post-hoc from `report.results` by `computeHoldoutBreakdown`
489
+ * (`src/eval-core/holdout.ts`), sharing the same testSetHash watermark as
490
+ * gapReports (gap-spec §7.1). A large train − holdout composite gap is the
491
+ * sample-set-overfitting signal the verdict's overfitting gate reads. */
492
+ export interface HoldoutBreakdown {
493
+ /** Held-out fraction requested via --holdout-ratio. */
494
+ ratio: number;
495
+ /** true when either side fell below the minimum subset size → scored full-set,
496
+ * no usable split. `perVariant` is empty and the verdict gate stays inert. */
497
+ disabled?: boolean;
498
+ /** Per-variant train vs holdout composite (1-5 scale). `*Count` is the authored
499
+ * split size; `*Scorable` is how many of those actually produced a composite (> 0)
500
+ * — they diverge under partial errors, and the overfitting gate trusts `*Scorable`. */
501
+ perVariant: Record<string, {
502
+ trainScore: number;
503
+ holdoutScore: number;
504
+ trainCount: number;
505
+ holdoutCount: number;
506
+ trainScorable: number;
507
+ holdoutScorable: number;
508
+ }>;
509
+ testSetPath?: string | null;
510
+ testSetHash?: string | null;
483
511
  }
484
512
  /** Aggregated sample design coverage stats. Built by
485
513
  * `buildSampleQualityAggregate(samples)` from `Sample.capability` /
@@ -502,6 +530,27 @@ export interface SampleQualityAggregate {
502
530
  sampleCountWithDifficulty: number;
503
531
  sampleCountWithConstruct: number;
504
532
  sampleCountWithProvenance: number;
533
+ /** Relative-balance / skew of the sample set (derived from the distributions
534
+ * above). Flags over-representation — "70% of samples are easy" — without an
535
+ * external denominator. Diagnostic only; never feeds grading / judge / verdict. */
536
+ representativeness?: Representativeness;
537
+ }
538
+ /** Distribution skew over what the sample set declares. Pure relative balance —
539
+ * there is no authored "expected" capability list to measure absolute coverage
540
+ * against (capabilities are free-form strings), so this reports concentration
541
+ * (dominant bucket share, 0-1) and the dominant label per dimension. */
542
+ export interface Representativeness {
543
+ /** Distinct capabilities declared across the set. */
544
+ capabilityCount: number;
545
+ /** Dominant capability's share of all capability tags (0-1); 0 when none declared. */
546
+ capabilityConcentration: number;
547
+ dominantCapability?: string;
548
+ /** Dominant difficulty bucket's share of samples that declared a difficulty (0-1). */
549
+ difficultyConcentration: number;
550
+ dominantDifficulty?: 'easy' | 'medium' | 'hard';
551
+ /** Dominant construct's share of samples that declared a construct (0-1). */
552
+ constructConcentration: number;
553
+ dominantConstruct?: string;
505
554
  }
506
555
  export interface HedgingVerdict {
507
556
  isUncertainty: boolean;
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "oh-my-knowledge",
3
- "version": "0.41.0",
3
+ "version": "0.42.0",
4
4
  "packageManager": "yarn@4.16.0",
5
5
  "description": "Evaluation framework for LLM knowledge inputs — prompts, RAG corpora, skills, agent workflows. Fix the model, vary the artifact. Built-in statistical rigor: bootstrap CI, Krippendorff α, length-debias, saturation curves.",
6
6
  "type": "module",