oh-my-knowledge 0.41.0 → 0.42.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +7 -2
- package/README.zh.md +7 -2
- package/dist/analysis/report-diagnostics.d.ts +8 -1
- package/dist/analysis/report-diagnostics.js +82 -1
- package/dist/assets/agent-skills/omk/references/commands.md +1 -0
- package/dist/authoring/evolver.d.ts +3 -14
- package/dist/authoring/evolver.js +1 -52
- package/dist/authoring/generator.d.ts +24 -0
- package/dist/authoring/generator.js +33 -6
- package/dist/cli/commands/eval/index.d.ts +1 -0
- package/dist/cli/commands/eval/index.js +40 -5
- package/dist/cli/commands/init.js +10 -7
- package/dist/cli/lib/i18n-dict/init.d.ts +1 -1
- package/dist/cli/lib/i18n-dict/init.js +14 -11
- package/dist/cli/lib/i18n-dict/run.d.ts +1 -1
- package/dist/cli/lib/i18n-dict/run.js +4 -0
- package/dist/cli/lib/parse-run-config.d.ts +3 -0
- package/dist/eval-core/evaluation-job.d.ts +2 -1
- package/dist/eval-core/evaluation-job.js +2 -1
- package/dist/eval-core/evaluation-reporting.js +7 -3
- package/dist/eval-core/holdout.d.ts +66 -0
- package/dist/eval-core/holdout.js +118 -0
- package/dist/eval-core/verdict.d.ts +44 -1
- package/dist/eval-core/verdict.js +175 -13
- package/dist/eval-workflows/evaluation-pipeline/report-finalize.js +19 -1
- package/dist/eval-workflows/evaluation-pipeline/run-state.d.ts +2 -1
- package/dist/eval-workflows/evaluation-pipeline/run-state.js +2 -1
- package/dist/eval-workflows/evaluation-pipeline.d.ts +3 -1
- package/dist/eval-workflows/evaluation-pipeline.js +2 -1
- package/dist/eval-workflows/run-evaluation.d.ts +5 -2
- package/dist/eval-workflows/run-evaluation.js +8 -5
- package/dist/inputs/eval-config.js +6 -0
- package/dist/renderer/summary.js +36 -3
- package/dist/types/eval.d.ts +7 -0
- package/dist/types/report.d.ts +49 -0
- package/package.json +1 -1
|
@@ -22,7 +22,7 @@ export interface EvaluationRunState {
|
|
|
22
22
|
runningJob: EvaluationJob;
|
|
23
23
|
resolvedJobStore: JobStore | null;
|
|
24
24
|
}
|
|
25
|
-
export declare function initializeEvaluationRunState({ samplesPath, skillDir, artifacts, model, judgeModel, noJudge, executorName, judgeExecutorName, concurrency, timeoutMs, noCache, project, owner, tags, runId, jobStore, persistJob, repeat, batch, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget, effort, }: {
|
|
25
|
+
export declare function initializeEvaluationRunState({ samplesPath, skillDir, artifacts, model, judgeModel, noJudge, executorName, judgeExecutorName, concurrency, timeoutMs, noCache, project, owner, tags, runId, jobStore, persistJob, repeat, holdoutRatio, batch, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget, effort, }: {
|
|
26
26
|
samplesPath: string;
|
|
27
27
|
skillDir: string;
|
|
28
28
|
artifacts: Artifact[];
|
|
@@ -41,6 +41,7 @@ export declare function initializeEvaluationRunState({ samplesPath, skillDir, ar
|
|
|
41
41
|
jobStore?: JobStore | null;
|
|
42
42
|
persistJob?: boolean;
|
|
43
43
|
repeat?: number;
|
|
44
|
+
holdoutRatio?: number;
|
|
44
45
|
batch?: boolean;
|
|
45
46
|
judgeRepeat?: number;
|
|
46
47
|
judgeModels?: import('../../types/index.js').JudgeConfig[];
|
|
@@ -14,7 +14,7 @@
|
|
|
14
14
|
import { buildEvaluationRequest, createFailedJob, createEvaluationRun, createQueuedJob, createSucceededJob, finalizeEvaluationRun, markJobRunning, failEvaluationRun, } from '../../eval-core/evaluation-job.js';
|
|
15
15
|
import { createFileJobStore } from '../../server/job-store.js';
|
|
16
16
|
import { DEFAULT_JOBS_DIR } from '../../eval-core/default-dirs.js';
|
|
17
|
-
export async function initializeEvaluationRunState({ samplesPath, skillDir, artifacts, model, judgeModel, noJudge, executorName, judgeExecutorName, concurrency, timeoutMs, noCache, project, owner, tags, runId, jobStore, persistJob, repeat, batch, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget, effort, }) {
|
|
17
|
+
export async function initializeEvaluationRunState({ samplesPath, skillDir, artifacts, model, judgeModel, noJudge, executorName, judgeExecutorName, concurrency, timeoutMs, noCache, project, owner, tags, runId, jobStore, persistJob, repeat, holdoutRatio, batch, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget, effort, }) {
|
|
18
18
|
const effectiveJudges = judgeModels && judgeModels.length > 0
|
|
19
19
|
? judgeModels
|
|
20
20
|
: [{ executor: judgeExecutorName, model: judgeModel }];
|
|
@@ -33,6 +33,7 @@ export async function initializeEvaluationRunState({ samplesPath, skillDir, arti
|
|
|
33
33
|
owner,
|
|
34
34
|
tags,
|
|
35
35
|
repeat,
|
|
36
|
+
holdoutRatio,
|
|
36
37
|
batch,
|
|
37
38
|
judgeRepeat,
|
|
38
39
|
judgeModels: effectiveJudges,
|
|
@@ -61,6 +61,8 @@ export interface EvaluationPipelineOptions {
|
|
|
61
61
|
layeredStats?: boolean;
|
|
62
62
|
/** 透传到 meta.request.repeat */
|
|
63
63
|
repeat?: number;
|
|
64
|
+
/** 透传到 meta.request.holdoutRatio;> 0 时 report-finalize 算 train/holdout 子集综合分。 */
|
|
65
|
+
holdoutRatio?: number;
|
|
64
66
|
/** 透传到 meta.request.batch */
|
|
65
67
|
batch?: boolean;
|
|
66
68
|
/** 透传到 meta.request.judgeRepeat 与 grade(),每条 sample × dimension judge N 次 */
|
|
@@ -89,7 +91,7 @@ export interface EvaluationPipelineOptions {
|
|
|
89
91
|
noDiagnostic?: boolean;
|
|
90
92
|
}
|
|
91
93
|
type VariantResult = import('../types/index.js').VariantResult;
|
|
92
|
-
export declare function executeEvaluationPipeline({ samplesPath, samplesBaseDir, samplesSourceFiles, skillDir, samples, tasks, artifacts, model, judgeModel, noJudge, executorName, judgeExecutorName, executor, judgeExecutor, outputDir, project, owner, tags, concurrency, timeoutMs, noCache, jobStore, persistJob, onProgress, skipConnectivity, verbose, retry, existingResults, requires: _requires, layeredStats, repeat, batch, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget, strictBaseline, runId, lang, effort, noDiagnostic, }: EvaluationPipelineOptions): Promise<{
|
|
94
|
+
export declare function executeEvaluationPipeline({ samplesPath, samplesBaseDir, samplesSourceFiles, skillDir, samples, tasks, artifacts, model, judgeModel, noJudge, executorName, judgeExecutorName, executor, judgeExecutor, outputDir, project, owner, tags, concurrency, timeoutMs, noCache, jobStore, persistJob, onProgress, skipConnectivity, verbose, retry, existingResults, requires: _requires, layeredStats, repeat, holdoutRatio, batch, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget, strictBaseline, runId, lang, effort, noDiagnostic, }: EvaluationPipelineOptions): Promise<{
|
|
93
95
|
report: Report;
|
|
94
96
|
filePath: string | null;
|
|
95
97
|
}>;
|
|
@@ -33,7 +33,7 @@ export { _computeTestSetHashForTest } from './evaluation-pipeline/test-set-hash.
|
|
|
33
33
|
export async function executeEvaluationPipeline({ samplesPath, samplesBaseDir, samplesSourceFiles, skillDir, samples, tasks, artifacts, model, judgeModel, noJudge, executorName, judgeExecutorName, executor, judgeExecutor, outputDir = DEFAULT_OUTPUT_DIR, project, owner, tags, concurrency = 1, timeoutMs, noCache = false, jobStore = null, persistJob = true, onProgress = null, skipConnectivity = false, verbose = false, retry = 0, existingResults,
|
|
34
34
|
// requires 现在由 runEvaluation 上游传给 doctor 处理; eval-pipeline 不再用
|
|
35
35
|
// 但保留接口字段,避免破 programmatic API (类型层面接收, 内部忽略)
|
|
36
|
-
requires: _requires, layeredStats = false, repeat, batch, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias = true, budget, strictBaseline, runId, lang = 'zh', effort, noDiagnostic, }) {
|
|
36
|
+
requires: _requires, layeredStats = false, repeat, holdoutRatio, batch, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias = true, budget, strictBaseline, runId, lang = 'zh', effort, noDiagnostic, }) {
|
|
37
37
|
const variantNames = artifacts.map((artifact) => artifact.name);
|
|
38
38
|
const runState = await initializeEvaluationRunState({
|
|
39
39
|
samplesPath,
|
|
@@ -54,6 +54,7 @@ requires: _requires, layeredStats = false, repeat, batch, judgeRepeat, judgeMode
|
|
|
54
54
|
jobStore,
|
|
55
55
|
persistJob,
|
|
56
56
|
repeat,
|
|
57
|
+
holdoutRatio,
|
|
57
58
|
batch,
|
|
58
59
|
judgeRepeat,
|
|
59
60
|
judgeModels,
|
|
@@ -32,6 +32,9 @@ interface CommonEvaluationOptions {
|
|
|
32
32
|
/** --repeat N. 1 表示单次(默认); > 1 时在 runMultiple 层聚合 variance。
|
|
33
33
|
* 记入 report.meta.request.repeat 让 meta 如实反映用户输入。 */
|
|
34
34
|
repeat?: number;
|
|
35
|
+
/** --holdout-ratio R. 0 / 缺省 = 不切分(默认)。> 0 时 report-finalize 算 train/holdout
|
|
36
|
+
* 子集综合分(report.analysis.holdout),供 verdict 过拟合门控读取。 */
|
|
37
|
+
holdoutRatio?: number;
|
|
35
38
|
/** --batch 模式标记, true 表示当前评测是 skill batch 流程。
|
|
36
39
|
* 记入 report.meta.request.batch。 */
|
|
37
40
|
batch?: boolean;
|
|
@@ -117,7 +120,7 @@ export interface DryRunReport extends DryRunBase {
|
|
|
117
120
|
samplesPath: string;
|
|
118
121
|
tasks: DryRunTask[];
|
|
119
122
|
}
|
|
120
|
-
export declare function runEvaluation({ samplesPath, skillDir, variantSpecs, model, outputDir, project, owner, tags, noJudge, dryRun, concurrency, timeoutMs, noCache, executorName, jobStore, persistJob, onProgress, skipConnectivity, skipDoctor, lang, mcpConfig, verbose, retry, resume, runId, layeredStats, repeat, batch, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget, strictBaseline, effort, noDiagnostic, }: RunEvaluationOptions): Promise<{
|
|
123
|
+
export declare function runEvaluation({ samplesPath, skillDir, variantSpecs, model, outputDir, project, owner, tags, noJudge, dryRun, concurrency, timeoutMs, noCache, executorName, jobStore, persistJob, onProgress, skipConnectivity, skipDoctor, lang, mcpConfig, verbose, retry, resume, runId, layeredStats, repeat, holdoutRatio, batch, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget, strictBaseline, effort, noDiagnostic, }: RunEvaluationOptions): Promise<{
|
|
121
124
|
report: Report | DryRunReport;
|
|
122
125
|
filePath: string | null;
|
|
123
126
|
}>;
|
|
@@ -133,7 +136,7 @@ export interface DryRunBatchReport extends DryRunBase {
|
|
|
133
136
|
artifacts: DryRunBatchSkill[];
|
|
134
137
|
}
|
|
135
138
|
export declare function buildVarianceData(runs: Report[], bootstrapSamples?: number, seed?: number): VarianceData | null;
|
|
136
|
-
export declare function runBatchEvaluation({ skillDir, model, outputDir, project, owner, tags, noJudge, dryRun, concurrency, timeoutMs, executorName, jobStore, persistJob, onProgress, onSkillProgress, skipConnectivity, skipDoctor, lang, mcpConfig, verbose, repeat, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget, noCache, strictBaseline, variantAllowedSkills, }: RunBatchEvaluationOptions): Promise<{
|
|
139
|
+
export declare function runBatchEvaluation({ skillDir, model, outputDir, project, owner, tags, noJudge, dryRun, concurrency, timeoutMs, executorName, jobStore, persistJob, onProgress, onSkillProgress, skipConnectivity, skipDoctor, lang, mcpConfig, verbose, repeat, holdoutRatio, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget, noCache, strictBaseline, variantAllowedSkills, }: RunBatchEvaluationOptions): Promise<{
|
|
137
140
|
report: BatchEvaluationReport | DryRunBatchReport;
|
|
138
141
|
filePath: string | null;
|
|
139
142
|
}>;
|
|
@@ -8,7 +8,7 @@ import { buildDryRunTaskReport, prepareEvaluationRun, } from './evaluation-prepa
|
|
|
8
8
|
import { executeEvaluationPipeline } from './evaluation-pipeline.js';
|
|
9
9
|
import { findSaturationPoint } from '../analysis/saturation.js';
|
|
10
10
|
import { bootstrapMeanCI, DEFAULT_BOOTSTRAP_ALPHA, DEFAULT_BOOTSTRAP_SAMPLES } from '../eval-core/bootstrap.js';
|
|
11
|
-
export async function runEvaluation({ samplesPath, skillDir, variantSpecs = [], model = DEFAULT_MODEL, outputDir = DEFAULT_OUTPUT_DIR, project, owner, tags, noJudge = false, dryRun = false, concurrency = 1, timeoutMs, noCache = false, executorName = 'claude', jobStore = null, persistJob = true, onProgress = null, skipConnectivity = false, skipDoctor = false, lang = 'zh', mcpConfig, verbose = false, retry = 0, resume, runId, layeredStats = false, repeat, batch, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget, strictBaseline, effort, noDiagnostic, }) {
|
|
11
|
+
export async function runEvaluation({ samplesPath, skillDir, variantSpecs = [], model = DEFAULT_MODEL, outputDir = DEFAULT_OUTPUT_DIR, project, owner, tags, noJudge = false, dryRun = false, concurrency = 1, timeoutMs, noCache = false, executorName = 'claude', jobStore = null, persistJob = true, onProgress = null, skipConnectivity = false, skipDoctor = false, lang = 'zh', mcpConfig, verbose = false, retry = 0, resume, runId, layeredStats = false, repeat, holdoutRatio, batch, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget, strictBaseline, effort, noDiagnostic, }) {
|
|
12
12
|
// Unified judgeModels → derive single-judge fields for downstream pipeline / grading
|
|
13
13
|
// (which still operate on string `judgeModel` + `judgeExecutorName` fields per call).
|
|
14
14
|
const effectiveJudgeModels = judgeModels && judgeModels.length > 0
|
|
@@ -165,6 +165,7 @@ export async function runEvaluation({ samplesPath, skillDir, variantSpecs = [],
|
|
|
165
165
|
requires,
|
|
166
166
|
layeredStats,
|
|
167
167
|
repeat,
|
|
168
|
+
holdoutRatio,
|
|
168
169
|
batch,
|
|
169
170
|
judgeRepeat,
|
|
170
171
|
judgeModels,
|
|
@@ -366,7 +367,7 @@ export function buildVarianceData(runs, bootstrapSamples = DEFAULT_BOOTSTRAP_SAM
|
|
|
366
367
|
...(saturation ? { saturation } : {}),
|
|
367
368
|
};
|
|
368
369
|
}
|
|
369
|
-
export async function runBatchEvaluation({ skillDir, model = DEFAULT_MODEL, outputDir = DEFAULT_OUTPUT_DIR, project, owner, tags, noJudge = false, dryRun = false, concurrency = 1, timeoutMs, executorName = 'claude', jobStore = null, persistJob = true, onProgress = null, onSkillProgress = null, skipConnectivity = false, skipDoctor = false, lang = 'zh', mcpConfig, verbose = false, repeat, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget, noCache = false, strictBaseline, variantAllowedSkills, }) {
|
|
370
|
+
export async function runBatchEvaluation({ skillDir, model = DEFAULT_MODEL, outputDir = DEFAULT_OUTPUT_DIR, project, owner, tags, noJudge = false, dryRun = false, concurrency = 1, timeoutMs, executorName = 'claude', jobStore = null, persistJob = true, onProgress = null, onSkillProgress = null, skipConnectivity = false, skipDoctor = false, lang = 'zh', mcpConfig, verbose = false, repeat, holdoutRatio, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget, noCache = false, strictBaseline, variantAllowedSkills, }) {
|
|
370
371
|
// Same unified judge derivation as runEvaluation (downstream pipeline / report build
|
|
371
372
|
// still uses single judgeModel + judgeExecutorName per call).
|
|
372
373
|
const effectiveJudgeModels = judgeModels && judgeModels.length > 0
|
|
@@ -485,12 +486,14 @@ export async function runBatchEvaluation({ skillDir, model = DEFAULT_MODEL, outp
|
|
|
485
486
|
strictBaseline,
|
|
486
487
|
variantAllowedSkills,
|
|
487
488
|
runSingleEvaluation: async (options) => {
|
|
488
|
-
// repeat > 1 时走 runMultiple 做 variance; batch=true 标记让 meta.request
|
|
489
|
+
// repeat > 1 时走 runMultiple 做 variance; batch=true 标记让 meta.request 如实反映。
|
|
490
|
+
// holdoutRatio 显式转发(executeBatchEvaluationRuns 不一定把它塞进 options),否则
|
|
491
|
+
// batch 子报告 meta.request.holdoutRatio 丢失、过拟合门控对 batch 静默失效。
|
|
489
492
|
if (repeat && repeat > 1) {
|
|
490
|
-
const multi = await runMultiple({ ...options, repeat, batch: true, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget });
|
|
493
|
+
const multi = await runMultiple({ ...options, repeat, holdoutRatio, batch: true, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget });
|
|
491
494
|
return { report: multi.report, filePath: multi.filePath };
|
|
492
495
|
}
|
|
493
|
-
const result = await runEvaluation({ ...options, batch: true, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget });
|
|
496
|
+
const result = await runEvaluation({ ...options, batch: true, holdoutRatio, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget });
|
|
494
497
|
return { report: result.report, filePath: result.filePath };
|
|
495
498
|
},
|
|
496
499
|
});
|
|
@@ -171,6 +171,11 @@ function validateEvalConfig(parsed, configPath) {
|
|
|
171
171
|
};
|
|
172
172
|
assertPositiveIntOpt('repeat');
|
|
173
173
|
assertPositiveIntOpt('judgeRepeat');
|
|
174
|
+
if (obj.holdoutRatio !== undefined) {
|
|
175
|
+
if (typeof obj.holdoutRatio !== 'number' || !Number.isFinite(obj.holdoutRatio) || obj.holdoutRatio <= 0 || obj.holdoutRatio >= 1) {
|
|
176
|
+
throw new Error(`${configPath}: holdoutRatio must be a number in (0, 1)`);
|
|
177
|
+
}
|
|
178
|
+
}
|
|
174
179
|
if (obj.bootstrapSamples !== undefined) {
|
|
175
180
|
if (typeof obj.bootstrapSamples !== 'number' || !Number.isFinite(obj.bootstrapSamples) || obj.bootstrapSamples < 100) {
|
|
176
181
|
throw new Error(`${configPath}: bootstrapSamples must be a number ≥ 100`);
|
|
@@ -256,6 +261,7 @@ function validateEvalConfig(parsed, configPath) {
|
|
|
256
261
|
variants,
|
|
257
262
|
budget,
|
|
258
263
|
repeat: obj.repeat,
|
|
264
|
+
holdoutRatio: obj.holdoutRatio,
|
|
259
265
|
judgeRepeat: obj.judgeRepeat,
|
|
260
266
|
bootstrap: obj.bootstrap,
|
|
261
267
|
bootstrapSamples: obj.bootstrapSamples,
|
package/dist/renderer/summary.js
CHANGED
|
@@ -101,6 +101,30 @@ function computeMedianCVPercent(report) {
|
|
|
101
101
|
const stab = medianStabilityCV(report);
|
|
102
102
|
return stab ? stab.cv * 100 : null;
|
|
103
103
|
}
|
|
104
|
+
// Verdict caveats(过拟合 / 知识缺口)渲染进 pill —— 让 HTML 报告和 CLI 说同一件事:
|
|
105
|
+
// CLI 在 verbose rationale 里给这两条,HTML 之前只剩一个泛化后的 level、看不到触发原因。
|
|
106
|
+
// 用 result.caveats 的结构化数据 i18n,而不是重解析 zh rationale 串。
|
|
107
|
+
function renderVerdictCaveats(caveats, lang) {
|
|
108
|
+
if (!caveats)
|
|
109
|
+
return '';
|
|
110
|
+
const lines = [];
|
|
111
|
+
if (caveats.overfitting) {
|
|
112
|
+
const c = caveats.overfitting;
|
|
113
|
+
lines.push(lang === 'zh'
|
|
114
|
+
? `⚠ 过拟合敞口:${c.variant} 训练 ${c.trainScore.toFixed(2)} / 留出 ${c.holdoutScore.toFixed(2)}(差 ${c.gap.toFixed(2)}),提升可能不泛化`
|
|
115
|
+
: `⚠ Overfitting: ${c.variant} train ${c.trainScore.toFixed(2)} / holdout ${c.holdoutScore.toFixed(2)} (gap ${c.gap.toFixed(2)}) — gain may not generalize`);
|
|
116
|
+
}
|
|
117
|
+
if (caveats.gapSignal) {
|
|
118
|
+
const g = caveats.gapSignal;
|
|
119
|
+
const wm = g.testSetHash ? g.testSetHash.slice(0, 8) : (g.testSetPath ?? '');
|
|
120
|
+
lines.push(lang === 'zh'
|
|
121
|
+
? `知识缺口率 ${g.gapRatePct}%(test set ${wm},informational)`
|
|
122
|
+
: `Knowledge gap ${g.gapRatePct}% (test set ${wm}, informational)`);
|
|
123
|
+
}
|
|
124
|
+
if (lines.length === 0)
|
|
125
|
+
return '';
|
|
126
|
+
return `<div class="page-verdict-caveats">${lines.map((l) => `<span class="page-verdict-caveat">${e(l)}</span>`).join('')}</div>`;
|
|
127
|
+
}
|
|
104
128
|
export function renderVerdictPill(report, lang) {
|
|
105
129
|
let result;
|
|
106
130
|
try {
|
|
@@ -110,8 +134,16 @@ export function renderVerdictPill(report, lang) {
|
|
|
110
134
|
return '';
|
|
111
135
|
}
|
|
112
136
|
const level = result.level;
|
|
113
|
-
|
|
137
|
+
// representative = top-level worst pair(与 CLI 同口径),不是第一对。多 treatment 报告里
|
|
138
|
+
// worst pair 不一定是 perPair[0],用它才不会把错的 treatment 名写进结论。fallback 兼容旧路径。
|
|
139
|
+
const pair = result.representative ?? result.perPair?.[0];
|
|
114
140
|
const oneLine = verdictOneLine(level, lang, pair?.treatment, pair?.control);
|
|
141
|
+
// Δ/CI 证据必须跟文案指同一对:按 representative 匹配对应的 pairComparison(alpha 也走这对),
|
|
142
|
+
// 否则多 treatment 报告会出现「文案 t2、数字 t1」的混搭。匹配不到 / 无 representative 时 fallback [0]。
|
|
143
|
+
const pairComparisons = report.meta?.pairComparisons;
|
|
144
|
+
const activeComparison = (pair
|
|
145
|
+
? pairComparisons?.find((p) => p.treatment === pair.treatment && p.control === pair.control)
|
|
146
|
+
: undefined) ?? pairComparisons?.[0];
|
|
115
147
|
const tooltip = levelTooltip(level, lang);
|
|
116
148
|
const prefix = lang === 'zh' ? '测评结论' : 'Verdict';
|
|
117
149
|
// 机器可读 enum 永远是 level token; 显示给用户的文字按 lang i18n.
|
|
@@ -120,7 +152,7 @@ export function renderVerdictPill(report, lang) {
|
|
|
120
152
|
// hero 只放「答案」: 分差是 verdict 的核心证据数字, 单独一枚 chip。
|
|
121
153
|
// 评测规模 (用例数 × 轮次) 走「实验配置」section 的 subtitle 那条 canonical 路径,
|
|
122
154
|
// 不在 hero 里重复; CV / CI 走 chip tooltip + 方法学审计 / 波动表。
|
|
123
|
-
const ci =
|
|
155
|
+
const ci = activeComparison?.diffBootstrapCI;
|
|
124
156
|
const cvPct = computeMedianCVPercent(report);
|
|
125
157
|
const metrics = [];
|
|
126
158
|
if (ci) {
|
|
@@ -128,7 +160,7 @@ export function renderVerdictPill(report, lang) {
|
|
|
128
160
|
const cvSuffix = cvPct != null
|
|
129
161
|
? (lang === 'zh' ? `;多轮稳定性 CV=${cvPct.toFixed(1)}% (${cvPct < 5 ? '稳' : cvPct <= STABILITY_UNSTABLE_CV * 100 ? '中' : '不稳'})` : `; CV=${cvPct.toFixed(1)}% (${cvPct < 5 ? 'stable' : cvPct <= STABILITY_UNSTABLE_CV * 100 ? 'moderate' : 'unstable'})`)
|
|
130
162
|
: '';
|
|
131
|
-
const pctLabel = ciLevelLabel(
|
|
163
|
+
const pctLabel = ciLevelLabel(activeComparison?.alpha);
|
|
132
164
|
const ciTipBase = lang === 'zh'
|
|
133
165
|
? `实验组与对照组综合分均值差(Δ)。bootstrap ${pctLabel} 可信区间 [${ci.low}, ${ci.high}],${ci.significant ? '不含 0 = 差异显著' : '跨过 0 = 差异不显著'}${cvSuffix}`
|
|
134
166
|
: `Treatment minus control mean composite score (Δ). Bootstrap ${pctLabel} CI [${ci.low}, ${ci.high}], ${ci.significant ? 'excludes 0 ⇒ significant' : 'spans 0 ⇒ not significant'}${cvSuffix}`;
|
|
@@ -146,6 +178,7 @@ export function renderVerdictPill(report, lang) {
|
|
|
146
178
|
<span class="page-verdict-badge"><span class="page-verdict-badge-dot" aria-hidden="true">●</span>${e(levelDisplay)}</span>
|
|
147
179
|
<span class="page-verdict-text">${e(oneLine)}</span>
|
|
148
180
|
</div>
|
|
181
|
+
${renderVerdictCaveats(result.caveats, lang)}
|
|
149
182
|
${metricChips ? `<div class="page-verdict-metrics">${metricChips}</div>` : ''}
|
|
150
183
|
</section>`;
|
|
151
184
|
}
|
package/dist/types/eval.d.ts
CHANGED
|
@@ -220,6 +220,9 @@ export interface EvalConfig {
|
|
|
220
220
|
budget?: EvalBudget;
|
|
221
221
|
/** --repeat N. Multi-run variance analysis. */
|
|
222
222
|
repeat?: number;
|
|
223
|
+
/** --holdout-ratio R (0 < R < 1). Hold out a deterministic sample slice and
|
|
224
|
+
* report train vs holdout composite as a generalization / overfitting signal. */
|
|
225
|
+
holdoutRatio?: number;
|
|
223
226
|
/** --judge-repeat N. Each (sample × dimension) judged N times for self-consistency stddev. */
|
|
224
227
|
judgeRepeat?: number;
|
|
225
228
|
/** --bootstrap. Distribution-free CI per variant + pairwise diff. */
|
|
@@ -257,6 +260,10 @@ export interface EvaluationRequest {
|
|
|
257
260
|
dryRun: boolean;
|
|
258
261
|
/** --repeat N; 1 表示单次跑,> 1 走 runMultiple 做 variance 分析 */
|
|
259
262
|
repeat?: number;
|
|
263
|
+
/** --holdout-ratio R; 0 / 缺省表示不切分(默认)。> 0 时 report-finalize 在结果上
|
|
264
|
+
* post-hoc 切出 train / holdout 子集算综合分(`report.analysis.holdout`),供 verdict
|
|
265
|
+
* 的过拟合门控读取。see src/eval-core/holdout.ts */
|
|
266
|
+
holdoutRatio?: number;
|
|
260
267
|
/** --batch; default absent/false. True means skill-batch mode. */
|
|
261
268
|
batch?: boolean;
|
|
262
269
|
/** --judge-repeat N; 每条 sample × dimension 用 LLM judge 跑 N 次, 输出 stddev. 默认 1 (单次). */
|
package/dist/types/report.d.ts
CHANGED
|
@@ -480,6 +480,34 @@ export interface AnalysisResult {
|
|
|
480
480
|
* (capability / difficulty / construct / provenance); persisted on report
|
|
481
481
|
* for studio to surface coverage gaps. See docs/specs/sample-design-spec.md. */
|
|
482
482
|
sampleQuality?: SampleQualityAggregate;
|
|
483
|
+
/** Opt-in train/holdout generalization breakdown (`omk eval --holdout-ratio`).
|
|
484
|
+
* Absent on default runs; present only when a holdout ratio was requested. */
|
|
485
|
+
holdout?: HoldoutBreakdown;
|
|
486
|
+
}
|
|
487
|
+
/** Train vs holdout composite breakdown for `omk eval --holdout-ratio`.
|
|
488
|
+
* Computed post-hoc from `report.results` by `computeHoldoutBreakdown`
|
|
489
|
+
* (`src/eval-core/holdout.ts`), sharing the same testSetHash watermark as
|
|
490
|
+
* gapReports (gap-spec §7.1). A large train − holdout composite gap is the
|
|
491
|
+
* sample-set-overfitting signal the verdict's overfitting gate reads. */
|
|
492
|
+
export interface HoldoutBreakdown {
|
|
493
|
+
/** Held-out fraction requested via --holdout-ratio. */
|
|
494
|
+
ratio: number;
|
|
495
|
+
/** true when either side fell below the minimum subset size → scored full-set,
|
|
496
|
+
* no usable split. `perVariant` is empty and the verdict gate stays inert. */
|
|
497
|
+
disabled?: boolean;
|
|
498
|
+
/** Per-variant train vs holdout composite (1-5 scale). `*Count` is the authored
|
|
499
|
+
* split size; `*Scorable` is how many of those actually produced a composite (> 0)
|
|
500
|
+
* — they diverge under partial errors, and the overfitting gate trusts `*Scorable`. */
|
|
501
|
+
perVariant: Record<string, {
|
|
502
|
+
trainScore: number;
|
|
503
|
+
holdoutScore: number;
|
|
504
|
+
trainCount: number;
|
|
505
|
+
holdoutCount: number;
|
|
506
|
+
trainScorable: number;
|
|
507
|
+
holdoutScorable: number;
|
|
508
|
+
}>;
|
|
509
|
+
testSetPath?: string | null;
|
|
510
|
+
testSetHash?: string | null;
|
|
483
511
|
}
|
|
484
512
|
/** Aggregated sample design coverage stats. Built by
|
|
485
513
|
* `buildSampleQualityAggregate(samples)` from `Sample.capability` /
|
|
@@ -502,6 +530,27 @@ export interface SampleQualityAggregate {
|
|
|
502
530
|
sampleCountWithDifficulty: number;
|
|
503
531
|
sampleCountWithConstruct: number;
|
|
504
532
|
sampleCountWithProvenance: number;
|
|
533
|
+
/** Relative-balance / skew of the sample set (derived from the distributions
|
|
534
|
+
* above). Flags over-representation — "70% of samples are easy" — without an
|
|
535
|
+
* external denominator. Diagnostic only; never feeds grading / judge / verdict. */
|
|
536
|
+
representativeness?: Representativeness;
|
|
537
|
+
}
|
|
538
|
+
/** Distribution skew over what the sample set declares. Pure relative balance —
|
|
539
|
+
* there is no authored "expected" capability list to measure absolute coverage
|
|
540
|
+
* against (capabilities are free-form strings), so this reports concentration
|
|
541
|
+
* (dominant bucket share, 0-1) and the dominant label per dimension. */
|
|
542
|
+
export interface Representativeness {
|
|
543
|
+
/** Distinct capabilities declared across the set. */
|
|
544
|
+
capabilityCount: number;
|
|
545
|
+
/** Dominant capability's share of all capability tags (0-1); 0 when none declared. */
|
|
546
|
+
capabilityConcentration: number;
|
|
547
|
+
dominantCapability?: string;
|
|
548
|
+
/** Dominant difficulty bucket's share of samples that declared a difficulty (0-1). */
|
|
549
|
+
difficultyConcentration: number;
|
|
550
|
+
dominantDifficulty?: 'easy' | 'medium' | 'hard';
|
|
551
|
+
/** Dominant construct's share of samples that declared a construct (0-1). */
|
|
552
|
+
constructConcentration: number;
|
|
553
|
+
dominantConstruct?: string;
|
|
505
554
|
}
|
|
506
555
|
export interface HedgingVerdict {
|
|
507
556
|
isUncertainty: boolean;
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "oh-my-knowledge",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.42.0",
|
|
4
4
|
"packageManager": "yarn@4.16.0",
|
|
5
5
|
"description": "Evaluation framework for LLM knowledge inputs — prompts, RAG corpora, skills, agent workflows. Fix the model, vary the artifact. Built-in statistical rigor: bootstrap CI, Krippendorff α, length-debias, saturation curves.",
|
|
6
6
|
"type": "module",
|