oh-my-knowledge 0.41.0 → 0.43.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +7 -2
- package/README.zh.md +7 -2
- package/dist/analysis/report-diagnostics.d.ts +8 -1
- package/dist/analysis/report-diagnostics.js +82 -1
- package/dist/artifact-graph/doctor.d.ts +21 -0
- package/dist/artifact-graph/doctor.js +569 -0
- package/dist/assets/agent-skills/omk/SKILL.md +9 -9
- package/dist/assets/agent-skills/omk/references/commands.md +2 -1
- package/dist/authoring/evolver.d.ts +3 -14
- package/dist/authoring/evolver.js +1 -52
- package/dist/authoring/generator.d.ts +24 -0
- package/dist/authoring/generator.js +33 -6
- package/dist/cli/commands/doctor.js +62 -61
- package/dist/cli/commands/eval/index.d.ts +1 -0
- package/dist/cli/commands/eval/index.js +60 -7
- package/dist/cli/commands/init.js +11 -7
- package/dist/cli/commands/observe/index.d.ts +2 -2
- package/dist/cli/commands/observe/index.js +8 -7
- package/dist/cli/commands/sample.js +22 -16
- package/dist/cli/lib/i18n-dict/common.d.ts +1 -1
- package/dist/cli/lib/i18n-dict/common.js +8 -0
- package/dist/cli/lib/i18n-dict/help.js +12 -10
- package/dist/cli/lib/i18n-dict/init.d.ts +1 -1
- package/dist/cli/lib/i18n-dict/init.js +14 -11
- package/dist/cli/lib/i18n-dict/run.d.ts +1 -1
- package/dist/cli/lib/i18n-dict/run.js +4 -0
- package/dist/cli/lib/parse-run-config/samples-discovery.d.ts +5 -7
- package/dist/cli/lib/parse-run-config/samples-discovery.js +10 -32
- package/dist/cli/lib/parse-run-config.d.ts +3 -0
- package/dist/cli/lib/parse-run-config.js +4 -4
- package/dist/cli/lib/resolve-skill-input.js +10 -12
- package/dist/doctor/messages.js +2 -2
- package/dist/eval-core/artifact-file-names.d.ts +15 -0
- package/dist/eval-core/artifact-file-names.js +46 -0
- package/dist/eval-core/evaluation-job.d.ts +2 -1
- package/dist/eval-core/evaluation-job.js +2 -1
- package/dist/eval-core/evaluation-reporting.js +9 -4
- package/dist/eval-core/holdout.d.ts +66 -0
- package/dist/eval-core/holdout.js +118 -0
- package/dist/eval-core/measurement-dirs.js +13 -7
- package/dist/eval-core/report-file-migration.d.ts +10 -0
- package/dist/eval-core/report-file-migration.js +90 -0
- package/dist/eval-core/verdict.d.ts +44 -1
- package/dist/eval-core/verdict.js +175 -13
- package/dist/eval-workflows/evaluation-pipeline/report-finalize.js +19 -1
- package/dist/eval-workflows/evaluation-pipeline/run-state.d.ts +2 -1
- package/dist/eval-workflows/evaluation-pipeline/run-state.js +2 -1
- package/dist/eval-workflows/evaluation-pipeline.d.ts +3 -1
- package/dist/eval-workflows/evaluation-pipeline.js +2 -1
- package/dist/eval-workflows/run-evaluation.d.ts +5 -2
- package/dist/eval-workflows/run-evaluation.js +8 -5
- package/dist/inputs/eval-config.js +6 -0
- package/dist/inputs/sample-locator.d.ts +23 -0
- package/dist/inputs/sample-locator.js +195 -0
- package/dist/inputs/skill-loader.js +7 -17
- package/dist/observability/inbox.js +7 -3
- package/dist/renderer/summary.js +36 -3
- package/dist/server/report-server.js +10 -4
- package/dist/server/report-store.js +17 -9
- package/dist/server/skill-index.js +16 -11
- package/dist/types/artifact-graph.d.ts +93 -0
- package/dist/types/artifact-graph.js +1 -0
- package/dist/types/eval.d.ts +7 -0
- package/dist/types/index.d.ts +1 -0
- package/dist/types/index.js +1 -0
- package/dist/types/report.d.ts +49 -0
- package/package.json +1 -1
|
@@ -9,6 +9,7 @@ import { CliExit } from '../lib/cli-exit.js';
|
|
|
9
9
|
import { tCli } from '../lib/i18n.js';
|
|
10
10
|
import { projectReportsDir, globalReportsDir } from '../../eval-core/measurement-dirs.js';
|
|
11
11
|
import { loadSamples, parseYaml, listSampleFilesInDir } from '../../inputs/load-samples.js';
|
|
12
|
+
import { defaultFlatSkillSamplesFile, defaultSkillLocalSamplesFile, findFlatSkillSamplesPath, findSkillSamplesPath, } from '../../inputs/sample-locator.js';
|
|
12
13
|
import { hashSample } from '../../eval-core/evaluation-reporting.js';
|
|
13
14
|
import { hashArtifactSource } from '../../inputs/content-hash.js';
|
|
14
15
|
function isRecord(value) {
|
|
@@ -233,23 +234,23 @@ async function runSampleFix(args, flags, lang) {
|
|
|
233
234
|
console.error(lang === 'zh' ? '请指定 skill 路径,如: omk sample skills/my-skill/SKILL.md --fix' : 'Specify skill path: omk sample skills/my-skill/SKILL.md --fix');
|
|
234
235
|
throw new CliExit(1);
|
|
235
236
|
}
|
|
236
|
-
const
|
|
237
|
-
|
|
238
|
-
|
|
237
|
+
const { resolveSkillInput } = await import('../lib/resolve-skill-input.js');
|
|
238
|
+
let resolvedInput;
|
|
239
|
+
try {
|
|
240
|
+
resolvedInput = resolveSkillInput(skillPath, lang);
|
|
241
|
+
}
|
|
242
|
+
catch (err) {
|
|
243
|
+
console.error(err instanceof Error ? err.message : String(err));
|
|
239
244
|
throw new CliExit(1);
|
|
240
245
|
}
|
|
241
|
-
const
|
|
242
|
-
const skillDir = isDir ? dirname(resolvedSkillPath) : dirname(resolvedSkillPath);
|
|
243
|
-
const samplesInput = isDir
|
|
244
|
-
? join(skillDir, '.omk')
|
|
245
|
-
: resolve('eval-samples.json');
|
|
246
|
+
const samplesInput = resolvedInput.samplesPath;
|
|
246
247
|
if (!existsSync(samplesInput)) {
|
|
247
248
|
console.error(lang === 'zh' ? `samples 路径不存在: ${samplesInput},先运行 omk sample 生成` : `Samples path not found: ${samplesInput}, run omk sample first`);
|
|
248
249
|
throw new CliExit(1);
|
|
249
250
|
}
|
|
250
|
-
const defaultTreatmentName =
|
|
251
|
-
? basename(skillDir)
|
|
252
|
-
: basename(
|
|
251
|
+
const defaultTreatmentName = resolvedInput.isDirectorySkill
|
|
252
|
+
? basename(resolvedInput.skillDir)
|
|
253
|
+
: basename(resolvedInput.skillPath, extname(resolvedInput.skillPath));
|
|
253
254
|
const treatmentName = flags.treatment ?? defaultTreatmentName;
|
|
254
255
|
process.stderr.write(lang === 'zh' ? `🔍 正在查找 ${treatmentName} 的最新评测报告...\n` : `🔍 Scanning latest report for ${treatmentName}...\n`);
|
|
255
256
|
// 显式 --reports-dir 固定该目录;默认 overlay(项目 .omk/reports 盖全局),findByVariant 记录优先看项目、
|
|
@@ -277,7 +278,7 @@ async function runSampleFix(args, flags, lang) {
|
|
|
277
278
|
throw new CliExit(1);
|
|
278
279
|
}
|
|
279
280
|
const samples = loadedSamples.samples;
|
|
280
|
-
const skillContent = readFileSync(
|
|
281
|
+
const skillContent = readFileSync(resolvedInput.skillPath, 'utf-8');
|
|
281
282
|
const sampleDesignIds = collectSampleDesignFailureIds(report, treatmentName);
|
|
282
283
|
const sampleDesignCount = sampleDesignIds.size;
|
|
283
284
|
if (sampleDesignCount === 0) {
|
|
@@ -286,7 +287,7 @@ async function runSampleFix(args, flags, lang) {
|
|
|
286
287
|
}
|
|
287
288
|
// 当前内容指纹走整树哈,与 eval 报告口径一致:dir-skill(用户传 .../SKILL.md)哈整棵 skill 目录、
|
|
288
289
|
// 单文件 .md 哈单文件字节。
|
|
289
|
-
const currentContentHash = hashArtifactSource(
|
|
290
|
+
const currentContentHash = hashArtifactSource(resolvedInput.isDirectorySkill ? resolvedInput.skillDir : resolvedInput.skillPath, resolvedInput.isDirectorySkill);
|
|
290
291
|
try {
|
|
291
292
|
assertFixReportMatchesCurrentInputs({
|
|
292
293
|
report,
|
|
@@ -435,24 +436,29 @@ async function runSample(args, flags, lang) {
|
|
|
435
436
|
let name;
|
|
436
437
|
let skillPath;
|
|
437
438
|
let samplesPath;
|
|
439
|
+
let existingSamplesPath;
|
|
438
440
|
const fullPath = join(skillDir, entry);
|
|
439
441
|
if (entry.endsWith('.md') && !entry.endsWith('.eval-samples.json')) {
|
|
440
442
|
name = entry.slice(0, -3);
|
|
441
443
|
skillPath = fullPath;
|
|
442
|
-
samplesPath =
|
|
444
|
+
samplesPath = defaultFlatSkillSamplesFile(skillDir, name);
|
|
445
|
+
existingSamplesPath = findFlatSkillSamplesPath(skillDir, name);
|
|
443
446
|
}
|
|
444
447
|
else if (statSync(fullPath).isDirectory()) {
|
|
445
448
|
const skillMd = join(fullPath, 'SKILL.md');
|
|
446
449
|
if (!existsSync(skillMd))
|
|
447
450
|
continue;
|
|
451
|
+
if (existsSync(join(skillDir, `${entry}.md`)))
|
|
452
|
+
continue;
|
|
448
453
|
name = entry;
|
|
449
454
|
skillPath = skillMd;
|
|
450
|
-
samplesPath =
|
|
455
|
+
samplesPath = defaultSkillLocalSamplesFile(fullPath);
|
|
456
|
+
existingSamplesPath = findSkillSamplesPath(fullPath);
|
|
451
457
|
}
|
|
452
458
|
else {
|
|
453
459
|
continue;
|
|
454
460
|
}
|
|
455
|
-
if (
|
|
461
|
+
if (existingSamplesPath) {
|
|
456
462
|
process.stderr.write(tCli('cli.gen.skill_skipped_existing', lang, { name }));
|
|
457
463
|
continue;
|
|
458
464
|
}
|
|
@@ -1,3 +1,3 @@
|
|
|
1
1
|
import type { CliMessage } from './types.js';
|
|
2
|
-
export type CommonMessageKey = 'cli.common.unknown_domain' | 'cli.common.error_prefix' | 'cli.common.skill_dir_not_found' | 'cli.common.skill_file_not_found' | 'cli.common.skill_dir_no_skill_md' | 'cli.common.report_not_found' | 'cli.common.no_judge_model' | 'cli.common.judge_models_single_only' | 'cli.common.warn_load_samples_failed' | 'cli.update.new_version_available' | 'cli.update.box_title' | 'cli.update.box_version_line' | 'cli.update.box_upgrade_line' | 'cli.update.box_silence_line' | 'cli.observe.view_hint' | 'cli.observe.observation_recorded' | 'cli.observe.production_gap' | 'cli.studio.started' | 'cli.studio.stop_hint' | 'cli.studio.open_failed' | 'cli.doctor.no_skill_found' | 'cli.doctor.samples_detected' | 'cli.doctor.progress_skill_start' | 'cli.doctor.progress_skill_done';
|
|
2
|
+
export type CommonMessageKey = 'cli.common.unknown_domain' | 'cli.common.error_prefix' | 'cli.common.skill_dir_not_found' | 'cli.common.skill_file_not_found' | 'cli.common.skill_dir_no_skill_md' | 'cli.common.report_not_found' | 'cli.common.no_judge_model' | 'cli.common.judge_models_single_only' | 'cli.common.warn_load_samples_failed' | 'cli.common.deprecated_skill_samples_path' | 'cli.common.samples_not_found' | 'cli.update.new_version_available' | 'cli.update.box_title' | 'cli.update.box_version_line' | 'cli.update.box_upgrade_line' | 'cli.update.box_silence_line' | 'cli.observe.view_hint' | 'cli.observe.observation_recorded' | 'cli.observe.production_gap' | 'cli.studio.started' | 'cli.studio.stop_hint' | 'cli.studio.open_failed' | 'cli.doctor.no_skill_found' | 'cli.doctor.samples_detected' | 'cli.doctor.progress_skill_start' | 'cli.doctor.progress_skill_done';
|
|
3
3
|
export declare const commonDict: Record<CommonMessageKey, CliMessage>;
|
|
@@ -35,6 +35,14 @@ export const commonDict = {
|
|
|
35
35
|
zh: '⚠ 加载 samples 文件失败 ({path}): {message}\n',
|
|
36
36
|
en: '⚠ Failed to load samples file ({path}): {message}\n',
|
|
37
37
|
},
|
|
38
|
+
'cli.common.deprecated_skill_samples_path': {
|
|
39
|
+
zh: '⚠ 发现旧的目录 skill 用例位置:{oldPath}。目录 skill 的私有用例已改放到 {newPath},请迁移。\n',
|
|
40
|
+
en: '⚠ Found deprecated directory skill samples path: {oldPath}. Directory skill samples now live at {newPath}; please migrate.\n',
|
|
41
|
+
},
|
|
42
|
+
'cli.common.samples_not_found': {
|
|
43
|
+
zh: '未找到评测用例:{path}。请通过 --samples 指定文件,或创建项目级 eval-samples.json;单 treatment 目录 skill 请使用 <skill>/.omk/samples.json。',
|
|
44
|
+
en: 'Eval samples not found: {path}. Pass --samples, create project-level eval-samples.json, or use <skill>/.omk/samples.json for a single-treatment directory skill.',
|
|
45
|
+
},
|
|
38
46
|
'cli.update.new_version_available': {
|
|
39
47
|
zh: '\n💡 新版本可用:{old} → {new},运行 npm i -g oh-my-knowledge@latest 升级\n\n',
|
|
40
48
|
en: '\n💡 New version available: {old} → {new}, run npm i -g oh-my-knowledge@latest to upgrade\n\n',
|
|
@@ -170,20 +170,21 @@ omk sample——生成或补齐 eval-samples 评测用例
|
|
|
170
170
|
omk sample --batch [--skill-dir <dir>] [options]
|
|
171
171
|
|
|
172
172
|
输出位置(默认):
|
|
173
|
-
|
|
174
|
-
|
|
173
|
+
目录 skill(<skill>/SKILL.md) → <skill>/.omk/samples.json(omk 标准约定)
|
|
174
|
+
扁平 .md(单次) → 当前目录的 eval-samples.json(项目级兜底)
|
|
175
|
+
扁平 .md(--batch) → <skill-dir>/<name>.eval-samples.json(兼容 paired 布局)
|
|
175
176
|
|
|
176
177
|
选项:
|
|
177
178
|
--count <n> 强制生成 N 条(不指定时由 LLM 按 skill 类型自动判断:
|
|
178
179
|
工作流型 6-8 条 / 原子型 4-6 条 / 混合型 5-7 条)
|
|
179
180
|
--model <name> 生成模型(默认:opus;lean+effort-low 已自动开,想省钱可改 sonnet/haiku)
|
|
180
181
|
--focus <text> 自然语言指定希望覆盖的场景(追加到 prompt,优先级高于自由发挥)
|
|
181
|
-
--batch 为 skill 目录下缺少
|
|
182
|
+
--batch 为 skill 目录下缺少 samples 的 skill 批量生成
|
|
182
183
|
--skill-dir <path> skill 目录(batch 使用,默认:skills)
|
|
183
184
|
|
|
184
185
|
示例:
|
|
185
|
-
omk
|
|
186
|
-
omk
|
|
186
|
+
omk sample skills/req-tool.md
|
|
187
|
+
omk sample skills/req-tool.md --count 8 \\
|
|
187
188
|
--focus "重点覆盖 tag 查询走 PROJECT 空 → WORKSPACE 兜底的多步流程,以及 search 失败的错误路径"
|
|
188
189
|
`,
|
|
189
190
|
en: `
|
|
@@ -194,20 +195,21 @@ Usage:
|
|
|
194
195
|
omk sample --batch [--skill-dir <dir>] [options]
|
|
195
196
|
|
|
196
197
|
Output path (default):
|
|
197
|
-
<skill>/SKILL.md
|
|
198
|
-
|
|
198
|
+
directory skill (<skill>/SKILL.md) → <skill>/.omk/samples.json (omk standard layout)
|
|
199
|
+
flat .md (single) → ./eval-samples.json in current directory (project fallback)
|
|
200
|
+
flat .md (--batch) → <skill-dir>/<name>.eval-samples.json (compatible paired layout)
|
|
199
201
|
|
|
200
202
|
Options:
|
|
201
203
|
--count <n> Force N samples (omit to let LLM auto-decide by skill type:
|
|
202
204
|
workflow 6-8 / atomic 4-6 / mixed 5-7)
|
|
203
205
|
--model <name> Generation model (default: opus; lean+effort-low applied; pass --model sonnet/haiku to save cost)
|
|
204
206
|
--focus <text> Natural-language scenario hints appended to the prompt (overrides freeform diversity)
|
|
205
|
-
--batch Generate for skills that are missing
|
|
207
|
+
--batch Generate for skills that are missing samples
|
|
206
208
|
--skill-dir <path> Skill directory for batch mode (default: skills)
|
|
207
209
|
|
|
208
210
|
Examples:
|
|
209
|
-
omk
|
|
210
|
-
omk
|
|
211
|
+
omk sample skills/req-tool.md
|
|
212
|
+
omk sample skills/req-tool.md --count 8 \\
|
|
211
213
|
--focus "Cover PROJECT-empty → WORKSPACE-fallback multi-step tag lookup and the search-failure error path"
|
|
212
214
|
`,
|
|
213
215
|
},
|
|
@@ -1,3 +1,3 @@
|
|
|
1
1
|
import type { CliMessage } from './types.js';
|
|
2
|
-
export type InitMessageKey = 'cli.init.scaffolded' | 'cli.init.next_steps_title' | 'cli.init.
|
|
2
|
+
export type InitMessageKey = 'cli.init.scaffolded' | 'cli.init.next_steps_title' | 'cli.init.next_step_run' | 'cli.init.next_step_executor' | 'cli.init.next_step_customize' | 'cli.init.note_codex_executor';
|
|
3
3
|
export declare const initDict: Record<InitMessageKey, CliMessage>;
|
|
@@ -1,23 +1,26 @@
|
|
|
1
1
|
export const initDict = {
|
|
2
2
|
'cli.init.scaffolded': {
|
|
3
|
-
zh: '已初始化 omk
|
|
3
|
+
zh: '已初始化 omk 项目:{dir}',
|
|
4
4
|
en: 'omk project initialized at: {dir}',
|
|
5
5
|
},
|
|
6
6
|
'cli.init.next_steps_title': {
|
|
7
|
-
zh: '
|
|
7
|
+
zh: '下一步:',
|
|
8
8
|
en: 'Next steps:',
|
|
9
9
|
},
|
|
10
|
-
|
|
11
|
-
|
|
12
|
-
|
|
10
|
+
// 先让用户「无需改任何文件直接跑通」——脚手架的用例与 skill 本身可跑(已过合规校验),
|
|
11
|
+
// 跑出第一份报告是冷启动最该先发生的事;「换成你自己的」放到跑通之后。这也消除了
|
|
12
|
+
// 主 README「不用改任何文件」与旧 init「先编辑」的矛盾。
|
|
13
|
+
'cli.init.next_step_run': {
|
|
14
|
+
zh: ' 1. 直接跑通(无需先改任何文件):omk eval --control code-review-v1 --treatment code-review-v2',
|
|
15
|
+
en: ' 1. Run it as-is (no edits needed): omk eval --control code-review-v1 --treatment code-review-v2',
|
|
13
16
|
},
|
|
14
|
-
'cli.init.
|
|
15
|
-
zh: '
|
|
16
|
-
en: '
|
|
17
|
+
'cli.init.next_step_executor': {
|
|
18
|
+
zh: ' 默认执行器与评委用 claude CLI,需先装好并登录;想换别的模型或离线跑(无需 API key)见文档「执行器」。',
|
|
19
|
+
en: ' The default executor and judge use the claude CLI (install and log in first); to use another model or run offline (no API key) see the Executors docs.',
|
|
17
20
|
},
|
|
18
|
-
'cli.init.
|
|
19
|
-
zh: '
|
|
20
|
-
en: '
|
|
21
|
+
'cli.init.next_step_customize': {
|
|
22
|
+
zh: ' 2. 跑通后,把 skills/code-review-v1/SKILL.md 和 skills/code-review-v2/SKILL.md 与 eval-samples.json 换成你自己的 skill 和用例',
|
|
23
|
+
en: ' 2. Once it runs, replace skills/code-review-v1/SKILL.md and skills/code-review-v2/SKILL.md and eval-samples.json with your own skills and cases',
|
|
21
24
|
},
|
|
22
25
|
'cli.init.note_codex_executor': {
|
|
23
26
|
zh: '\n注: omk 评测时把 SKILL.md 整文(含 frontmatter)作为 system prompt 注入——跨 executor 一致(claude / codex / openai-api / gemini 都走同一条路径,不依赖任何 executor 的 native skill auto-discovery 或 Skill 工具机制)。frontmatter 在 prompt 头部对 model 行为无显著影响。\n模板带 Claude Code 兼容的 frontmatter(name + description)是为了让同一份 directory-skill 也能 deploy 到 Claude Code:把整个目录复制到 ~/.claude/skills/code-review-v1/(整目录,不是单个 SKILL.md),Claude SDK 才能识别。这是 omk 评测之外的 bonus,一份文件双向 dogfood。',
|
|
@@ -1,3 +1,3 @@
|
|
|
1
1
|
import type { CliMessage } from './types.js';
|
|
2
|
-
export type RunMessageKey = 'cli.progress.preflight_starting' | 'cli.progress.sample_retry' | 'cli.progress.sample_error' | 'cli.progress.sample_executing' | 'cli.progress.sample_exec_done' | 'cli.progress.output_preview' | 'cli.progress.judging' | 'cli.progress.judged' | 'cli.progress.skipped' | 'cli.progress.sample_done' | 'cli.progress.sample_failed_done' | 'cli.run.invalid_repeat' | 'cli.run.invalid_judge_repeat' | 'cli.run.no_debias_length_active' | 'cli.run.invalid_bootstrap_samples' | 'cli.run.bootstrap_samples_too_large' | 'cli.run.dry_run_no_scores' | 'cli.run.skill_section' | 'cli.run.run_section' | 'cli.run.batch_complete' | 'cli.run.batch_verdict_header' | 'cli.run.batch_child_report_missing' | 'cli.run.eval_complete' | 'cli.run.tally' | 'cli.run.report_saved' | 'cli.run.evidence_recorded' | 'cli.run.evidence_recorded_unbound' | 'cli.run.report_only_gate_skipped' | 'cli.run.report_server_running' | 'cli.run.report_server_view' | 'cli.run.report_server_stop' | 'cli.run.no_serve_in_non_tty' | 'cli.run.no_serve_view_hint' | 'cli.run.gold_load_failed' | 'cli.run.gold_load_issue' | 'cli.run.contamination_warning' | 'cli.run.skip_connectivity_warning';
|
|
2
|
+
export type RunMessageKey = 'cli.progress.preflight_starting' | 'cli.progress.sample_retry' | 'cli.progress.sample_error' | 'cli.progress.sample_executing' | 'cli.progress.sample_exec_done' | 'cli.progress.output_preview' | 'cli.progress.judging' | 'cli.progress.judged' | 'cli.progress.skipped' | 'cli.progress.sample_done' | 'cli.progress.sample_failed_done' | 'cli.run.invalid_repeat' | 'cli.run.invalid_holdout_ratio' | 'cli.run.invalid_judge_repeat' | 'cli.run.no_debias_length_active' | 'cli.run.invalid_bootstrap_samples' | 'cli.run.bootstrap_samples_too_large' | 'cli.run.dry_run_no_scores' | 'cli.run.skill_section' | 'cli.run.run_section' | 'cli.run.batch_complete' | 'cli.run.batch_verdict_header' | 'cli.run.batch_child_report_missing' | 'cli.run.eval_complete' | 'cli.run.tally' | 'cli.run.report_saved' | 'cli.run.evidence_recorded' | 'cli.run.evidence_recorded_unbound' | 'cli.run.report_only_gate_skipped' | 'cli.run.report_server_running' | 'cli.run.report_server_view' | 'cli.run.report_server_stop' | 'cli.run.no_serve_in_non_tty' | 'cli.run.no_serve_view_hint' | 'cli.run.gold_load_failed' | 'cli.run.gold_load_issue' | 'cli.run.contamination_warning' | 'cli.run.skip_connectivity_warning';
|
|
3
3
|
export declare const runDict: Record<RunMessageKey, CliMessage>;
|
|
@@ -51,6 +51,10 @@ export const runDict = {
|
|
|
51
51
|
zh: '⚠ --judge-repeat "{value}" 无效 (期望 ≥ 1 的整数), 已按 1 次 judge 执行\n',
|
|
52
52
|
en: '⚠ --judge-repeat "{value}" is invalid (expected an integer ≥ 1), falling back to 1 judge call\n',
|
|
53
53
|
},
|
|
54
|
+
'cli.run.invalid_holdout_ratio': {
|
|
55
|
+
zh: '⚠ --holdout-ratio "{value}" 无效 (期望 0 到 1 之间的小数), 已忽略、不做 holdout 切分\n',
|
|
56
|
+
en: '⚠ --holdout-ratio "{value}" is invalid (expected a fraction in (0, 1)), ignored — no holdout split\n',
|
|
57
|
+
},
|
|
54
58
|
'cli.run.no_debias_length_active': {
|
|
55
59
|
zh: 'ℹ --no-debias-length 已生效:judge prompt 去掉长度去偏指令(debias-off 变体),hash 与默认开启时不同。\n',
|
|
56
60
|
en: 'ℹ --no-debias-length is active: the judge prompt drops the length-debias instruction (debias-off variant); its hash differs from the default.\n',
|
|
@@ -1,18 +1,16 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* 当 CLI 未传 `--samples` 时,自动发现 sample 路径。
|
|
3
3
|
*
|
|
4
|
-
*
|
|
5
|
-
* 1. treatment
|
|
6
|
-
* 2.
|
|
7
|
-
* 3. fallback `<skillDir>/<treatmentName>/.omk/`
|
|
8
|
-
* 4. 终极兜底 cwd 下的 `eval-samples.{json,yaml,yml}`
|
|
4
|
+
* 发现顺序:
|
|
5
|
+
* 1. 单 treatment → 找该 skill 私有 samples(`<skill>/.omk/`)或扁平 skill paired 文件
|
|
6
|
+
* 2. 终极兜底 cwd 下的项目级 `eval-samples.{json,yaml,yml}`
|
|
9
7
|
*
|
|
10
8
|
* `loadSamples` 在下游自己处理「文件 vs 目录」—— 目录模式 glob `*.{json,yaml,yml}`
|
|
11
9
|
* 合并,跳过 reserved 前缀(`report-` / `health-` / `_`)。所以一个 skill 可以把
|
|
12
10
|
* sample 拆到多文件(`workflow.json` + `platform.json`),也可以单 `samples.json`,
|
|
13
11
|
* 两种都行。
|
|
14
12
|
*
|
|
15
|
-
* 多 treatment
|
|
16
|
-
* 没有唯一答案)
|
|
13
|
+
* 多 treatment 评测不会触发 skill-local 自动发现(因为「找哪个 skill 的 bundled samples」
|
|
14
|
+
* 没有唯一答案),只回到项目级 samples 兜底。
|
|
17
15
|
*/
|
|
18
16
|
export declare function discoverSamplesPath(values: Record<string, unknown>, skillDir: string): string;
|
|
@@ -1,50 +1,28 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* 当 CLI 未传 `--samples` 时,自动发现 sample 路径。
|
|
3
3
|
*
|
|
4
|
-
*
|
|
5
|
-
* 1. treatment
|
|
6
|
-
* 2.
|
|
7
|
-
* 3. fallback `<skillDir>/<treatmentName>/.omk/`
|
|
8
|
-
* 4. 终极兜底 cwd 下的 `eval-samples.{json,yaml,yml}`
|
|
4
|
+
* 发现顺序:
|
|
5
|
+
* 1. 单 treatment → 找该 skill 私有 samples(`<skill>/.omk/`)或扁平 skill paired 文件
|
|
6
|
+
* 2. 终极兜底 cwd 下的项目级 `eval-samples.{json,yaml,yml}`
|
|
9
7
|
*
|
|
10
8
|
* `loadSamples` 在下游自己处理「文件 vs 目录」—— 目录模式 glob `*.{json,yaml,yml}`
|
|
11
9
|
* 合并,跳过 reserved 前缀(`report-` / `health-` / `_`)。所以一个 skill 可以把
|
|
12
10
|
* sample 拆到多文件(`workflow.json` + `platform.json`),也可以单 `samples.json`,
|
|
13
11
|
* 两种都行。
|
|
14
12
|
*
|
|
15
|
-
* 多 treatment
|
|
16
|
-
* 没有唯一答案)
|
|
13
|
+
* 多 treatment 评测不会触发 skill-local 自动发现(因为「找哪个 skill 的 bundled samples」
|
|
14
|
+
* 没有唯一答案),只回到项目级 samples 兜底。
|
|
17
15
|
*/
|
|
18
|
-
import {
|
|
19
|
-
import { existsSync, statSync } from 'node:fs';
|
|
16
|
+
import { findProjectSamplesFile, findSingleTreatmentSamplesPath } from '../../../inputs/sample-locator.js';
|
|
20
17
|
export function discoverSamplesPath(values, skillDir) {
|
|
21
18
|
const treatmentRaw = values.treatment;
|
|
22
19
|
const treatments = treatmentRaw
|
|
23
20
|
? treatmentRaw.split(',').map((v) => v.trim()).filter(Boolean)
|
|
24
21
|
: [];
|
|
25
22
|
if (treatments.length === 1) {
|
|
26
|
-
const
|
|
27
|
-
|
|
28
|
-
|
|
29
|
-
const treatmentDir = statSync(resolved).isDirectory() ? resolved : dirname(resolved);
|
|
30
|
-
const omkDir = join(treatmentDir, '.omk');
|
|
31
|
-
if (existsSync(omkDir))
|
|
32
|
-
return omkDir;
|
|
33
|
-
for (const name of ['eval-samples.json', 'eval-samples.yaml', 'eval-samples.yml']) {
|
|
34
|
-
if (existsSync(join(treatmentDir, name)))
|
|
35
|
-
return join(treatmentDir, name);
|
|
36
|
-
}
|
|
37
|
-
}
|
|
38
|
-
const omkDir = join(skillDir, expr, '.omk');
|
|
39
|
-
if (existsSync(omkDir))
|
|
40
|
-
return omkDir;
|
|
23
|
+
const samplesPath = findSingleTreatmentSamplesPath(treatments[0], skillDir, process.cwd());
|
|
24
|
+
if (samplesPath)
|
|
25
|
+
return samplesPath;
|
|
41
26
|
}
|
|
42
|
-
|
|
43
|
-
if (!existsSync(resolve(cwdFile))) {
|
|
44
|
-
if (existsSync(resolve('eval-samples.yaml')))
|
|
45
|
-
cwdFile = 'eval-samples.yaml';
|
|
46
|
-
else if (existsSync(resolve('eval-samples.yml')))
|
|
47
|
-
cwdFile = 'eval-samples.yml';
|
|
48
|
-
}
|
|
49
|
-
return cwdFile;
|
|
27
|
+
return findProjectSamplesFile(process.cwd()) ?? 'eval-samples.json';
|
|
50
28
|
}
|
|
@@ -45,6 +45,9 @@ export interface RunConfig {
|
|
|
45
45
|
retry?: number;
|
|
46
46
|
resume?: string;
|
|
47
47
|
layeredStats?: boolean;
|
|
48
|
+
/** --holdout-ratio R (0 < R < 1). Hold out a deterministic sample slice; report-finalize
|
|
49
|
+
* computes train vs holdout composite (report.analysis.holdout) for the overfitting gate. */
|
|
50
|
+
holdoutRatio?: number;
|
|
48
51
|
/** --judge-repeat N. Calls LLM judge N times per (sample × dimension). Default 1. */
|
|
49
52
|
judgeRepeat?: number;
|
|
50
53
|
/** Unified judge config. Always non-empty; 1 entry = single judge, ≥ 2 = ensemble.
|
|
@@ -41,10 +41,10 @@ export function parseRunConfig(values) {
|
|
|
41
41
|
? loadEvalConfig(values.config)
|
|
42
42
|
: null;
|
|
43
43
|
const skillDir = resolve(values['skill-dir'] ?? 'skills');
|
|
44
|
-
// 2) Resolve samples path: CLI > config >
|
|
45
|
-
//
|
|
46
|
-
// which skill's bundled samples to use. The dir form (loadSamples handles
|
|
47
|
-
// means a skill can split samples across multiple files
|
|
44
|
+
// 2) Resolve samples path: CLI > config > single-treatment skill-local discovery > cwd project default.
|
|
45
|
+
// Skill-local discovery only fires when exactly one --treatment is given, so omk knows
|
|
46
|
+
// which skill's bundled samples to use. The `<skill>/.omk/` dir form (loadSamples handles
|
|
47
|
+
// both file + dir) means a skill can split samples across multiple files.
|
|
48
48
|
const cliSamples = values.samples;
|
|
49
49
|
let samplesFile;
|
|
50
50
|
if (cliSamples) {
|
|
@@ -1,10 +1,11 @@
|
|
|
1
1
|
import { resolve, join, dirname, basename } from 'node:path';
|
|
2
2
|
import { existsSync, statSync } from 'node:fs';
|
|
3
3
|
import { tCli } from './i18n.js';
|
|
4
|
+
import { findFlatSkillSamplesPath, findProjectSamplesFile, findSkillSamplesPath, skillLocalSamplesDir, } from '../../inputs/sample-locator.js';
|
|
4
5
|
// 统一 skill 入参解析:既接受 SKILL.md 文件(老式 + flat skill),也接受 directory-skill
|
|
5
|
-
// 目录(skills/foo/ 自动找 foo/SKILL.md)
|
|
6
|
-
//
|
|
7
|
-
// .
|
|
6
|
+
// 目录(skills/foo/ 自动找 foo/SKILL.md)。目录-skill 以 `<skill>/.omk/` 为标准
|
|
7
|
+
// samples 命名空间;扁平 .md 兼容 paired sidecar,找不到时回到项目级
|
|
8
|
+
// `eval-samples.json`。
|
|
8
9
|
//
|
|
9
10
|
// 错误用 tCli 走 i18n,调用方直接 console.error err.message 给用户看,zh/en 都要正确。
|
|
10
11
|
export function resolveSkillInput(input, lang) {
|
|
@@ -26,16 +27,13 @@ export function resolveSkillInput(input, lang) {
|
|
|
26
27
|
skillPath = resolved;
|
|
27
28
|
skillDir = dirname(resolved);
|
|
28
29
|
}
|
|
29
|
-
const omkDir = join(skillDir, '.omk');
|
|
30
|
-
const candidates = [
|
|
31
|
-
...(existsSync(omkDir) ? [omkDir] : []),
|
|
32
|
-
join(skillDir, 'eval-samples.json'),
|
|
33
|
-
join(skillDir, 'eval-samples.yaml'),
|
|
34
|
-
join(skillDir, 'eval-samples.yml'),
|
|
35
|
-
];
|
|
36
|
-
// fallback 到 .omk/ 目录:上游会报 "no sample files found in directory" 引导用户创建
|
|
37
|
-
const samplesPath = candidates.find(existsSync) ?? omkDir;
|
|
38
30
|
// 形态以解析后的 skillPath 命名为准:目录-skill 的 skillPath 总是 `.../SKILL.md`。
|
|
39
31
|
const isDirectorySkill = basename(skillPath) === 'SKILL.md';
|
|
32
|
+
const flatSkillName = basename(skillPath).replace(/\.md$/i, '');
|
|
33
|
+
const samplesPath = isDirectorySkill
|
|
34
|
+
? findSkillSamplesPath(skillDir) ?? skillLocalSamplesDir(skillDir)
|
|
35
|
+
: findFlatSkillSamplesPath(skillDir, flatSkillName)
|
|
36
|
+
?? findProjectSamplesFile(process.cwd())
|
|
37
|
+
?? 'eval-samples.json';
|
|
40
38
|
return { skillPath, skillDir, samplesPath, isDirectorySkill };
|
|
41
39
|
}
|
package/dist/doctor/messages.js
CHANGED
|
@@ -133,8 +133,8 @@ export const DOCTOR_MESSAGES = {
|
|
|
133
133
|
en: 'directory-skill missing SKILL.md entry file',
|
|
134
134
|
},
|
|
135
135
|
'cli.doctor.skill_metadata.hint.frontmatter': {
|
|
136
|
-
zh: 'front-matter 用 YAML
|
|
137
|
-
en: 'front-matter uses YAML syntax (key: value or - item). See examples/
|
|
136
|
+
zh: 'front-matter 用 YAML 语法,key: value 或 - item 形式。可参考 examples/skill-map-showcase/skills/release-readiness 的目录式 skill 写法',
|
|
137
|
+
en: 'front-matter uses YAML syntax (key: value or - item). See examples/skill-map-showcase/skills/release-readiness for a directory-skill reference',
|
|
138
138
|
},
|
|
139
139
|
'cli.doctor.skill_metadata.hint.hardrules': {
|
|
140
140
|
zh: 'hardRules 必须写在 SKILL.md front-matter 中,格式为 hardRules: [{ id, rule, expectedBehavior }];id 要稳定且唯一,expectedBehavior 写可观察行为。',
|
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
export declare const REPORT_FILE_SUFFIX = ".report.json";
|
|
2
|
+
export declare const GRAPH_FILE_SUFFIX = ".graph.json";
|
|
3
|
+
export declare const CARD_FILE_SUFFIX = ".card.md";
|
|
4
|
+
export declare function safeArtifactFileStem(id: string): string;
|
|
5
|
+
export declare function reportFileName(stem: string): string;
|
|
6
|
+
export declare function reportFilePath(dir: string, stem: string): string;
|
|
7
|
+
export declare function reportFileStem(fileName: string): string | null;
|
|
8
|
+
export declare function isReportFileName(fileName: string): boolean;
|
|
9
|
+
export declare function graphFileName(stem: string): string;
|
|
10
|
+
export declare function cardFileName(stem: string): string;
|
|
11
|
+
export declare function runTimestamp(date?: Date): string;
|
|
12
|
+
export declare function randomRunToken(): string;
|
|
13
|
+
export declare function runFileSuffix(counter?: number): string;
|
|
14
|
+
export declare function stripDomainPrefix(id: string, domain: string): string;
|
|
15
|
+
export declare function doctorReportFileStem(skillName: string, reportId: string): string;
|
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
import { join } from 'node:path';
|
|
2
|
+
export const REPORT_FILE_SUFFIX = '.report.json';
|
|
3
|
+
export const GRAPH_FILE_SUFFIX = '.graph.json';
|
|
4
|
+
export const CARD_FILE_SUFFIX = '.card.md';
|
|
5
|
+
export function safeArtifactFileStem(id) {
|
|
6
|
+
return id.replaceAll(/[/\\:*?"<>|]/g, '_');
|
|
7
|
+
}
|
|
8
|
+
export function reportFileName(stem) {
|
|
9
|
+
return `${safeArtifactFileStem(stem)}${REPORT_FILE_SUFFIX}`;
|
|
10
|
+
}
|
|
11
|
+
export function reportFilePath(dir, stem) {
|
|
12
|
+
return join(dir, reportFileName(stem));
|
|
13
|
+
}
|
|
14
|
+
export function reportFileStem(fileName) {
|
|
15
|
+
return fileName.endsWith(REPORT_FILE_SUFFIX)
|
|
16
|
+
? fileName.slice(0, -REPORT_FILE_SUFFIX.length)
|
|
17
|
+
: null;
|
|
18
|
+
}
|
|
19
|
+
export function isReportFileName(fileName) {
|
|
20
|
+
return reportFileStem(fileName) !== null;
|
|
21
|
+
}
|
|
22
|
+
export function graphFileName(stem) {
|
|
23
|
+
return `${safeArtifactFileStem(stem)}${GRAPH_FILE_SUFFIX}`;
|
|
24
|
+
}
|
|
25
|
+
export function cardFileName(stem) {
|
|
26
|
+
return `${safeArtifactFileStem(stem)}${CARD_FILE_SUFFIX}`;
|
|
27
|
+
}
|
|
28
|
+
export function runTimestamp(date = new Date()) {
|
|
29
|
+
const pad = (n) => String(n).padStart(2, '0');
|
|
30
|
+
return `${date.getFullYear()}${pad(date.getMonth() + 1)}${pad(date.getDate())}T${pad(date.getHours())}${pad(date.getMinutes())}${pad(date.getSeconds())}`;
|
|
31
|
+
}
|
|
32
|
+
export function randomRunToken() {
|
|
33
|
+
return Math.random().toString(36).slice(2, 6);
|
|
34
|
+
}
|
|
35
|
+
export function runFileSuffix(counter) {
|
|
36
|
+
const middle = counter === undefined ? '' : `-${counter}`;
|
|
37
|
+
return `${runTimestamp()}${middle}-${randomRunToken()}`;
|
|
38
|
+
}
|
|
39
|
+
export function stripDomainPrefix(id, domain) {
|
|
40
|
+
const safeId = safeArtifactFileStem(id);
|
|
41
|
+
const prefix = `${domain}-`;
|
|
42
|
+
return safeId.startsWith(prefix) ? safeId.slice(prefix.length) : safeId;
|
|
43
|
+
}
|
|
44
|
+
export function doctorReportFileStem(skillName, reportId) {
|
|
45
|
+
return `${safeArtifactFileStem(skillName)}-${stripDomainPrefix(reportId, 'doctor')}`;
|
|
46
|
+
}
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import type { Artifact, EvaluationErrorCategory, EvaluationJob, EvaluationRequest, EvaluationRun, JudgeConfig } from '../types/index.js';
|
|
2
|
-
export declare function buildEvaluationRequest({ samplesPath, skillDir, artifacts, model, executor, noJudge, concurrency, timeoutMs, noCache, dryRun, project, owner, tags, repeat, batch, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget, effort, }: {
|
|
2
|
+
export declare function buildEvaluationRequest({ samplesPath, skillDir, artifacts, model, executor, noJudge, concurrency, timeoutMs, noCache, dryRun, project, owner, tags, repeat, holdoutRatio, batch, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget, effort, }: {
|
|
3
3
|
samplesPath: string;
|
|
4
4
|
skillDir: string;
|
|
5
5
|
artifacts: Artifact[];
|
|
@@ -14,6 +14,7 @@ export declare function buildEvaluationRequest({ samplesPath, skillDir, artifact
|
|
|
14
14
|
owner?: string;
|
|
15
15
|
tags?: string[];
|
|
16
16
|
repeat?: number;
|
|
17
|
+
holdoutRatio?: number;
|
|
17
18
|
batch?: boolean;
|
|
18
19
|
judgeRepeat?: number;
|
|
19
20
|
judgeModels: JudgeConfig[];
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
function nowIso() {
|
|
2
2
|
return new Date().toISOString();
|
|
3
3
|
}
|
|
4
|
-
export function buildEvaluationRequest({ samplesPath, skillDir, artifacts, model, executor, noJudge, concurrency, timeoutMs, noCache, dryRun, project, owner, tags, repeat, batch, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget, effort, }) {
|
|
4
|
+
export function buildEvaluationRequest({ samplesPath, skillDir, artifacts, model, executor, noJudge, concurrency, timeoutMs, noCache, dryRun, project, owner, tags, repeat, holdoutRatio, batch, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget, effort, }) {
|
|
5
5
|
return {
|
|
6
6
|
samplesPath,
|
|
7
7
|
skillDir,
|
|
@@ -17,6 +17,7 @@ export function buildEvaluationRequest({ samplesPath, skillDir, artifacts, model
|
|
|
17
17
|
owner,
|
|
18
18
|
tags,
|
|
19
19
|
repeat,
|
|
20
|
+
holdoutRatio,
|
|
20
21
|
batch,
|
|
21
22
|
judgeRepeat,
|
|
22
23
|
judgeModels,
|
|
@@ -5,6 +5,7 @@ import { createHash } from 'node:crypto';
|
|
|
5
5
|
import { fileURLToPath } from 'node:url';
|
|
6
6
|
import { DEFAULT_REPORTS_DIR } from './default-dirs.js';
|
|
7
7
|
import { indexReportWrite } from './artifact-index.js';
|
|
8
|
+
import { reportFilePath } from './artifact-file-names.js';
|
|
8
9
|
import { buildVariantSummary } from './schema.js';
|
|
9
10
|
import { buildVariantConfig, resolveExecutionStrategy } from './execution-strategy.js';
|
|
10
11
|
import { getJudgePromptHash } from '../grading/judge.js';
|
|
@@ -61,10 +62,14 @@ export function getCliVersion() {
|
|
|
61
62
|
return PKG.version;
|
|
62
63
|
}
|
|
63
64
|
export function getGitInfo() {
|
|
65
|
+
// stdio 静默 stderr:在非 git 目录(如 omk init 出来的 demo)里 rev-parse 会打印
|
|
66
|
+
// `fatal: not a git repository` 到终端。catch 已把失败兜成 null(报告省略 git 信息),
|
|
67
|
+
// 这条 fatal 对用户是纯噪声,吞掉它。与 skill-loader 的 GIT_PROBE_STDIO 同口径。
|
|
68
|
+
const gitProbeStdio = ['ignore', 'pipe', 'ignore'];
|
|
64
69
|
try {
|
|
65
|
-
const commit = execFileSync('git', ['rev-parse', 'HEAD'], { encoding: 'utf-8' }).trim();
|
|
66
|
-
const branch = execFileSync('git', ['rev-parse', '--abbrev-ref', 'HEAD'], { encoding: 'utf-8' }).trim();
|
|
67
|
-
const dirty = execFileSync('git', ['status', '--porcelain'], { encoding: 'utf-8' }).trim().length > 0;
|
|
70
|
+
const commit = execFileSync('git', ['rev-parse', 'HEAD'], { encoding: 'utf-8', stdio: gitProbeStdio }).trim();
|
|
71
|
+
const branch = execFileSync('git', ['rev-parse', '--abbrev-ref', 'HEAD'], { encoding: 'utf-8', stdio: gitProbeStdio }).trim();
|
|
72
|
+
const dirty = execFileSync('git', ['status', '--porcelain'], { encoding: 'utf-8', stdio: gitProbeStdio }).trim().length > 0;
|
|
68
73
|
return { commit, commitShort: commit.slice(0, 7), branch, dirty };
|
|
69
74
|
}
|
|
70
75
|
catch {
|
|
@@ -276,7 +281,7 @@ export function persistReport(report, outputDir) {
|
|
|
276
281
|
return null;
|
|
277
282
|
if (!existsSync(outputDir))
|
|
278
283
|
mkdirSync(outputDir, { recursive: true });
|
|
279
|
-
const filePath =
|
|
284
|
+
const filePath = reportFilePath(outputDir, report.id);
|
|
280
285
|
writeFileSync(filePath, JSON.stringify(report, null, 2));
|
|
281
286
|
// 产物发现索引:报告落项目本地后,best-effort 追加全局轻卡片,让 omk studio 跨项目聚合成机器级总览。
|
|
282
287
|
// 永不抛、永不阻断报告落盘(正文是 source of truth)。
|
|
@@ -0,0 +1,66 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Holdout split + train/holdout composite breakdown.
|
|
3
|
+
*
|
|
4
|
+
* Deterministic (no RNG) train/holdout partitioning of a sample set, plus the
|
|
5
|
+
* subset-composite recompute that lets `omk eval --holdout-ratio` and `omk evolve`
|
|
6
|
+
* score a variant on a withheld slice using the *same* aggregation as the headline
|
|
7
|
+
* composite (`buildVariantSummary`). Lives in eval-core so both the eval pipeline
|
|
8
|
+
* and the authoring/evolve loop depend *down* into it (authoring → eval-core is the
|
|
9
|
+
* established direction).
|
|
10
|
+
*
|
|
11
|
+
* A large train − holdout composite gap is the generalization / sample-set-overfitting
|
|
12
|
+
* signal the verdict's overfitting gate reads (`src/eval-core/verdict.ts`).
|
|
13
|
+
*/
|
|
14
|
+
import type { Report, HoldoutBreakdown } from '../types/index.js';
|
|
15
|
+
/** A train / holdout partition of a sample set. */
|
|
16
|
+
export interface HoldoutSplit {
|
|
17
|
+
trainIds: Set<string>;
|
|
18
|
+
holdoutIds: Set<string>;
|
|
19
|
+
}
|
|
20
|
+
/** Below this many samples on any side, a split is too small to be meaningful —
|
|
21
|
+
* callers fall back to full-set scoring and mark the breakdown `disabled`. */
|
|
22
|
+
export declare const MIN_HOLDOUT_SUBSET = 3;
|
|
23
|
+
/** Pick `count` ids at an even stride across `ids` (deterministic, no RNG) so the
|
|
24
|
+
* picked subset is representative of the ordering and stable across rounds/runs. */
|
|
25
|
+
export declare function pickByStride(ids: string[], count: number): Set<string>;
|
|
26
|
+
/**
|
|
27
|
+
* Deterministically split sample ids into train / holdout by `ratio` (fraction
|
|
28
|
+
* held out). Holdout members are picked at an even stride so the partition is
|
|
29
|
+
* representative of the ordering, and the split is stable across rounds and runs
|
|
30
|
+
* (no RNG). Returns null when ratio ≤ 0 or either side would drop below
|
|
31
|
+
* MIN_HOLDOUT_SUBSET — the caller then scores on the full set.
|
|
32
|
+
*/
|
|
33
|
+
export declare function splitHoldout(sampleIds: string[], ratio: number): HoldoutSplit | null;
|
|
34
|
+
/**
|
|
35
|
+
* Mean composite over the subset of a report's results whose sample_id is in
|
|
36
|
+
* `ids`, using the same aggregation as the full-run summary
|
|
37
|
+
* (`buildVariantSummary`) so train / holdout scores stay comparable to the
|
|
38
|
+
* headline composite. Returns 0 when the subset has no scorable entries.
|
|
39
|
+
*/
|
|
40
|
+
export declare function subsetCompositeScore(report: Report, variantKey: string, ids: Set<string>): number;
|
|
41
|
+
/**
|
|
42
|
+
* How many subset results actually produced a usable composite (> 0) for a variant.
|
|
43
|
+
* `buildVariantSummary` averages only `compositeScore > 0` entries (schema.ts), so
|
|
44
|
+
* the mean can rest on far fewer samples than the authored split size when runs
|
|
45
|
+
* flake / partial-error / budget-abort. The overfitting gate must trust THIS count,
|
|
46
|
+
* not the authored `trainCount` / `holdoutCount`, or a 1-of-3 holdout gets dressed
|
|
47
|
+
* up as a 3-sample-backed conclusion.
|
|
48
|
+
*/
|
|
49
|
+
export declare function subsetScorableCount(report: Report, variantKey: string, ids: Set<string>): number;
|
|
50
|
+
/**
|
|
51
|
+
* Train vs holdout composite breakdown per variant for `omk eval --holdout-ratio`.
|
|
52
|
+
* Post-hoc — never perturbs the headline aggregation or bootstrap CI.
|
|
53
|
+
*
|
|
54
|
+
* The split is taken over `sampleIdOrder` — the **stable authored sample order**
|
|
55
|
+
* (the loaded `samples` file order), NOT `report.results`, whose insertion order
|
|
56
|
+
* is the concurrent-completion order and drifts run-to-run. Binding the stride pick
|
|
57
|
+
* to the authored order is what makes the holdout (and the verdict overfitting gate
|
|
58
|
+
* it feeds) deterministic and reproducible. Subset scores are then read from
|
|
59
|
+
* `report.results` by id-set membership, which is order-independent.
|
|
60
|
+
*
|
|
61
|
+
* When the split is too small on either side (< MIN_HOLDOUT_SUBSET) it returns
|
|
62
|
+
* `{ disabled: true }` with an empty `perVariant`, so the verdict overfitting gate
|
|
63
|
+
* stays inert. The testSetHash watermark (gap-spec §7.1) is attached by the caller,
|
|
64
|
+
* shared with gapReports.
|
|
65
|
+
*/
|
|
66
|
+
export declare function computeHoldoutBreakdown(report: Report, variantNames: string[], ratio: number, sampleIdOrder: string[]): HoldoutBreakdown;
|