oh-my-knowledge 0.40.0 → 0.42.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (71) hide show
  1. package/README.md +8 -4
  2. package/README.zh.md +8 -4
  3. package/dist/analysis/report-diagnostics.d.ts +8 -1
  4. package/dist/analysis/report-diagnostics.js +109 -1
  5. package/dist/assets/agent-skills/omk/references/commands.md +2 -2
  6. package/dist/authoring/evolver.d.ts +3 -14
  7. package/dist/authoring/evolver.js +1 -52
  8. package/dist/authoring/generator.d.ts +24 -0
  9. package/dist/authoring/generator.js +36 -8
  10. package/dist/cli/commands/eval/index.d.ts +1 -1
  11. package/dist/cli/commands/eval/index.js +50 -15
  12. package/dist/cli/commands/init.js +10 -7
  13. package/dist/cli/lib/cmd-flags.d.ts +0 -1
  14. package/dist/cli/lib/i18n-dict/init.d.ts +1 -1
  15. package/dist/cli/lib/i18n-dict/init.js +14 -11
  16. package/dist/cli/lib/i18n-dict/run.d.ts +1 -1
  17. package/dist/cli/lib/i18n-dict/run.js +6 -2
  18. package/dist/cli/lib/parse-run-config.d.ts +3 -1
  19. package/dist/cli/lib/parse-run-config.js +0 -2
  20. package/dist/eval-core/evaluation-job.d.ts +2 -2
  21. package/dist/eval-core/evaluation-job.js +2 -2
  22. package/dist/eval-core/evaluation-reporting.d.ts +0 -1
  23. package/dist/eval-core/evaluation-reporting.js +9 -41
  24. package/dist/eval-core/execution-strategy.js +3 -2
  25. package/dist/eval-core/holdout.d.ts +66 -0
  26. package/dist/eval-core/holdout.js +118 -0
  27. package/dist/eval-core/judge-independence.d.ts +28 -0
  28. package/dist/eval-core/judge-independence.js +29 -0
  29. package/dist/eval-core/verdict.d.ts +53 -2
  30. package/dist/eval-core/verdict.js +216 -16
  31. package/dist/eval-workflows/batch-evaluation-workflow.d.ts +1 -1
  32. package/dist/eval-workflows/batch-evaluation-workflow.js +1 -2
  33. package/dist/eval-workflows/evaluation-pipeline/report-finalize.d.ts +1 -5
  34. package/dist/eval-workflows/evaluation-pipeline/report-finalize.js +19 -8
  35. package/dist/eval-workflows/evaluation-pipeline/run-state.d.ts +2 -2
  36. package/dist/eval-workflows/evaluation-pipeline/run-state.js +2 -2
  37. package/dist/eval-workflows/evaluation-pipeline.d.ts +3 -2
  38. package/dist/eval-workflows/evaluation-pipeline.js +3 -4
  39. package/dist/eval-workflows/run-evaluation.d.ts +6 -4
  40. package/dist/eval-workflows/run-evaluation.js +8 -6
  41. package/dist/executors/claude-cli.js +5 -6
  42. package/dist/executors/claude-sdk.d.ts +5 -2
  43. package/dist/executors/claude-sdk.js +13 -8
  44. package/dist/executors/codex-cli.js +3 -4
  45. package/dist/executors/shared.d.ts +2 -0
  46. package/dist/executors/shared.js +15 -0
  47. package/dist/grading/assertions.js +6 -122
  48. package/dist/grading/gold-cli.js +1 -1
  49. package/dist/grading/human-gold.d.ts +5 -3
  50. package/dist/grading/human-gold.js +5 -3
  51. package/dist/grading/index.d.ts +4 -4
  52. package/dist/grading/judge.d.ts +6 -14
  53. package/dist/grading/judge.js +5 -88
  54. package/dist/inputs/eval-config.js +12 -2
  55. package/dist/managed/evidence.js +1 -2
  56. package/dist/managed/version-scores.js +1 -1
  57. package/dist/renderer/html-renderer.js +0 -9
  58. package/dist/renderer/layout.js +4 -4
  59. package/dist/renderer/summary.js +59 -4
  60. package/dist/shared/llm-prompts/debias-instructions.d.ts +4 -0
  61. package/dist/shared/llm-prompts/debias-instructions.js +44 -0
  62. package/dist/shared/llm-prompts/judge-prompts.d.ts +30 -0
  63. package/dist/shared/llm-prompts/judge-prompts.js +205 -0
  64. package/dist/shared/llm-prompts/registry.d.ts +27 -0
  65. package/dist/shared/llm-prompts/registry.js +69 -0
  66. package/dist/types/eval.d.ts +15 -8
  67. package/dist/types/judge.d.ts +1 -1
  68. package/dist/types/report.d.ts +50 -3
  69. package/package.json +1 -1
  70. package/dist/grading/debias-validate.d.ts +0 -83
  71. package/dist/grading/debias-validate.js +0 -176
@@ -0,0 +1,205 @@
1
+ import { createHash } from 'node:crypto';
2
+ import { LENGTH_DEBIAS_INSTRUCTION, PRESENTATION_NEUTRALITY_INSTRUCTION, RAG_LENGTH_DEBIAS, RAG_PRESENTATION_NEUTRALITY, } from './debias-instructions.js';
3
+ // 评分类 prompt 单一来源 —— 直接决定分数的 LLM 评委 prompt(rubric 主评委 + RAG + 语义相似度)。
4
+ // 这些是测量学不变量:文本字节决定可比性,改动须配合 prompt-registry 的冻结 hash bump
5
+ // (BREAKING-COMPARABILITY)。执行器调用 / JSON 解析逻辑留在 grading/judge.ts 与 grading/assertions.ts。
6
+ // ===========================================================================
7
+ // Rubric 主评委 prompt
8
+ // ===========================================================================
9
+ /**
10
+ * Judge prompt template version.
11
+ *
12
+ * The version string encodes WHICH debias / context features the judge sees, so reports
13
+ * tagged with the same hash are score-comparable; a mismatched hash means "we changed how
14
+ * we ask the judge to think" and the reports should not be compared blind. Bump it (and the
15
+ * frozen hashes in `test/shared/prompt-registry-freeze.test.ts`) whenever the template's bytes
16
+ * change — that change is BREAKING-COMPARABILITY.
17
+ *
18
+ * Naming: a single main version (`v5`) + a `-feature` suffix per debias/context capability.
19
+ * The only asymmetry between the two strings is the `-len` suffix, gated by the
20
+ * `--no-debias-length` toggle; every other feature is always-on and appears in both.
21
+ */
22
+ // 版本演进(内嵌"评委看到什么 / 被要求忽略什么"的语义):
23
+ // v2-cot : 仅看 output + rubric + 工具名分布(早期 buildTraceSummary)
24
+ // v3-cot-length : 加 length-debias 指令
25
+ // v3-cot-toolargs(off) / v4-cot-len-args(on)
26
+ // : 加 tool input 预览(让评委识别 wrapper-style skill 在 Bash 命令里的
27
+ // 真实语义调用,如 `Bash: mcporter --tool X`,否则误判"只调了 Bash")
28
+ // v5-cot-toolargs-fmt(off) / v5-cot-toolargs-fmt-len(on)
29
+ // : 加排版 / 语气中性化指令(始终开启,不给开关)。研究表明 LLM 评委除了
30
+ // 偏向更长的回答,还隐性偏向排版精致(标题 / 列表 / 加粗)与语气自信 / 自我
31
+ // 表扬(谄媚 / 权威偏置)的回答;显式指令要求评委只对照评分标准核内容。
32
+ // 同时借此把命名统一成单一主序号(v5)+ feature 后缀,`-len` 仍是开关那条。
33
+ const JUDGE_PROMPT_VERSION_DEBIAS_OFF = 'v5-cot-toolargs-fmt';
34
+ const JUDGE_PROMPT_VERSION_DEBIAS_ON = 'v5-cot-toolargs-fmt-len';
35
+ export const JUDGE_SYSTEM_PROMPT = '你是一个严格的 AI 输出质量评审员。先逐条对照评分标准做推理,再给最终分数。只返回 JSON,不要其他内容。';
36
+ export function buildJudgePrompt(prompt, rubric, output, traceSummary, lengthDebias = true) {
37
+ const version = lengthDebias ? JUDGE_PROMPT_VERSION_DEBIAS_ON : JUDGE_PROMPT_VERSION_DEBIAS_OFF;
38
+ const traceSection = traceSummary
39
+ ? ['', '## Agent 执行过程', traceSummary, '', '请同时考虑执行过程的合理性(工具选择、步骤效率、错误恢复)。']
40
+ : [];
41
+ // 排版 / 语气中性化始终开启(不受 --no-debias-length 影响);length-debias 仍受开关控。
42
+ const neutralitySection = ['', PRESENTATION_NEUTRALITY_INSTRUCTION];
43
+ const debiasSection = lengthDebias ? ['', LENGTH_DEBIAS_INSTRUCTION] : [];
44
+ return [
45
+ `请对以下 AI 输出进行质量评分(template ${version})。`,
46
+ '',
47
+ '## 原始任务',
48
+ prompt,
49
+ '',
50
+ '## 评分标准',
51
+ rubric,
52
+ '',
53
+ '## AI 输出',
54
+ output,
55
+ ...traceSection,
56
+ ...neutralitySection,
57
+ ...debiasSection,
58
+ '',
59
+ '## 评分流程',
60
+ '1. 逐条对照评分标准,先做推理(reasoning):列出 AI 输出哪些点对应哪条标准,哪些缺失,哪些有歧义。',
61
+ '2. 基于推理给出最终分数(1-5 的整数)和简短理由。',
62
+ '',
63
+ '请返回 JSON(不要包含 markdown 代码块标记):',
64
+ '{"reasoning": "<对照标准的逐条推理>", "score": <1-5的整数>, "reason": "<最终结论的简短理由>"}',
65
+ '',
66
+ '评分标准:1=完全不达标, 2=部分涉及, 3=基本达标, 4=较好, 5=优秀',
67
+ ].join('\n');
68
+ }
69
+ /**
70
+ * Stable hash of the judge prompt template. Saved into ReportMeta.judgePromptHash so
71
+ * downstream readers can detect "the judge prompt changed between these two reports".
72
+ *
73
+ * `lengthDebias` defaults to true. Pass false (via `--no-debias-length`) to drop the
74
+ * length-debias instruction; that produces the debias-off prompt variant, whose hash
75
+ * differs from the default so the two are never compared blind.
76
+ */
77
+ export function getJudgePromptHash(lengthDebias = true) {
78
+ const version = lengthDebias ? JUDGE_PROMPT_VERSION_DEBIAS_ON : JUDGE_PROMPT_VERSION_DEBIAS_OFF;
79
+ // Hash the template-shaping function source + the version tag together. We hash a
80
+ // deterministic stringified form of the template (with placeholder inputs) so any
81
+ // structural edit shows up.
82
+ const sample = buildJudgePrompt('<P>', '<R>', '<O>', '<T>', lengthDebias);
83
+ return createHash('sha256').update(version + '\n' + sample).digest('hex').slice(0, 12);
84
+ }
85
+ // ===========================================================================
86
+ // 语义相似度评委 prompt(semantic_similarity 断言)
87
+ // ===========================================================================
88
+ export const SEMANTIC_SIMILARITY_SYSTEM = '你是语义相似度评审员。只返回 JSON,不要其他内容。';
89
+ export function buildSemanticSimilarityPrompt(reference, output) {
90
+ return [
91
+ '请判断以下两段文本的语义相似度。',
92
+ '',
93
+ '## 参考文本',
94
+ reference,
95
+ '',
96
+ '## 待评估文本',
97
+ output,
98
+ '',
99
+ '请返回 JSON(不要包含 markdown 代码块标记):',
100
+ '{"score": <1-5的整数>, "reason": "<简短理由>"}',
101
+ '',
102
+ '评分:1=完全无关, 2=略有关联, 3=部分相似, 4=大致相同, 5=高度一致',
103
+ ].join('\n');
104
+ }
105
+ /** RAG 评委 system + 用户 prompt。缺字段的早退 / 解析逻辑留在 assertions.ts。 */
106
+ export function buildRagJudgePrompt(type, fields) {
107
+ if (type === 'faithfulness') {
108
+ return {
109
+ system: '你是 RAG 评审员,专注判断输出是否被参考 context 支持。只返回 JSON。',
110
+ prompt: [
111
+ '请判断"待评估输出"中的事实性陈述是否被"参考 context"支持。',
112
+ '',
113
+ '## 参考 context',
114
+ fields.context ?? '',
115
+ '',
116
+ '## 待评估输出',
117
+ fields.output,
118
+ '',
119
+ RAG_LENGTH_DEBIAS,
120
+ RAG_PRESENTATION_NEUTRALITY,
121
+ '',
122
+ '## 评分流程',
123
+ '1. 列出待评估输出中所有事实性陈述',
124
+ '2. 逐条判断是否能在 context 中找到支持',
125
+ '3. 给出 1-5 分:',
126
+ ' 5 = 全部陈述都有 context 支持,无编造',
127
+ ' 4 = 多数有支持,有 1-2 处不重要的编造',
128
+ ' 3 = 一半有支持',
129
+ ' 2 = 多数无支持',
130
+ ' 1 = 完全编造或与 context 矛盾',
131
+ '',
132
+ '请返回 JSON(不要 markdown 代码块):',
133
+ '{"score": <1-5的整数>, "reason": "<简短理由>"}',
134
+ ].join('\n'),
135
+ };
136
+ }
137
+ if (type === 'answer_relevancy') {
138
+ return {
139
+ system: '你是答题切题度评审员。只返回 JSON。',
140
+ prompt: [
141
+ '请判断"AI 输出"是否直接、切题地回答了"用户问题"。',
142
+ '',
143
+ '## 用户问题',
144
+ fields.question ?? '',
145
+ '',
146
+ '## AI 输出',
147
+ fields.output,
148
+ '',
149
+ RAG_LENGTH_DEBIAS,
150
+ RAG_PRESENTATION_NEUTRALITY,
151
+ '',
152
+ '## 评分',
153
+ '5 = 完整切题回答,无冗余无遗漏',
154
+ '4 = 切题但有少量冗余或小遗漏',
155
+ '3 = 部分切题,部分跑题或避而不答',
156
+ '2 = 大部分跑题',
157
+ '1 = 完全跑题或拒答',
158
+ '',
159
+ '请返回 JSON(不要 markdown 代码块):',
160
+ '{"score": <1-5的整数>, "reason": "<简短理由>"}',
161
+ ].join('\n'),
162
+ };
163
+ }
164
+ // context_recall
165
+ return {
166
+ system: '你是 context 覆盖率评审员。只返回 JSON。',
167
+ prompt: [
168
+ '请判断"参考 gold"中的关键事实在"AI 输出"中被覆盖的程度。',
169
+ '',
170
+ '## 参考 gold',
171
+ fields.reference ?? '',
172
+ '',
173
+ '## AI 输出',
174
+ fields.output,
175
+ '',
176
+ RAG_LENGTH_DEBIAS,
177
+ RAG_PRESENTATION_NEUTRALITY,
178
+ '',
179
+ '## 评分流程',
180
+ '1. 列出参考中的关键事实(忽略修饰性内容)',
181
+ '2. 检查每条是否在输出中被提及/使用',
182
+ '3. 给出 1-5 分:',
183
+ ' 5 = 全部关键事实被覆盖',
184
+ ' 4 = 大部分覆盖,缺 1-2 条次要事实',
185
+ ' 3 = 一半覆盖',
186
+ ' 2 = 仅覆盖少量',
187
+ ' 1 = 完全未覆盖',
188
+ '',
189
+ '请返回 JSON(不要 markdown 代码块):',
190
+ '{"score": <1-5的整数>, "reason": "<简短理由>"}',
191
+ ].join('\n'),
192
+ };
193
+ }
194
+ /** 评分类断言 prompt 的 drift-detection hash(占位符输入,12 位,仅供 prompt-registry 冻结)。 */
195
+ function hashPromptSample(label, system, prompt) {
196
+ return createHash('sha256').update(`${label}\n${system}\n${prompt}`).digest('hex').slice(0, 12);
197
+ }
198
+ export function getSemanticPromptHash() {
199
+ return hashPromptSample('semantic_similarity', SEMANTIC_SIMILARITY_SYSTEM, buildSemanticSimilarityPrompt('<R>', '<O>'));
200
+ }
201
+ export function getRagJudgePromptHash(type) {
202
+ const fields = { output: '<O>', context: '<C>', question: '<Q>', reference: '<R>' };
203
+ const { system, prompt } = buildRagJudgePrompt(type, fields);
204
+ return hashPromptSample(`rag:${type}`, system, prompt);
205
+ }
@@ -0,0 +1,27 @@
1
+ /**
2
+ * omk 所有 LLM prompt 的单一编目 —— 发现入口 + 测量学冻结入口。
3
+ *
4
+ * 解决「prompt 散在多个目录、难管理」与「只有 2 个 prompt 被冻结」两个问题:
5
+ * - 每条登记 prompt 的位置(module)+ 用途,一处可查全。
6
+ * - measurementInvariant 为 true 的(直接决定分数的评委 prompt)带 getHash,
7
+ * 由 test/shared/prompt-registry-freeze.test.ts 统一冻结,文本漂移即 CI 红。
8
+ *
9
+ * 注意:hash 形状按各自历史口径,不强行统一 ——
10
+ * - rubric 评委:12 位(getJudgePromptHash,持久化进 report.meta.judgePromptHash 的同口径);
11
+ * - RAG / semantic:12 位(getRagJudgePromptHash / getSemanticPromptHash,仅供冻结 drift 检测);
12
+ * - observe 复盘:64 位(readPromptDocument 读 .md 全文 sha256)。
13
+ * 用限定名 `promptId`(非裸 `kind`,见 terminology-spec §5.4)。
14
+ */
15
+ export interface PromptRegistryEntry {
16
+ /** 稳定标识(冻结表的 key)。 */
17
+ promptId: string;
18
+ /** 人类可读用途。 */
19
+ purpose: string;
20
+ /** 定义位置(发现用,相对仓库根的源文件路径)。 */
21
+ module: string;
22
+ /** 是否直接决定分数 → 须冻结防漂移。 */
23
+ measurementInvariant: boolean;
24
+ /** 仅 measurementInvariant 条目提供:当前 prompt 的 hash。 */
25
+ getHash?: () => string;
26
+ }
27
+ export declare const PROMPT_REGISTRY: PromptRegistryEntry[];
@@ -0,0 +1,69 @@
1
+ import { getJudgePromptHash, getSemanticPromptHash, getRagJudgePromptHash } from './judge-prompts.js';
2
+ import { readPromptDocument } from './index.js';
3
+ import { PROMPTS_DIR, SOFT_STANDARD_PROMPT_ID, SOFT_STANDARD_PROMPT_VERSION } from '../../observability/soft-standards/constants.js';
4
+ export const PROMPT_REGISTRY = [
5
+ // —— 测量学不变量:直接决定分数的评委 prompt,统一冻结 ——
6
+ {
7
+ promptId: 'rubric-judge-debias-on',
8
+ purpose: 'rubric 主评委(length-debias 开,默认)',
9
+ module: 'src/shared/llm-prompts/judge-prompts.ts',
10
+ measurementInvariant: true,
11
+ getHash: () => getJudgePromptHash(true),
12
+ },
13
+ {
14
+ promptId: 'rubric-judge-debias-off',
15
+ purpose: 'rubric 主评委(--no-debias-length,length-debias 关)',
16
+ module: 'src/shared/llm-prompts/judge-prompts.ts',
17
+ measurementInvariant: true,
18
+ getHash: () => getJudgePromptHash(false),
19
+ },
20
+ {
21
+ promptId: 'semantic-similarity',
22
+ purpose: 'semantic_similarity 断言评委',
23
+ module: 'src/shared/llm-prompts/judge-prompts.ts',
24
+ measurementInvariant: true,
25
+ getHash: () => getSemanticPromptHash(),
26
+ },
27
+ {
28
+ promptId: 'rag-faithfulness',
29
+ purpose: 'RAG faithfulness(输出是否被 context 支持)',
30
+ module: 'src/shared/llm-prompts/judge-prompts.ts',
31
+ measurementInvariant: true,
32
+ getHash: () => getRagJudgePromptHash('faithfulness'),
33
+ },
34
+ {
35
+ promptId: 'rag-answer-relevancy',
36
+ purpose: 'RAG answer_relevancy(是否切题)',
37
+ module: 'src/shared/llm-prompts/judge-prompts.ts',
38
+ measurementInvariant: true,
39
+ getHash: () => getRagJudgePromptHash('answer_relevancy'),
40
+ },
41
+ {
42
+ promptId: 'rag-context-recall',
43
+ purpose: 'RAG context_recall(关键事实覆盖率)',
44
+ module: 'src/shared/llm-prompts/judge-prompts.ts',
45
+ measurementInvariant: true,
46
+ getHash: () => getRagJudgePromptHash('context_recall'),
47
+ },
48
+ {
49
+ promptId: 'observe-llm-enhanced-review',
50
+ purpose: 'observe 运行期软标准抽取 / LLM 增强复盘',
51
+ module: 'src/observability/prompts/llm-enhanced-review.prompt.md',
52
+ measurementInvariant: true,
53
+ getHash: () => readPromptDocument({
54
+ dir: PROMPTS_DIR,
55
+ fileName: `${SOFT_STANDARD_PROMPT_ID}.prompt.md`,
56
+ id: SOFT_STANDARD_PROMPT_ID,
57
+ version: SOFT_STANDARD_PROMPT_VERSION,
58
+ }).hash,
59
+ },
60
+ // —— 非评分类:不决定 composite / verdict / assertion 分数,不冻结,仅登记供发现 ——
61
+ { promptId: 'failure-diagnostic', purpose: '失败诊断(root cause / 修复建议)', module: 'src/grading/diagnostic.ts', measurementInvariant: false },
62
+ { promptId: 'failure-clusterer', purpose: '失败案例聚类', module: 'src/analysis/failure-clusterer.ts', measurementInvariant: false },
63
+ { promptId: 'hedging-classifier', purpose: 'hedging 判定(喂 gap-signal,非评分)', module: 'src/analysis/hedging-classifier.ts', measurementInvariant: false },
64
+ { promptId: 'sample-generator', purpose: '用例生成(skill / trace → samples)', module: 'src/authoring/generator.ts', measurementInvariant: false },
65
+ { promptId: 'sample-fixer', purpose: '坏用例修复', module: 'src/authoring/sample-fixer.ts', measurementInvariant: false },
66
+ { promptId: 'skill-improve', purpose: 'skill 迭代改进(evolve)', module: 'src/authoring/evolver.ts', measurementInvariant: false },
67
+ { promptId: 'doctor-fixer', purpose: 'doctor 健康项修复向导', module: 'src/doctor/fixer.ts', measurementInvariant: false },
68
+ { promptId: 'skill-health', purpose: 'skill 健康检查打分', module: 'src/shared/llm-prompts/skill-health.ts', measurementInvariant: false },
69
+ ];
@@ -211,7 +211,6 @@ export interface EvalConfig {
211
211
  timeoutMs?: number;
212
212
  noCache?: boolean;
213
213
  noJudge?: boolean;
214
- blind?: boolean;
215
214
  mcpConfig?: string;
216
215
  variants: EvalConfigVariant[];
217
216
  /** hard budget caps. When any limit is hit during a run, remaining
@@ -221,6 +220,9 @@ export interface EvalConfig {
221
220
  budget?: EvalBudget;
222
221
  /** --repeat N. Multi-run variance analysis. */
223
222
  repeat?: number;
223
+ /** --holdout-ratio R (0 < R < 1). Hold out a deterministic sample slice and
224
+ * report train vs holdout composite as a generalization / overfitting signal. */
225
+ holdoutRatio?: number;
224
226
  /** --judge-repeat N. Each (sample × dimension) judged N times for self-consistency stddev. */
225
227
  judgeRepeat?: number;
226
228
  /** --bootstrap. Distribution-free CI per variant + pairwise diff. */
@@ -229,7 +231,7 @@ export interface EvalConfig {
229
231
  bootstrapSamples?: number;
230
232
  /** --gold-dir. After-run automatic comparison against a human-anchor dataset. */
231
233
  goldDir?: string;
232
- /** --no-debias-length flips this to false. Default true (judge prompt v3-cot-length). */
234
+ /** --no-debias-length flips this to false. Default true (length-debias instruction on). */
233
235
  lengthDebias?: boolean;
234
236
  /** --no-strict-baseline flips this to false. Default true (baseline-kind allowedSkills=[]). */
235
237
  strictBaseline?: boolean;
@@ -256,9 +258,12 @@ export interface EvaluationRequest {
256
258
  timeoutMs?: number;
257
259
  noCache: boolean;
258
260
  dryRun: boolean;
259
- blind: boolean;
260
261
  /** --repeat N; 1 表示单次跑,> 1 走 runMultiple 做 variance 分析 */
261
262
  repeat?: number;
263
+ /** --holdout-ratio R; 0 / 缺省表示不切分(默认)。> 0 时 report-finalize 在结果上
264
+ * post-hoc 切出 train / holdout 子集算综合分(`report.analysis.holdout`),供 verdict
265
+ * 的过拟合门控读取。see src/eval-core/holdout.ts */
266
+ holdoutRatio?: number;
262
267
  /** --batch; default absent/false. True means skill-batch mode. */
263
268
  batch?: boolean;
264
269
  /** --judge-repeat N; 每条 sample × dimension 用 LLM judge 跑 N 次, 输出 stddev. 默认 1 (单次). */
@@ -266,8 +271,9 @@ export interface EvaluationRequest {
266
271
  /** Unified judge config — always non-empty.
267
272
  * - length === 1: single judge (degenerate ensemble of size 1).
268
273
  * - length >= 2: multi-judge ensemble. Each (sample × dimension) is scored by every judge,
269
- * inter-judge agreement (Pearson + mean absolute difference) reported as the hard
270
- * rebuttal to "judge same-model bias".
274
+ * inter-judge agreement (Pearson + mean absolute difference) reported as a rebuttal to
275
+ * "judge same-model bias" — but only when judges span vendors; a single-vendor ensemble's
276
+ * high agreement reflects shared bias, not independence (flagged by `single_vendor_ensemble`).
271
277
  * When `noJudge: true` the entry is preserved for audit but no judge call actually runs. */
272
278
  judgeModels: JudgeConfig[];
273
279
  /** --bootstrap; true 时 aggregateReport 加跑 bootstrap mean/diff CI, 写入 VariantSummary.
@@ -275,9 +281,10 @@ export interface EvaluationRequest {
275
281
  bootstrap?: boolean;
276
282
  /** --bootstrap-samples N; bootstrap 重采样次数, 默认 1000. > 10000 时 stderr 警告. */
277
283
  bootstrapSamples?: number;
278
- /** length-debias toggle. Default true (judge prompt v3-cot-length).
279
- * CLI flag --no-debias-length flips to false (legacy v2-cot prompt). The active
280
- * value is reflected in ReportMeta.judgePromptHash and ReportMeta.debiasMode. */
284
+ /** length-debias toggle. Default true — judge prompt carries the length-debias
285
+ * instruction. CLI flag --no-debias-length flips to false (drops that instruction,
286
+ * the debias-off prompt variant). The active value is reflected in
287
+ * ReportMeta.judgePromptHash and ReportMeta.debiasMode. */
281
288
  lengthDebias?: boolean;
282
289
  /** hard budget caps. See EvalBudget. */
283
290
  budget?: EvalBudget;
@@ -18,7 +18,7 @@ export interface JudgeRuntimeEntry {
18
18
  }
19
19
  /** Per-judge ensemble entry: which judge gave what score (mean over judge-repeat if N>1). */
20
20
  export interface EnsembleJudgeResult {
21
- /** "executor:model" identifier — e.g. "claude:opus" or "openai:gpt-4o". */
21
+ /** "executor:model" identifier — e.g. "claude:opus" or "openai-api:gpt-4o". */
22
22
  judge: string;
23
23
  /** Mean score from this judge over judge-repeat calls (or single score if repeat=1). */
24
24
  score: number;
@@ -317,7 +317,7 @@ export interface ReportMeta {
317
317
  humanAgreement?: ReportHumanAgreement;
318
318
  variantConfigs?: VariantConfig[];
319
319
  /** Skill isolation 快照(per-variant)。
320
- * key = variant name;value = allowedSkills(undefined → null,SDK 默认全发现 / [] → 完全隔离 / [...] → 白名单)。
320
+ * key = variant name;value = allowedSkills(undefined → null,SDK 默认全发现 / [] → 完全隔离;非空白名单已移除)。
321
321
  * 跨报告对比 verdict / Δ 时,isolation 状态不一致会被 stderr warn 标"不可比"。
322
322
  * 字段缺失意味着报告产自 之前(默认全发现,construct validity 不保证)。 */
323
323
  skillIsolation?: Record<string, string[] | null>;
@@ -325,8 +325,6 @@ export interface ReportMeta {
325
325
  run?: EvaluationRun;
326
326
  job?: EvaluationJob;
327
327
  gitInfo?: GitInfo | null;
328
- blind?: boolean;
329
- blindMap?: Record<string, string>;
330
328
  layeredStats?: boolean;
331
329
  /** Evolve 合并报告的原始 skill 归属。variants 会被 relabel 为 round-0/round-1,
332
330
  * Studio skill 索引用该字段把报告归回 skill 卡片。 */
@@ -482,6 +480,34 @@ export interface AnalysisResult {
482
480
  * (capability / difficulty / construct / provenance); persisted on report
483
481
  * for studio to surface coverage gaps. See docs/specs/sample-design-spec.md. */
484
482
  sampleQuality?: SampleQualityAggregate;
483
+ /** Opt-in train/holdout generalization breakdown (`omk eval --holdout-ratio`).
484
+ * Absent on default runs; present only when a holdout ratio was requested. */
485
+ holdout?: HoldoutBreakdown;
486
+ }
487
+ /** Train vs holdout composite breakdown for `omk eval --holdout-ratio`.
488
+ * Computed post-hoc from `report.results` by `computeHoldoutBreakdown`
489
+ * (`src/eval-core/holdout.ts`), sharing the same testSetHash watermark as
490
+ * gapReports (gap-spec §7.1). A large train − holdout composite gap is the
491
+ * sample-set-overfitting signal the verdict's overfitting gate reads. */
492
+ export interface HoldoutBreakdown {
493
+ /** Held-out fraction requested via --holdout-ratio. */
494
+ ratio: number;
495
+ /** true when either side fell below the minimum subset size → scored full-set,
496
+ * no usable split. `perVariant` is empty and the verdict gate stays inert. */
497
+ disabled?: boolean;
498
+ /** Per-variant train vs holdout composite (1-5 scale). `*Count` is the authored
499
+ * split size; `*Scorable` is how many of those actually produced a composite (> 0)
500
+ * — they diverge under partial errors, and the overfitting gate trusts `*Scorable`. */
501
+ perVariant: Record<string, {
502
+ trainScore: number;
503
+ holdoutScore: number;
504
+ trainCount: number;
505
+ holdoutCount: number;
506
+ trainScorable: number;
507
+ holdoutScorable: number;
508
+ }>;
509
+ testSetPath?: string | null;
510
+ testSetHash?: string | null;
485
511
  }
486
512
  /** Aggregated sample design coverage stats. Built by
487
513
  * `buildSampleQualityAggregate(samples)` from `Sample.capability` /
@@ -504,6 +530,27 @@ export interface SampleQualityAggregate {
504
530
  sampleCountWithDifficulty: number;
505
531
  sampleCountWithConstruct: number;
506
532
  sampleCountWithProvenance: number;
533
+ /** Relative-balance / skew of the sample set (derived from the distributions
534
+ * above). Flags over-representation — "70% of samples are easy" — without an
535
+ * external denominator. Diagnostic only; never feeds grading / judge / verdict. */
536
+ representativeness?: Representativeness;
537
+ }
538
+ /** Distribution skew over what the sample set declares. Pure relative balance —
539
+ * there is no authored "expected" capability list to measure absolute coverage
540
+ * against (capabilities are free-form strings), so this reports concentration
541
+ * (dominant bucket share, 0-1) and the dominant label per dimension. */
542
+ export interface Representativeness {
543
+ /** Distinct capabilities declared across the set. */
544
+ capabilityCount: number;
545
+ /** Dominant capability's share of all capability tags (0-1); 0 when none declared. */
546
+ capabilityConcentration: number;
547
+ dominantCapability?: string;
548
+ /** Dominant difficulty bucket's share of samples that declared a difficulty (0-1). */
549
+ difficultyConcentration: number;
550
+ dominantDifficulty?: 'easy' | 'medium' | 'hard';
551
+ /** Dominant construct's share of samples that declared a construct (0-1). */
552
+ constructConcentration: number;
553
+ dominantConstruct?: string;
507
554
  }
508
555
  export interface HedgingVerdict {
509
556
  isUncertainty: boolean;
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "oh-my-knowledge",
3
- "version": "0.40.0",
3
+ "version": "0.42.0",
4
4
  "packageManager": "yarn@4.16.0",
5
5
  "description": "Evaluation framework for LLM knowledge inputs — prompts, RAG corpora, skills, agent workflows. Fix the model, vary the artifact. Built-in statistical rigor: bootstrap CI, Krippendorff α, length-debias, saturation curves.",
6
6
  "type": "module",
@@ -1,83 +0,0 @@
1
- /**
2
- * Measure how much the judge's scores shift when the length-debias instruction
3
- * is toggled.
4
- *
5
- * What it actually measures
6
- * -------------------------
7
- * Given a finished report, this command re-judges every (sample, variant) pair
8
- * using the OPPOSITE length-debias setting from the original run. It then
9
- * compares the two score distributions with a bootstrap CI on the mean
10
- * difference.
11
- *
12
- * If the original report ran with debias-on (v3-cot-length), we re-judge with
13
- * v2-cot (legacy). If the original ran with debias-off, we re-judge with
14
- * v3-cot-length. Significant difference → the prompt change moves scores → the
15
- * judge is sensitive to the length-debias instruction. That's *consistent with*
16
- * length bias being present, but it doesn't prove it directly — a perfectly
17
- * length-neutral judge could in principle also be sensitive to the wording for
18
- * other reasons. We label the verdict accordingly.
19
- *
20
- * Cost
21
- * ----
22
- * Re-judging is a full second pass over all (sample, variant) cells. Cost
23
- * doubles vs the original run. The CLI surfaces this so users running on
24
- * large/expensive evaluations can opt in deliberately.
25
- */
26
- import type { ExecutorFn, Report, Sample } from '../types/index.js';
27
- import { type BootstrapDiffCI } from '../eval-core/bootstrap.js';
28
- export interface DebiasValidateInput {
29
- report: Report;
30
- samples: Sample[];
31
- judgeExecutor: ExecutorFn;
32
- judgeModel: string;
33
- /** Variant to validate. Defaults to first variant. */
34
- variant?: string;
35
- /** Bootstrap iterations for the diff CI. Default 1000. */
36
- bootstrapSamples?: number;
37
- seed?: number;
38
- /** Progress hook. */
39
- onProgress?: (info: {
40
- sample_id: string;
41
- completed: number;
42
- total: number;
43
- }) => void;
44
- }
45
- export interface DebiasValidateResult {
46
- variant: string;
47
- /** Original lengthDebias setting (true if debias-on at run time). */
48
- originalLengthDebias: boolean;
49
- /** Pairs of (originalScore, alternateScore) per sample. */
50
- pairs: Array<{
51
- sample_id: string;
52
- originalScore: number;
53
- alternateScore: number;
54
- }>;
55
- meanOriginal: number;
56
- meanAlternate: number;
57
- /** Mean of (alternate - original). Positive = alternate prompt scored higher. */
58
- diffCI: BootstrapDiffCI;
59
- /** Verdict in the {未检测, 弱, 中, 强} bucket plus an English shadow. */
60
- verdict: {
61
- zh: string;
62
- en: string;
63
- level: 'none' | 'weak' | 'medium' | 'strong';
64
- };
65
- /** Total cost burned re-judging. */
66
- alternateJudgeCostUSD: number;
67
- /** Sample_ids that the report had but lacked judge scores. */
68
- unscored: string[];
69
- /** Sample_ids in the samples file that are missing from the report. */
70
- missing: string[];
71
- }
72
- /**
73
- * Re-judge every sample of `variant` in the given report with the OPPOSITE
74
- * lengthDebias setting and compute the bootstrap CI on the mean difference.
75
- *
76
- * The judge call uses the rubric from the samples file and the output stored
77
- * in the report — we do NOT re-execute the model. Only judging is repeated.
78
- *
79
- * Multi-dimensional samples currently use the rubric as fallback when there's
80
- * no top-level rubric. Per-dimension validation can be added later if needed.
81
- */
82
- export declare function validateLengthDebias(input: DebiasValidateInput): Promise<DebiasValidateResult>;
83
- export declare function formatDebiasValidate(result: DebiasValidateResult): string;