oh-my-knowledge 0.40.0 → 0.42.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +8 -4
- package/README.zh.md +8 -4
- package/dist/analysis/report-diagnostics.d.ts +8 -1
- package/dist/analysis/report-diagnostics.js +109 -1
- package/dist/assets/agent-skills/omk/references/commands.md +2 -2
- package/dist/authoring/evolver.d.ts +3 -14
- package/dist/authoring/evolver.js +1 -52
- package/dist/authoring/generator.d.ts +24 -0
- package/dist/authoring/generator.js +36 -8
- package/dist/cli/commands/eval/index.d.ts +1 -1
- package/dist/cli/commands/eval/index.js +50 -15
- package/dist/cli/commands/init.js +10 -7
- package/dist/cli/lib/cmd-flags.d.ts +0 -1
- package/dist/cli/lib/i18n-dict/init.d.ts +1 -1
- package/dist/cli/lib/i18n-dict/init.js +14 -11
- package/dist/cli/lib/i18n-dict/run.d.ts +1 -1
- package/dist/cli/lib/i18n-dict/run.js +6 -2
- package/dist/cli/lib/parse-run-config.d.ts +3 -1
- package/dist/cli/lib/parse-run-config.js +0 -2
- package/dist/eval-core/evaluation-job.d.ts +2 -2
- package/dist/eval-core/evaluation-job.js +2 -2
- package/dist/eval-core/evaluation-reporting.d.ts +0 -1
- package/dist/eval-core/evaluation-reporting.js +9 -41
- package/dist/eval-core/execution-strategy.js +3 -2
- package/dist/eval-core/holdout.d.ts +66 -0
- package/dist/eval-core/holdout.js +118 -0
- package/dist/eval-core/judge-independence.d.ts +28 -0
- package/dist/eval-core/judge-independence.js +29 -0
- package/dist/eval-core/verdict.d.ts +53 -2
- package/dist/eval-core/verdict.js +216 -16
- package/dist/eval-workflows/batch-evaluation-workflow.d.ts +1 -1
- package/dist/eval-workflows/batch-evaluation-workflow.js +1 -2
- package/dist/eval-workflows/evaluation-pipeline/report-finalize.d.ts +1 -5
- package/dist/eval-workflows/evaluation-pipeline/report-finalize.js +19 -8
- package/dist/eval-workflows/evaluation-pipeline/run-state.d.ts +2 -2
- package/dist/eval-workflows/evaluation-pipeline/run-state.js +2 -2
- package/dist/eval-workflows/evaluation-pipeline.d.ts +3 -2
- package/dist/eval-workflows/evaluation-pipeline.js +3 -4
- package/dist/eval-workflows/run-evaluation.d.ts +6 -4
- package/dist/eval-workflows/run-evaluation.js +8 -6
- package/dist/executors/claude-cli.js +5 -6
- package/dist/executors/claude-sdk.d.ts +5 -2
- package/dist/executors/claude-sdk.js +13 -8
- package/dist/executors/codex-cli.js +3 -4
- package/dist/executors/shared.d.ts +2 -0
- package/dist/executors/shared.js +15 -0
- package/dist/grading/assertions.js +6 -122
- package/dist/grading/gold-cli.js +1 -1
- package/dist/grading/human-gold.d.ts +5 -3
- package/dist/grading/human-gold.js +5 -3
- package/dist/grading/index.d.ts +4 -4
- package/dist/grading/judge.d.ts +6 -14
- package/dist/grading/judge.js +5 -88
- package/dist/inputs/eval-config.js +12 -2
- package/dist/managed/evidence.js +1 -2
- package/dist/managed/version-scores.js +1 -1
- package/dist/renderer/html-renderer.js +0 -9
- package/dist/renderer/layout.js +4 -4
- package/dist/renderer/summary.js +59 -4
- package/dist/shared/llm-prompts/debias-instructions.d.ts +4 -0
- package/dist/shared/llm-prompts/debias-instructions.js +44 -0
- package/dist/shared/llm-prompts/judge-prompts.d.ts +30 -0
- package/dist/shared/llm-prompts/judge-prompts.js +205 -0
- package/dist/shared/llm-prompts/registry.d.ts +27 -0
- package/dist/shared/llm-prompts/registry.js +69 -0
- package/dist/types/eval.d.ts +15 -8
- package/dist/types/judge.d.ts +1 -1
- package/dist/types/report.d.ts +50 -3
- package/package.json +1 -1
- package/dist/grading/debias-validate.d.ts +0 -83
- package/dist/grading/debias-validate.js +0 -176
|
@@ -0,0 +1,205 @@
|
|
|
1
|
+
import { createHash } from 'node:crypto';
|
|
2
|
+
import { LENGTH_DEBIAS_INSTRUCTION, PRESENTATION_NEUTRALITY_INSTRUCTION, RAG_LENGTH_DEBIAS, RAG_PRESENTATION_NEUTRALITY, } from './debias-instructions.js';
|
|
3
|
+
// 评分类 prompt 单一来源 —— 直接决定分数的 LLM 评委 prompt(rubric 主评委 + RAG + 语义相似度)。
|
|
4
|
+
// 这些是测量学不变量:文本字节决定可比性,改动须配合 prompt-registry 的冻结 hash bump
|
|
5
|
+
// (BREAKING-COMPARABILITY)。执行器调用 / JSON 解析逻辑留在 grading/judge.ts 与 grading/assertions.ts。
|
|
6
|
+
// ===========================================================================
|
|
7
|
+
// Rubric 主评委 prompt
|
|
8
|
+
// ===========================================================================
|
|
9
|
+
/**
|
|
10
|
+
* Judge prompt template version.
|
|
11
|
+
*
|
|
12
|
+
* The version string encodes WHICH debias / context features the judge sees, so reports
|
|
13
|
+
* tagged with the same hash are score-comparable; a mismatched hash means "we changed how
|
|
14
|
+
* we ask the judge to think" and the reports should not be compared blind. Bump it (and the
|
|
15
|
+
* frozen hashes in `test/shared/prompt-registry-freeze.test.ts`) whenever the template's bytes
|
|
16
|
+
* change — that change is BREAKING-COMPARABILITY.
|
|
17
|
+
*
|
|
18
|
+
* Naming: a single main version (`v5`) + a `-feature` suffix per debias/context capability.
|
|
19
|
+
* The only asymmetry between the two strings is the `-len` suffix, gated by the
|
|
20
|
+
* `--no-debias-length` toggle; every other feature is always-on and appears in both.
|
|
21
|
+
*/
|
|
22
|
+
// 版本演进(内嵌"评委看到什么 / 被要求忽略什么"的语义):
|
|
23
|
+
// v2-cot : 仅看 output + rubric + 工具名分布(早期 buildTraceSummary)
|
|
24
|
+
// v3-cot-length : 加 length-debias 指令
|
|
25
|
+
// v3-cot-toolargs(off) / v4-cot-len-args(on)
|
|
26
|
+
// : 加 tool input 预览(让评委识别 wrapper-style skill 在 Bash 命令里的
|
|
27
|
+
// 真实语义调用,如 `Bash: mcporter --tool X`,否则误判"只调了 Bash")
|
|
28
|
+
// v5-cot-toolargs-fmt(off) / v5-cot-toolargs-fmt-len(on)
|
|
29
|
+
// : 加排版 / 语气中性化指令(始终开启,不给开关)。研究表明 LLM 评委除了
|
|
30
|
+
// 偏向更长的回答,还隐性偏向排版精致(标题 / 列表 / 加粗)与语气自信 / 自我
|
|
31
|
+
// 表扬(谄媚 / 权威偏置)的回答;显式指令要求评委只对照评分标准核内容。
|
|
32
|
+
// 同时借此把命名统一成单一主序号(v5)+ feature 后缀,`-len` 仍是开关那条。
|
|
33
|
+
const JUDGE_PROMPT_VERSION_DEBIAS_OFF = 'v5-cot-toolargs-fmt';
|
|
34
|
+
const JUDGE_PROMPT_VERSION_DEBIAS_ON = 'v5-cot-toolargs-fmt-len';
|
|
35
|
+
export const JUDGE_SYSTEM_PROMPT = '你是一个严格的 AI 输出质量评审员。先逐条对照评分标准做推理,再给最终分数。只返回 JSON,不要其他内容。';
|
|
36
|
+
export function buildJudgePrompt(prompt, rubric, output, traceSummary, lengthDebias = true) {
|
|
37
|
+
const version = lengthDebias ? JUDGE_PROMPT_VERSION_DEBIAS_ON : JUDGE_PROMPT_VERSION_DEBIAS_OFF;
|
|
38
|
+
const traceSection = traceSummary
|
|
39
|
+
? ['', '## Agent 执行过程', traceSummary, '', '请同时考虑执行过程的合理性(工具选择、步骤效率、错误恢复)。']
|
|
40
|
+
: [];
|
|
41
|
+
// 排版 / 语气中性化始终开启(不受 --no-debias-length 影响);length-debias 仍受开关控。
|
|
42
|
+
const neutralitySection = ['', PRESENTATION_NEUTRALITY_INSTRUCTION];
|
|
43
|
+
const debiasSection = lengthDebias ? ['', LENGTH_DEBIAS_INSTRUCTION] : [];
|
|
44
|
+
return [
|
|
45
|
+
`请对以下 AI 输出进行质量评分(template ${version})。`,
|
|
46
|
+
'',
|
|
47
|
+
'## 原始任务',
|
|
48
|
+
prompt,
|
|
49
|
+
'',
|
|
50
|
+
'## 评分标准',
|
|
51
|
+
rubric,
|
|
52
|
+
'',
|
|
53
|
+
'## AI 输出',
|
|
54
|
+
output,
|
|
55
|
+
...traceSection,
|
|
56
|
+
...neutralitySection,
|
|
57
|
+
...debiasSection,
|
|
58
|
+
'',
|
|
59
|
+
'## 评分流程',
|
|
60
|
+
'1. 逐条对照评分标准,先做推理(reasoning):列出 AI 输出哪些点对应哪条标准,哪些缺失,哪些有歧义。',
|
|
61
|
+
'2. 基于推理给出最终分数(1-5 的整数)和简短理由。',
|
|
62
|
+
'',
|
|
63
|
+
'请返回 JSON(不要包含 markdown 代码块标记):',
|
|
64
|
+
'{"reasoning": "<对照标准的逐条推理>", "score": <1-5的整数>, "reason": "<最终结论的简短理由>"}',
|
|
65
|
+
'',
|
|
66
|
+
'评分标准:1=完全不达标, 2=部分涉及, 3=基本达标, 4=较好, 5=优秀',
|
|
67
|
+
].join('\n');
|
|
68
|
+
}
|
|
69
|
+
/**
|
|
70
|
+
* Stable hash of the judge prompt template. Saved into ReportMeta.judgePromptHash so
|
|
71
|
+
* downstream readers can detect "the judge prompt changed between these two reports".
|
|
72
|
+
*
|
|
73
|
+
* `lengthDebias` defaults to true. Pass false (via `--no-debias-length`) to drop the
|
|
74
|
+
* length-debias instruction; that produces the debias-off prompt variant, whose hash
|
|
75
|
+
* differs from the default so the two are never compared blind.
|
|
76
|
+
*/
|
|
77
|
+
export function getJudgePromptHash(lengthDebias = true) {
|
|
78
|
+
const version = lengthDebias ? JUDGE_PROMPT_VERSION_DEBIAS_ON : JUDGE_PROMPT_VERSION_DEBIAS_OFF;
|
|
79
|
+
// Hash the template-shaping function source + the version tag together. We hash a
|
|
80
|
+
// deterministic stringified form of the template (with placeholder inputs) so any
|
|
81
|
+
// structural edit shows up.
|
|
82
|
+
const sample = buildJudgePrompt('<P>', '<R>', '<O>', '<T>', lengthDebias);
|
|
83
|
+
return createHash('sha256').update(version + '\n' + sample).digest('hex').slice(0, 12);
|
|
84
|
+
}
|
|
85
|
+
// ===========================================================================
|
|
86
|
+
// 语义相似度评委 prompt(semantic_similarity 断言)
|
|
87
|
+
// ===========================================================================
|
|
88
|
+
export const SEMANTIC_SIMILARITY_SYSTEM = '你是语义相似度评审员。只返回 JSON,不要其他内容。';
|
|
89
|
+
export function buildSemanticSimilarityPrompt(reference, output) {
|
|
90
|
+
return [
|
|
91
|
+
'请判断以下两段文本的语义相似度。',
|
|
92
|
+
'',
|
|
93
|
+
'## 参考文本',
|
|
94
|
+
reference,
|
|
95
|
+
'',
|
|
96
|
+
'## 待评估文本',
|
|
97
|
+
output,
|
|
98
|
+
'',
|
|
99
|
+
'请返回 JSON(不要包含 markdown 代码块标记):',
|
|
100
|
+
'{"score": <1-5的整数>, "reason": "<简短理由>"}',
|
|
101
|
+
'',
|
|
102
|
+
'评分:1=完全无关, 2=略有关联, 3=部分相似, 4=大致相同, 5=高度一致',
|
|
103
|
+
].join('\n');
|
|
104
|
+
}
|
|
105
|
+
/** RAG 评委 system + 用户 prompt。缺字段的早退 / 解析逻辑留在 assertions.ts。 */
|
|
106
|
+
export function buildRagJudgePrompt(type, fields) {
|
|
107
|
+
if (type === 'faithfulness') {
|
|
108
|
+
return {
|
|
109
|
+
system: '你是 RAG 评审员,专注判断输出是否被参考 context 支持。只返回 JSON。',
|
|
110
|
+
prompt: [
|
|
111
|
+
'请判断"待评估输出"中的事实性陈述是否被"参考 context"支持。',
|
|
112
|
+
'',
|
|
113
|
+
'## 参考 context',
|
|
114
|
+
fields.context ?? '',
|
|
115
|
+
'',
|
|
116
|
+
'## 待评估输出',
|
|
117
|
+
fields.output,
|
|
118
|
+
'',
|
|
119
|
+
RAG_LENGTH_DEBIAS,
|
|
120
|
+
RAG_PRESENTATION_NEUTRALITY,
|
|
121
|
+
'',
|
|
122
|
+
'## 评分流程',
|
|
123
|
+
'1. 列出待评估输出中所有事实性陈述',
|
|
124
|
+
'2. 逐条判断是否能在 context 中找到支持',
|
|
125
|
+
'3. 给出 1-5 分:',
|
|
126
|
+
' 5 = 全部陈述都有 context 支持,无编造',
|
|
127
|
+
' 4 = 多数有支持,有 1-2 处不重要的编造',
|
|
128
|
+
' 3 = 一半有支持',
|
|
129
|
+
' 2 = 多数无支持',
|
|
130
|
+
' 1 = 完全编造或与 context 矛盾',
|
|
131
|
+
'',
|
|
132
|
+
'请返回 JSON(不要 markdown 代码块):',
|
|
133
|
+
'{"score": <1-5的整数>, "reason": "<简短理由>"}',
|
|
134
|
+
].join('\n'),
|
|
135
|
+
};
|
|
136
|
+
}
|
|
137
|
+
if (type === 'answer_relevancy') {
|
|
138
|
+
return {
|
|
139
|
+
system: '你是答题切题度评审员。只返回 JSON。',
|
|
140
|
+
prompt: [
|
|
141
|
+
'请判断"AI 输出"是否直接、切题地回答了"用户问题"。',
|
|
142
|
+
'',
|
|
143
|
+
'## 用户问题',
|
|
144
|
+
fields.question ?? '',
|
|
145
|
+
'',
|
|
146
|
+
'## AI 输出',
|
|
147
|
+
fields.output,
|
|
148
|
+
'',
|
|
149
|
+
RAG_LENGTH_DEBIAS,
|
|
150
|
+
RAG_PRESENTATION_NEUTRALITY,
|
|
151
|
+
'',
|
|
152
|
+
'## 评分',
|
|
153
|
+
'5 = 完整切题回答,无冗余无遗漏',
|
|
154
|
+
'4 = 切题但有少量冗余或小遗漏',
|
|
155
|
+
'3 = 部分切题,部分跑题或避而不答',
|
|
156
|
+
'2 = 大部分跑题',
|
|
157
|
+
'1 = 完全跑题或拒答',
|
|
158
|
+
'',
|
|
159
|
+
'请返回 JSON(不要 markdown 代码块):',
|
|
160
|
+
'{"score": <1-5的整数>, "reason": "<简短理由>"}',
|
|
161
|
+
].join('\n'),
|
|
162
|
+
};
|
|
163
|
+
}
|
|
164
|
+
// context_recall
|
|
165
|
+
return {
|
|
166
|
+
system: '你是 context 覆盖率评审员。只返回 JSON。',
|
|
167
|
+
prompt: [
|
|
168
|
+
'请判断"参考 gold"中的关键事实在"AI 输出"中被覆盖的程度。',
|
|
169
|
+
'',
|
|
170
|
+
'## 参考 gold',
|
|
171
|
+
fields.reference ?? '',
|
|
172
|
+
'',
|
|
173
|
+
'## AI 输出',
|
|
174
|
+
fields.output,
|
|
175
|
+
'',
|
|
176
|
+
RAG_LENGTH_DEBIAS,
|
|
177
|
+
RAG_PRESENTATION_NEUTRALITY,
|
|
178
|
+
'',
|
|
179
|
+
'## 评分流程',
|
|
180
|
+
'1. 列出参考中的关键事实(忽略修饰性内容)',
|
|
181
|
+
'2. 检查每条是否在输出中被提及/使用',
|
|
182
|
+
'3. 给出 1-5 分:',
|
|
183
|
+
' 5 = 全部关键事实被覆盖',
|
|
184
|
+
' 4 = 大部分覆盖,缺 1-2 条次要事实',
|
|
185
|
+
' 3 = 一半覆盖',
|
|
186
|
+
' 2 = 仅覆盖少量',
|
|
187
|
+
' 1 = 完全未覆盖',
|
|
188
|
+
'',
|
|
189
|
+
'请返回 JSON(不要 markdown 代码块):',
|
|
190
|
+
'{"score": <1-5的整数>, "reason": "<简短理由>"}',
|
|
191
|
+
].join('\n'),
|
|
192
|
+
};
|
|
193
|
+
}
|
|
194
|
+
/** 评分类断言 prompt 的 drift-detection hash(占位符输入,12 位,仅供 prompt-registry 冻结)。 */
|
|
195
|
+
function hashPromptSample(label, system, prompt) {
|
|
196
|
+
return createHash('sha256').update(`${label}\n${system}\n${prompt}`).digest('hex').slice(0, 12);
|
|
197
|
+
}
|
|
198
|
+
export function getSemanticPromptHash() {
|
|
199
|
+
return hashPromptSample('semantic_similarity', SEMANTIC_SIMILARITY_SYSTEM, buildSemanticSimilarityPrompt('<R>', '<O>'));
|
|
200
|
+
}
|
|
201
|
+
export function getRagJudgePromptHash(type) {
|
|
202
|
+
const fields = { output: '<O>', context: '<C>', question: '<Q>', reference: '<R>' };
|
|
203
|
+
const { system, prompt } = buildRagJudgePrompt(type, fields);
|
|
204
|
+
return hashPromptSample(`rag:${type}`, system, prompt);
|
|
205
|
+
}
|
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* omk 所有 LLM prompt 的单一编目 —— 发现入口 + 测量学冻结入口。
|
|
3
|
+
*
|
|
4
|
+
* 解决「prompt 散在多个目录、难管理」与「只有 2 个 prompt 被冻结」两个问题:
|
|
5
|
+
* - 每条登记 prompt 的位置(module)+ 用途,一处可查全。
|
|
6
|
+
* - measurementInvariant 为 true 的(直接决定分数的评委 prompt)带 getHash,
|
|
7
|
+
* 由 test/shared/prompt-registry-freeze.test.ts 统一冻结,文本漂移即 CI 红。
|
|
8
|
+
*
|
|
9
|
+
* 注意:hash 形状按各自历史口径,不强行统一 ——
|
|
10
|
+
* - rubric 评委:12 位(getJudgePromptHash,持久化进 report.meta.judgePromptHash 的同口径);
|
|
11
|
+
* - RAG / semantic:12 位(getRagJudgePromptHash / getSemanticPromptHash,仅供冻结 drift 检测);
|
|
12
|
+
* - observe 复盘:64 位(readPromptDocument 读 .md 全文 sha256)。
|
|
13
|
+
* 用限定名 `promptId`(非裸 `kind`,见 terminology-spec §5.4)。
|
|
14
|
+
*/
|
|
15
|
+
export interface PromptRegistryEntry {
|
|
16
|
+
/** 稳定标识(冻结表的 key)。 */
|
|
17
|
+
promptId: string;
|
|
18
|
+
/** 人类可读用途。 */
|
|
19
|
+
purpose: string;
|
|
20
|
+
/** 定义位置(发现用,相对仓库根的源文件路径)。 */
|
|
21
|
+
module: string;
|
|
22
|
+
/** 是否直接决定分数 → 须冻结防漂移。 */
|
|
23
|
+
measurementInvariant: boolean;
|
|
24
|
+
/** 仅 measurementInvariant 条目提供:当前 prompt 的 hash。 */
|
|
25
|
+
getHash?: () => string;
|
|
26
|
+
}
|
|
27
|
+
export declare const PROMPT_REGISTRY: PromptRegistryEntry[];
|
|
@@ -0,0 +1,69 @@
|
|
|
1
|
+
import { getJudgePromptHash, getSemanticPromptHash, getRagJudgePromptHash } from './judge-prompts.js';
|
|
2
|
+
import { readPromptDocument } from './index.js';
|
|
3
|
+
import { PROMPTS_DIR, SOFT_STANDARD_PROMPT_ID, SOFT_STANDARD_PROMPT_VERSION } from '../../observability/soft-standards/constants.js';
|
|
4
|
+
export const PROMPT_REGISTRY = [
|
|
5
|
+
// —— 测量学不变量:直接决定分数的评委 prompt,统一冻结 ——
|
|
6
|
+
{
|
|
7
|
+
promptId: 'rubric-judge-debias-on',
|
|
8
|
+
purpose: 'rubric 主评委(length-debias 开,默认)',
|
|
9
|
+
module: 'src/shared/llm-prompts/judge-prompts.ts',
|
|
10
|
+
measurementInvariant: true,
|
|
11
|
+
getHash: () => getJudgePromptHash(true),
|
|
12
|
+
},
|
|
13
|
+
{
|
|
14
|
+
promptId: 'rubric-judge-debias-off',
|
|
15
|
+
purpose: 'rubric 主评委(--no-debias-length,length-debias 关)',
|
|
16
|
+
module: 'src/shared/llm-prompts/judge-prompts.ts',
|
|
17
|
+
measurementInvariant: true,
|
|
18
|
+
getHash: () => getJudgePromptHash(false),
|
|
19
|
+
},
|
|
20
|
+
{
|
|
21
|
+
promptId: 'semantic-similarity',
|
|
22
|
+
purpose: 'semantic_similarity 断言评委',
|
|
23
|
+
module: 'src/shared/llm-prompts/judge-prompts.ts',
|
|
24
|
+
measurementInvariant: true,
|
|
25
|
+
getHash: () => getSemanticPromptHash(),
|
|
26
|
+
},
|
|
27
|
+
{
|
|
28
|
+
promptId: 'rag-faithfulness',
|
|
29
|
+
purpose: 'RAG faithfulness(输出是否被 context 支持)',
|
|
30
|
+
module: 'src/shared/llm-prompts/judge-prompts.ts',
|
|
31
|
+
measurementInvariant: true,
|
|
32
|
+
getHash: () => getRagJudgePromptHash('faithfulness'),
|
|
33
|
+
},
|
|
34
|
+
{
|
|
35
|
+
promptId: 'rag-answer-relevancy',
|
|
36
|
+
purpose: 'RAG answer_relevancy(是否切题)',
|
|
37
|
+
module: 'src/shared/llm-prompts/judge-prompts.ts',
|
|
38
|
+
measurementInvariant: true,
|
|
39
|
+
getHash: () => getRagJudgePromptHash('answer_relevancy'),
|
|
40
|
+
},
|
|
41
|
+
{
|
|
42
|
+
promptId: 'rag-context-recall',
|
|
43
|
+
purpose: 'RAG context_recall(关键事实覆盖率)',
|
|
44
|
+
module: 'src/shared/llm-prompts/judge-prompts.ts',
|
|
45
|
+
measurementInvariant: true,
|
|
46
|
+
getHash: () => getRagJudgePromptHash('context_recall'),
|
|
47
|
+
},
|
|
48
|
+
{
|
|
49
|
+
promptId: 'observe-llm-enhanced-review',
|
|
50
|
+
purpose: 'observe 运行期软标准抽取 / LLM 增强复盘',
|
|
51
|
+
module: 'src/observability/prompts/llm-enhanced-review.prompt.md',
|
|
52
|
+
measurementInvariant: true,
|
|
53
|
+
getHash: () => readPromptDocument({
|
|
54
|
+
dir: PROMPTS_DIR,
|
|
55
|
+
fileName: `${SOFT_STANDARD_PROMPT_ID}.prompt.md`,
|
|
56
|
+
id: SOFT_STANDARD_PROMPT_ID,
|
|
57
|
+
version: SOFT_STANDARD_PROMPT_VERSION,
|
|
58
|
+
}).hash,
|
|
59
|
+
},
|
|
60
|
+
// —— 非评分类:不决定 composite / verdict / assertion 分数,不冻结,仅登记供发现 ——
|
|
61
|
+
{ promptId: 'failure-diagnostic', purpose: '失败诊断(root cause / 修复建议)', module: 'src/grading/diagnostic.ts', measurementInvariant: false },
|
|
62
|
+
{ promptId: 'failure-clusterer', purpose: '失败案例聚类', module: 'src/analysis/failure-clusterer.ts', measurementInvariant: false },
|
|
63
|
+
{ promptId: 'hedging-classifier', purpose: 'hedging 判定(喂 gap-signal,非评分)', module: 'src/analysis/hedging-classifier.ts', measurementInvariant: false },
|
|
64
|
+
{ promptId: 'sample-generator', purpose: '用例生成(skill / trace → samples)', module: 'src/authoring/generator.ts', measurementInvariant: false },
|
|
65
|
+
{ promptId: 'sample-fixer', purpose: '坏用例修复', module: 'src/authoring/sample-fixer.ts', measurementInvariant: false },
|
|
66
|
+
{ promptId: 'skill-improve', purpose: 'skill 迭代改进(evolve)', module: 'src/authoring/evolver.ts', measurementInvariant: false },
|
|
67
|
+
{ promptId: 'doctor-fixer', purpose: 'doctor 健康项修复向导', module: 'src/doctor/fixer.ts', measurementInvariant: false },
|
|
68
|
+
{ promptId: 'skill-health', purpose: 'skill 健康检查打分', module: 'src/shared/llm-prompts/skill-health.ts', measurementInvariant: false },
|
|
69
|
+
];
|
package/dist/types/eval.d.ts
CHANGED
|
@@ -211,7 +211,6 @@ export interface EvalConfig {
|
|
|
211
211
|
timeoutMs?: number;
|
|
212
212
|
noCache?: boolean;
|
|
213
213
|
noJudge?: boolean;
|
|
214
|
-
blind?: boolean;
|
|
215
214
|
mcpConfig?: string;
|
|
216
215
|
variants: EvalConfigVariant[];
|
|
217
216
|
/** hard budget caps. When any limit is hit during a run, remaining
|
|
@@ -221,6 +220,9 @@ export interface EvalConfig {
|
|
|
221
220
|
budget?: EvalBudget;
|
|
222
221
|
/** --repeat N. Multi-run variance analysis. */
|
|
223
222
|
repeat?: number;
|
|
223
|
+
/** --holdout-ratio R (0 < R < 1). Hold out a deterministic sample slice and
|
|
224
|
+
* report train vs holdout composite as a generalization / overfitting signal. */
|
|
225
|
+
holdoutRatio?: number;
|
|
224
226
|
/** --judge-repeat N. Each (sample × dimension) judged N times for self-consistency stddev. */
|
|
225
227
|
judgeRepeat?: number;
|
|
226
228
|
/** --bootstrap. Distribution-free CI per variant + pairwise diff. */
|
|
@@ -229,7 +231,7 @@ export interface EvalConfig {
|
|
|
229
231
|
bootstrapSamples?: number;
|
|
230
232
|
/** --gold-dir. After-run automatic comparison against a human-anchor dataset. */
|
|
231
233
|
goldDir?: string;
|
|
232
|
-
/** --no-debias-length flips this to false. Default true (
|
|
234
|
+
/** --no-debias-length flips this to false. Default true (length-debias instruction on). */
|
|
233
235
|
lengthDebias?: boolean;
|
|
234
236
|
/** --no-strict-baseline flips this to false. Default true (baseline-kind allowedSkills=[]). */
|
|
235
237
|
strictBaseline?: boolean;
|
|
@@ -256,9 +258,12 @@ export interface EvaluationRequest {
|
|
|
256
258
|
timeoutMs?: number;
|
|
257
259
|
noCache: boolean;
|
|
258
260
|
dryRun: boolean;
|
|
259
|
-
blind: boolean;
|
|
260
261
|
/** --repeat N; 1 表示单次跑,> 1 走 runMultiple 做 variance 分析 */
|
|
261
262
|
repeat?: number;
|
|
263
|
+
/** --holdout-ratio R; 0 / 缺省表示不切分(默认)。> 0 时 report-finalize 在结果上
|
|
264
|
+
* post-hoc 切出 train / holdout 子集算综合分(`report.analysis.holdout`),供 verdict
|
|
265
|
+
* 的过拟合门控读取。see src/eval-core/holdout.ts */
|
|
266
|
+
holdoutRatio?: number;
|
|
262
267
|
/** --batch; default absent/false. True means skill-batch mode. */
|
|
263
268
|
batch?: boolean;
|
|
264
269
|
/** --judge-repeat N; 每条 sample × dimension 用 LLM judge 跑 N 次, 输出 stddev. 默认 1 (单次). */
|
|
@@ -266,8 +271,9 @@ export interface EvaluationRequest {
|
|
|
266
271
|
/** Unified judge config — always non-empty.
|
|
267
272
|
* - length === 1: single judge (degenerate ensemble of size 1).
|
|
268
273
|
* - length >= 2: multi-judge ensemble. Each (sample × dimension) is scored by every judge,
|
|
269
|
-
* inter-judge agreement (Pearson + mean absolute difference) reported as
|
|
270
|
-
*
|
|
274
|
+
* inter-judge agreement (Pearson + mean absolute difference) reported as a rebuttal to
|
|
275
|
+
* "judge same-model bias" — but only when judges span vendors; a single-vendor ensemble's
|
|
276
|
+
* high agreement reflects shared bias, not independence (flagged by `single_vendor_ensemble`).
|
|
271
277
|
* When `noJudge: true` the entry is preserved for audit but no judge call actually runs. */
|
|
272
278
|
judgeModels: JudgeConfig[];
|
|
273
279
|
/** --bootstrap; true 时 aggregateReport 加跑 bootstrap mean/diff CI, 写入 VariantSummary.
|
|
@@ -275,9 +281,10 @@ export interface EvaluationRequest {
|
|
|
275
281
|
bootstrap?: boolean;
|
|
276
282
|
/** --bootstrap-samples N; bootstrap 重采样次数, 默认 1000. > 10000 时 stderr 警告. */
|
|
277
283
|
bootstrapSamples?: number;
|
|
278
|
-
/** length-debias toggle. Default true
|
|
279
|
-
* CLI flag --no-debias-length flips to false (
|
|
280
|
-
* value is reflected in
|
|
284
|
+
/** length-debias toggle. Default true — judge prompt carries the length-debias
|
|
285
|
+
* instruction. CLI flag --no-debias-length flips to false (drops that instruction,
|
|
286
|
+
* the debias-off prompt variant). The active value is reflected in
|
|
287
|
+
* ReportMeta.judgePromptHash and ReportMeta.debiasMode. */
|
|
281
288
|
lengthDebias?: boolean;
|
|
282
289
|
/** hard budget caps. See EvalBudget. */
|
|
283
290
|
budget?: EvalBudget;
|
package/dist/types/judge.d.ts
CHANGED
|
@@ -18,7 +18,7 @@ export interface JudgeRuntimeEntry {
|
|
|
18
18
|
}
|
|
19
19
|
/** Per-judge ensemble entry: which judge gave what score (mean over judge-repeat if N>1). */
|
|
20
20
|
export interface EnsembleJudgeResult {
|
|
21
|
-
/** "executor:model" identifier — e.g. "claude:opus" or "openai:gpt-4o". */
|
|
21
|
+
/** "executor:model" identifier — e.g. "claude:opus" or "openai-api:gpt-4o". */
|
|
22
22
|
judge: string;
|
|
23
23
|
/** Mean score from this judge over judge-repeat calls (or single score if repeat=1). */
|
|
24
24
|
score: number;
|
package/dist/types/report.d.ts
CHANGED
|
@@ -317,7 +317,7 @@ export interface ReportMeta {
|
|
|
317
317
|
humanAgreement?: ReportHumanAgreement;
|
|
318
318
|
variantConfigs?: VariantConfig[];
|
|
319
319
|
/** Skill isolation 快照(per-variant)。
|
|
320
|
-
* key = variant name;value = allowedSkills(undefined → null,SDK 默认全发现 / [] →
|
|
320
|
+
* key = variant name;value = allowedSkills(undefined → null,SDK 默认全发现 / [] → 完全隔离;非空白名单已移除)。
|
|
321
321
|
* 跨报告对比 verdict / Δ 时,isolation 状态不一致会被 stderr warn 标"不可比"。
|
|
322
322
|
* 字段缺失意味着报告产自 之前(默认全发现,construct validity 不保证)。 */
|
|
323
323
|
skillIsolation?: Record<string, string[] | null>;
|
|
@@ -325,8 +325,6 @@ export interface ReportMeta {
|
|
|
325
325
|
run?: EvaluationRun;
|
|
326
326
|
job?: EvaluationJob;
|
|
327
327
|
gitInfo?: GitInfo | null;
|
|
328
|
-
blind?: boolean;
|
|
329
|
-
blindMap?: Record<string, string>;
|
|
330
328
|
layeredStats?: boolean;
|
|
331
329
|
/** Evolve 合并报告的原始 skill 归属。variants 会被 relabel 为 round-0/round-1,
|
|
332
330
|
* Studio skill 索引用该字段把报告归回 skill 卡片。 */
|
|
@@ -482,6 +480,34 @@ export interface AnalysisResult {
|
|
|
482
480
|
* (capability / difficulty / construct / provenance); persisted on report
|
|
483
481
|
* for studio to surface coverage gaps. See docs/specs/sample-design-spec.md. */
|
|
484
482
|
sampleQuality?: SampleQualityAggregate;
|
|
483
|
+
/** Opt-in train/holdout generalization breakdown (`omk eval --holdout-ratio`).
|
|
484
|
+
* Absent on default runs; present only when a holdout ratio was requested. */
|
|
485
|
+
holdout?: HoldoutBreakdown;
|
|
486
|
+
}
|
|
487
|
+
/** Train vs holdout composite breakdown for `omk eval --holdout-ratio`.
|
|
488
|
+
* Computed post-hoc from `report.results` by `computeHoldoutBreakdown`
|
|
489
|
+
* (`src/eval-core/holdout.ts`), sharing the same testSetHash watermark as
|
|
490
|
+
* gapReports (gap-spec §7.1). A large train − holdout composite gap is the
|
|
491
|
+
* sample-set-overfitting signal the verdict's overfitting gate reads. */
|
|
492
|
+
export interface HoldoutBreakdown {
|
|
493
|
+
/** Held-out fraction requested via --holdout-ratio. */
|
|
494
|
+
ratio: number;
|
|
495
|
+
/** true when either side fell below the minimum subset size → scored full-set,
|
|
496
|
+
* no usable split. `perVariant` is empty and the verdict gate stays inert. */
|
|
497
|
+
disabled?: boolean;
|
|
498
|
+
/** Per-variant train vs holdout composite (1-5 scale). `*Count` is the authored
|
|
499
|
+
* split size; `*Scorable` is how many of those actually produced a composite (> 0)
|
|
500
|
+
* — they diverge under partial errors, and the overfitting gate trusts `*Scorable`. */
|
|
501
|
+
perVariant: Record<string, {
|
|
502
|
+
trainScore: number;
|
|
503
|
+
holdoutScore: number;
|
|
504
|
+
trainCount: number;
|
|
505
|
+
holdoutCount: number;
|
|
506
|
+
trainScorable: number;
|
|
507
|
+
holdoutScorable: number;
|
|
508
|
+
}>;
|
|
509
|
+
testSetPath?: string | null;
|
|
510
|
+
testSetHash?: string | null;
|
|
485
511
|
}
|
|
486
512
|
/** Aggregated sample design coverage stats. Built by
|
|
487
513
|
* `buildSampleQualityAggregate(samples)` from `Sample.capability` /
|
|
@@ -504,6 +530,27 @@ export interface SampleQualityAggregate {
|
|
|
504
530
|
sampleCountWithDifficulty: number;
|
|
505
531
|
sampleCountWithConstruct: number;
|
|
506
532
|
sampleCountWithProvenance: number;
|
|
533
|
+
/** Relative-balance / skew of the sample set (derived from the distributions
|
|
534
|
+
* above). Flags over-representation — "70% of samples are easy" — without an
|
|
535
|
+
* external denominator. Diagnostic only; never feeds grading / judge / verdict. */
|
|
536
|
+
representativeness?: Representativeness;
|
|
537
|
+
}
|
|
538
|
+
/** Distribution skew over what the sample set declares. Pure relative balance —
|
|
539
|
+
* there is no authored "expected" capability list to measure absolute coverage
|
|
540
|
+
* against (capabilities are free-form strings), so this reports concentration
|
|
541
|
+
* (dominant bucket share, 0-1) and the dominant label per dimension. */
|
|
542
|
+
export interface Representativeness {
|
|
543
|
+
/** Distinct capabilities declared across the set. */
|
|
544
|
+
capabilityCount: number;
|
|
545
|
+
/** Dominant capability's share of all capability tags (0-1); 0 when none declared. */
|
|
546
|
+
capabilityConcentration: number;
|
|
547
|
+
dominantCapability?: string;
|
|
548
|
+
/** Dominant difficulty bucket's share of samples that declared a difficulty (0-1). */
|
|
549
|
+
difficultyConcentration: number;
|
|
550
|
+
dominantDifficulty?: 'easy' | 'medium' | 'hard';
|
|
551
|
+
/** Dominant construct's share of samples that declared a construct (0-1). */
|
|
552
|
+
constructConcentration: number;
|
|
553
|
+
dominantConstruct?: string;
|
|
507
554
|
}
|
|
508
555
|
export interface HedgingVerdict {
|
|
509
556
|
isUncertainty: boolean;
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "oh-my-knowledge",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.42.0",
|
|
4
4
|
"packageManager": "yarn@4.16.0",
|
|
5
5
|
"description": "Evaluation framework for LLM knowledge inputs — prompts, RAG corpora, skills, agent workflows. Fix the model, vary the artifact. Built-in statistical rigor: bootstrap CI, Krippendorff α, length-debias, saturation curves.",
|
|
6
6
|
"type": "module",
|
|
@@ -1,83 +0,0 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* Measure how much the judge's scores shift when the length-debias instruction
|
|
3
|
-
* is toggled.
|
|
4
|
-
*
|
|
5
|
-
* What it actually measures
|
|
6
|
-
* -------------------------
|
|
7
|
-
* Given a finished report, this command re-judges every (sample, variant) pair
|
|
8
|
-
* using the OPPOSITE length-debias setting from the original run. It then
|
|
9
|
-
* compares the two score distributions with a bootstrap CI on the mean
|
|
10
|
-
* difference.
|
|
11
|
-
*
|
|
12
|
-
* If the original report ran with debias-on (v3-cot-length), we re-judge with
|
|
13
|
-
* v2-cot (legacy). If the original ran with debias-off, we re-judge with
|
|
14
|
-
* v3-cot-length. Significant difference → the prompt change moves scores → the
|
|
15
|
-
* judge is sensitive to the length-debias instruction. That's *consistent with*
|
|
16
|
-
* length bias being present, but it doesn't prove it directly — a perfectly
|
|
17
|
-
* length-neutral judge could in principle also be sensitive to the wording for
|
|
18
|
-
* other reasons. We label the verdict accordingly.
|
|
19
|
-
*
|
|
20
|
-
* Cost
|
|
21
|
-
* ----
|
|
22
|
-
* Re-judging is a full second pass over all (sample, variant) cells. Cost
|
|
23
|
-
* doubles vs the original run. The CLI surfaces this so users running on
|
|
24
|
-
* large/expensive evaluations can opt in deliberately.
|
|
25
|
-
*/
|
|
26
|
-
import type { ExecutorFn, Report, Sample } from '../types/index.js';
|
|
27
|
-
import { type BootstrapDiffCI } from '../eval-core/bootstrap.js';
|
|
28
|
-
export interface DebiasValidateInput {
|
|
29
|
-
report: Report;
|
|
30
|
-
samples: Sample[];
|
|
31
|
-
judgeExecutor: ExecutorFn;
|
|
32
|
-
judgeModel: string;
|
|
33
|
-
/** Variant to validate. Defaults to first variant. */
|
|
34
|
-
variant?: string;
|
|
35
|
-
/** Bootstrap iterations for the diff CI. Default 1000. */
|
|
36
|
-
bootstrapSamples?: number;
|
|
37
|
-
seed?: number;
|
|
38
|
-
/** Progress hook. */
|
|
39
|
-
onProgress?: (info: {
|
|
40
|
-
sample_id: string;
|
|
41
|
-
completed: number;
|
|
42
|
-
total: number;
|
|
43
|
-
}) => void;
|
|
44
|
-
}
|
|
45
|
-
export interface DebiasValidateResult {
|
|
46
|
-
variant: string;
|
|
47
|
-
/** Original lengthDebias setting (true if debias-on at run time). */
|
|
48
|
-
originalLengthDebias: boolean;
|
|
49
|
-
/** Pairs of (originalScore, alternateScore) per sample. */
|
|
50
|
-
pairs: Array<{
|
|
51
|
-
sample_id: string;
|
|
52
|
-
originalScore: number;
|
|
53
|
-
alternateScore: number;
|
|
54
|
-
}>;
|
|
55
|
-
meanOriginal: number;
|
|
56
|
-
meanAlternate: number;
|
|
57
|
-
/** Mean of (alternate - original). Positive = alternate prompt scored higher. */
|
|
58
|
-
diffCI: BootstrapDiffCI;
|
|
59
|
-
/** Verdict in the {未检测, 弱, 中, 强} bucket plus an English shadow. */
|
|
60
|
-
verdict: {
|
|
61
|
-
zh: string;
|
|
62
|
-
en: string;
|
|
63
|
-
level: 'none' | 'weak' | 'medium' | 'strong';
|
|
64
|
-
};
|
|
65
|
-
/** Total cost burned re-judging. */
|
|
66
|
-
alternateJudgeCostUSD: number;
|
|
67
|
-
/** Sample_ids that the report had but lacked judge scores. */
|
|
68
|
-
unscored: string[];
|
|
69
|
-
/** Sample_ids in the samples file that are missing from the report. */
|
|
70
|
-
missing: string[];
|
|
71
|
-
}
|
|
72
|
-
/**
|
|
73
|
-
* Re-judge every sample of `variant` in the given report with the OPPOSITE
|
|
74
|
-
* lengthDebias setting and compute the bootstrap CI on the mean difference.
|
|
75
|
-
*
|
|
76
|
-
* The judge call uses the rubric from the samples file and the output stored
|
|
77
|
-
* in the report — we do NOT re-execute the model. Only judging is repeated.
|
|
78
|
-
*
|
|
79
|
-
* Multi-dimensional samples currently use the rubric as fallback when there's
|
|
80
|
-
* no top-level rubric. Per-dimension validation can be added later if needed.
|
|
81
|
-
*/
|
|
82
|
-
export declare function validateLengthDebias(input: DebiasValidateInput): Promise<DebiasValidateResult>;
|
|
83
|
-
export declare function formatDebiasValidate(result: DebiasValidateResult): string;
|