oh-my-knowledge 0.40.0 → 0.41.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (61) hide show
  1. package/README.md +1 -2
  2. package/README.zh.md +1 -2
  3. package/dist/analysis/report-diagnostics.js +27 -0
  4. package/dist/assets/agent-skills/omk/references/commands.md +1 -2
  5. package/dist/authoring/generator.js +3 -2
  6. package/dist/cli/commands/eval/index.d.ts +0 -1
  7. package/dist/cli/commands/eval/index.js +10 -10
  8. package/dist/cli/lib/cmd-flags.d.ts +0 -1
  9. package/dist/cli/lib/i18n-dict/run.js +2 -2
  10. package/dist/cli/lib/parse-run-config.d.ts +0 -1
  11. package/dist/cli/lib/parse-run-config.js +0 -2
  12. package/dist/eval-core/evaluation-job.d.ts +1 -2
  13. package/dist/eval-core/evaluation-job.js +1 -2
  14. package/dist/eval-core/evaluation-reporting.d.ts +0 -1
  15. package/dist/eval-core/evaluation-reporting.js +2 -38
  16. package/dist/eval-core/execution-strategy.js +3 -2
  17. package/dist/eval-core/judge-independence.d.ts +28 -0
  18. package/dist/eval-core/judge-independence.js +29 -0
  19. package/dist/eval-core/verdict.d.ts +9 -1
  20. package/dist/eval-core/verdict.js +43 -5
  21. package/dist/eval-workflows/batch-evaluation-workflow.d.ts +1 -1
  22. package/dist/eval-workflows/batch-evaluation-workflow.js +1 -2
  23. package/dist/eval-workflows/evaluation-pipeline/report-finalize.d.ts +1 -5
  24. package/dist/eval-workflows/evaluation-pipeline/report-finalize.js +1 -8
  25. package/dist/eval-workflows/evaluation-pipeline/run-state.d.ts +1 -2
  26. package/dist/eval-workflows/evaluation-pipeline/run-state.js +1 -2
  27. package/dist/eval-workflows/evaluation-pipeline.d.ts +1 -2
  28. package/dist/eval-workflows/evaluation-pipeline.js +1 -3
  29. package/dist/eval-workflows/run-evaluation.d.ts +2 -3
  30. package/dist/eval-workflows/run-evaluation.js +1 -2
  31. package/dist/executors/claude-cli.js +5 -6
  32. package/dist/executors/claude-sdk.d.ts +5 -2
  33. package/dist/executors/claude-sdk.js +13 -8
  34. package/dist/executors/codex-cli.js +3 -4
  35. package/dist/executors/shared.d.ts +2 -0
  36. package/dist/executors/shared.js +15 -0
  37. package/dist/grading/assertions.js +6 -122
  38. package/dist/grading/gold-cli.js +1 -1
  39. package/dist/grading/human-gold.d.ts +5 -3
  40. package/dist/grading/human-gold.js +5 -3
  41. package/dist/grading/index.d.ts +4 -4
  42. package/dist/grading/judge.d.ts +6 -14
  43. package/dist/grading/judge.js +5 -88
  44. package/dist/inputs/eval-config.js +6 -2
  45. package/dist/managed/evidence.js +1 -2
  46. package/dist/managed/version-scores.js +1 -1
  47. package/dist/renderer/html-renderer.js +0 -9
  48. package/dist/renderer/layout.js +4 -4
  49. package/dist/renderer/summary.js +23 -1
  50. package/dist/shared/llm-prompts/debias-instructions.d.ts +4 -0
  51. package/dist/shared/llm-prompts/debias-instructions.js +44 -0
  52. package/dist/shared/llm-prompts/judge-prompts.d.ts +30 -0
  53. package/dist/shared/llm-prompts/judge-prompts.js +205 -0
  54. package/dist/shared/llm-prompts/registry.d.ts +27 -0
  55. package/dist/shared/llm-prompts/registry.js +69 -0
  56. package/dist/types/eval.d.ts +8 -8
  57. package/dist/types/judge.d.ts +1 -1
  58. package/dist/types/report.d.ts +1 -3
  59. package/package.json +1 -1
  60. package/dist/grading/debias-validate.d.ts +0 -83
  61. package/dist/grading/debias-validate.js +0 -176
@@ -1,4 +1,7 @@
1
- import { createHash } from 'node:crypto';
1
+ import { buildJudgePrompt, JUDGE_SYSTEM_PROMPT } from '../shared/llm-prompts/judge-prompts.js';
2
+ // 评分类 prompt 已收口到 shared/llm-prompts/judge-prompts.ts(单一来源 + prompt-registry 冻结)。
3
+ // 这里 re-export 保对外 API 不破:既有消费方仍从 grading/judge.js import 这两个符号。
4
+ export { buildJudgePrompt, getJudgePromptHash } from '../shared/llm-prompts/judge-prompts.js';
2
5
  function extractFirstJsonObject(text) {
3
6
  const start = text.indexOf('{');
4
7
  if (start === -1)
@@ -46,92 +49,6 @@ function salvageJudgeResponse(text) {
46
49
  reasoning: text.trim().slice(0, 2000),
47
50
  };
48
51
  }
49
- /**
50
- * Judge prompt template version.
51
- *
52
- * - 'v2-cot' — legacy; CoT scoring without explicit length-debias instruction.
53
- * Kept for `--no-debias-length` so users can reproduce historical
54
- * reports byte-for-byte.
55
- * - 'v3-cot-length' — adds a paragraph telling the judge that length is not a quality
56
- * signal. Default on (research consistently shows
57
- * LLM judges over-weight verbosity; explicit instruction mitigates).
58
- *
59
- * Bump when the prompt's intent or structure changes meaningfully — reports tagged
60
- * with the same hash are score-comparable; mismatched hashes mean "we changed how we
61
- * ask the judge to think" and should not be compared blind.
62
- */
63
- // 版本字符串内嵌"判官能看到什么"的语义。bump 时机:
64
- // v2-cot : 仅看 output + rubric + 工具名分布(早期 buildTraceSummary)
65
- // v3-cot-length : 加了 length-debias 指令(v0.21)
66
- // v3-cot-toolargs : 加了 tool input 预览(本次,BREAKING-COMPARABILITY)
67
- // v4-cot-len-args : v3-cot-length 的同步升级(BREAKING-COMPARABILITY)
68
- // bump 原因:之前 trace 只给 tool 名 + 分布,wrapper-style skill(mcporter / code-host CLI 等)
69
- // 被判官当成"只调了 Bash,没用 skill 指定的 MCP 工具"——结论事实错误。
70
- // tool input 预览让判官能识别 `Bash: mcporter --tool skylark_xxx` 内的真实语义调用。
71
- //
72
- // ⚠️ 命名约定(下次 bump 时执行):
73
- // 当前 debias OFF: v3-* / debias ON: v4-* 编号不对称(历史上 debias OFF 跑 v2→v3,
74
- // debias ON 跑 v3→v4)。这次再 bump 会让两条线持续偏移 v4/v5、v5/v6、…
75
- //
76
- // 下次 bump 时统一为单一主序号 + features 后缀,如:
77
- // JUDGE_PROMPT_VERSION_DEBIAS_OFF = 'v5-cot-<features>'
78
- // JUDGE_PROMPT_VERSION_DEBIAS_ON = 'v5-cot-<features>-len'
79
- // 这样 hash 测试自然分两条,主序号清晰对齐,features 字段独立描述差异。
80
- // 本 PR 不在意命名,只是把约定写下来,避免下次又跟着错误偏移走。
81
- const JUDGE_PROMPT_VERSION_DEBIAS_OFF = 'v3-cot-toolargs';
82
- const JUDGE_PROMPT_VERSION_DEBIAS_ON = 'v4-cot-len-args';
83
- const JUDGE_SYSTEM_PROMPT = '你是一个严格的 AI 输出质量评审员。先逐条对照评分标准做推理,再给最终分数。只返回 JSON,不要其他内容。';
84
- const LENGTH_DEBIAS_INSTRUCTION = [
85
- '## 重要:长度不是质量信号',
86
- '评分时聚焦内容实质与正确性。回答的篇幅、行文丰富度、结构复杂度本身不是质量指标 ——',
87
- '简洁正确的回答不应因短而扣分;冗长但偏题或重复的回答不应因长而加分。',
88
- '研究显示 LLM 评委容易隐性偏向更长的回答,请在打分前先警觉这一点。',
89
- ].join('\n');
90
- export function buildJudgePrompt(prompt, rubric, output, traceSummary, lengthDebias = true) {
91
- const version = lengthDebias ? JUDGE_PROMPT_VERSION_DEBIAS_ON : JUDGE_PROMPT_VERSION_DEBIAS_OFF;
92
- const traceSection = traceSummary
93
- ? ['', '## Agent 执行过程', traceSummary, '', '请同时考虑执行过程的合理性(工具选择、步骤效率、错误恢复)。']
94
- : [];
95
- const debiasSection = lengthDebias ? ['', LENGTH_DEBIAS_INSTRUCTION] : [];
96
- return [
97
- `请对以下 AI 输出进行质量评分(template ${version})。`,
98
- '',
99
- '## 原始任务',
100
- prompt,
101
- '',
102
- '## 评分标准',
103
- rubric,
104
- '',
105
- '## AI 输出',
106
- output,
107
- ...traceSection,
108
- ...debiasSection,
109
- '',
110
- '## 评分流程',
111
- '1. 逐条对照评分标准,先做推理(reasoning):列出 AI 输出哪些点对应哪条标准,哪些缺失,哪些有歧义。',
112
- '2. 基于推理给出最终分数(1-5 的整数)和简短理由。',
113
- '',
114
- '请返回 JSON(不要包含 markdown 代码块标记):',
115
- '{"reasoning": "<对照标准的逐条推理>", "score": <1-5的整数>, "reason": "<最终结论的简短理由>"}',
116
- '',
117
- '评分标准:1=完全不达标, 2=部分涉及, 3=基本达标, 4=较好, 5=优秀',
118
- ].join('\n');
119
- }
120
- /**
121
- * Stable hash of the judge prompt template. Saved into ReportMeta.judgePromptHash so
122
- * downstream readers can detect "the judge prompt changed between these two reports".
123
- *
124
- * `lengthDebias` defaults to true (v0.21+ default). Pass false when running under
125
- * `--no-debias-length` so the hash matches historical v2-cot reports.
126
- */
127
- export function getJudgePromptHash(lengthDebias = true) {
128
- const version = lengthDebias ? JUDGE_PROMPT_VERSION_DEBIAS_ON : JUDGE_PROMPT_VERSION_DEBIAS_OFF;
129
- // Hash the template-shaping function source + the version tag together. We hash a
130
- // deterministic stringified form of the template (with placeholder inputs) so any
131
- // structural edit shows up.
132
- const sample = buildJudgePrompt('<P>', '<R>', '<O>', '<T>', lengthDebias);
133
- return createHash('sha256').update(version + '\n' + sample).digest('hex').slice(0, 12);
134
- }
135
52
  function getErrorMessage(err) {
136
53
  return err instanceof Error ? err.message : String(err);
137
54
  }
@@ -412,7 +329,7 @@ export function computeJudgeAgreement(judgeScores) {
412
329
  }
413
330
  /**
414
331
  * Judge a single (output, rubric) pair with N judge models in parallel. Each judge
415
- * may use a different executor (e.g. claude:opus + openai:gpt-4o + gemini:pro). Each
332
+ * may use a different executor (e.g. claude:opus + openai-api:gpt-4o + gemini:pro). Each
416
333
  * judge can also be repeated `judgeRepeat` times — final per-judge score is its mean.
417
334
  *
418
335
  * Returns: aggregate DimensionResult (score = mean across judges; this is the "consensus"
@@ -107,6 +107,9 @@ function validateEvalConfig(parsed, configPath) {
107
107
  throw new Error(`${configPath}: variants[${i}].allowedSkills[${j}] must be a non-empty string`);
108
108
  }
109
109
  }
110
+ if (v.allowedSkills.length > 0) {
111
+ throw new Error(`${configPath}: variants[${i}].allowedSkills must be [] (strict isolation: no skills) or omitted (no isolation) — a non-empty skill whitelist is no longer supported because it could not be fully isolated (the subagent Skill tool and cwd filesystem channels leaked). For a multi-skill experiment, control the eval environment instead.`);
112
+ }
110
113
  allowedSkills = v.allowedSkills;
111
114
  }
112
115
  if (seen.has(v.name)) {
@@ -146,11 +149,13 @@ function validateEvalConfig(parsed, configPath) {
146
149
  if (obj.judgeModel !== undefined || obj.judgeExecutor !== undefined) {
147
150
  throw new Error(`${configPath}: \`judgeModel\` and \`judgeExecutor\` were removed in v0.25 — use \`judgeModels: [{executor, model}]\` instead (single judge is the 1-entry case). See README.`);
148
151
  }
152
+ if (obj.blind !== undefined) {
153
+ throw new Error(`${configPath}: \`blind\` (judge blind mode) was removed — delete it from your eval.yaml; reports are no longer blinded.`);
154
+ }
149
155
  assertNumberOpt('concurrency');
150
156
  assertNumberOpt('timeoutMs');
151
157
  assertBoolOpt('noCache');
152
158
  assertBoolOpt('noJudge');
153
- assertBoolOpt('blind');
154
159
  assertStringOpt('mcpConfig');
155
160
  assertStringOpt('goldDir');
156
161
  assertBoolOpt('bootstrap');
@@ -247,7 +252,6 @@ function validateEvalConfig(parsed, configPath) {
247
252
  timeoutMs: obj.timeoutMs,
248
253
  noCache: obj.noCache,
249
254
  noJudge: obj.noJudge,
250
- blind: obj.blind,
251
255
  mcpConfig: obj.mcpConfig,
252
256
  variants,
253
257
  budget,
@@ -13,8 +13,7 @@
13
13
  * `deriveManagedState` 按 contentHash 匹配裁定(重装新内容后旧证据留存供回滚,却不让新内容显得已测)。
14
14
  * - **跨源**:用三级身份消歧把被测 variant 对到受管记录(install 与 eval 命名不一致:记录名是
15
15
  * skill 短名 `review`,而 eval 报告 variant key 可能是整串表达式 `git:HEAD:skills/review`、
16
- * eval.yaml 别名 `candidate`、blind 模式的 `A`/`B`)。`applyBlindMode` 盲化 `variants` 但**不**动
17
- * `artifactHashes` / `variantConfigs` 的键面,故三级都按真实键工作:
16
+ * eval.yaml 别名 `candidate`)。三级都按 `artifactHashes` / `variantConfigs` 的真实键工作:
18
17
  * (a) 显式**同名** variant —— 最强身份(本地 `--treatment <name>` 与 drift);
19
18
  * (b) **结构化源匹配** —— `variantConfigs[].locator(+ref)` 与 `record.source` 对齐(git / 远端 /
20
19
  * 别名),即便内容撞哈也能精确消歧;
@@ -33,7 +33,7 @@ export function buildVersionScores(record, reportsById) {
33
33
  continue;
34
34
  const variant = Object.keys(hashes).find((v) => hashes[v] === ev.contentHash);
35
35
  if (!variant)
36
- continue; // 匹配不到变体(旧 schema 不绑 / blind 未带哈)→ 跳过
36
+ continue; // 匹配不到变体(旧 schema 不绑)→ 跳过
37
37
  const s = report.summary?.[variant];
38
38
  const composite = s?.avgCompositeScore;
39
39
  if (typeof composite !== 'number' || Number.isNaN(composite))
@@ -459,7 +459,6 @@ export function renderRunDetail(report, lang = DEFAULT_LANG, skillContext) {
459
459
  ${m.sampleHashes ? `<span class="meta-tag" title="${t('sampleHashCountDesc', lang)}">${t('sampleHashCount', lang)}: ${Object.keys(m.sampleHashes).length}/${m.sampleCount}</span>` : ''}
460
460
  ${m.evaluationFramework ? `<span class="meta-tag" title="${t('evalFrameworkDesc', lang)}">${t('evalFrameworkLabel', lang)}: ${m.evaluationFramework === 'bootstrap' ? t('evalFrameworkBootstrap', lang) : m.evaluationFramework === 'both' ? t('evalFrameworkBoth', lang) : t('evalFrameworkTTest', lang)}</span>` : ''}
461
461
  ${renderDebiasModeTag(m.debiasMode, lang)}
462
- ${m.blind ? `<span class="meta-tag" style="color:var(--green)" data-i18n="blindLabel">${t('blindLabel', lang)}</span>` : ''}
463
462
  </div>`;
464
463
  const auditFingerprints = (() => {
465
464
  const auditTags = [
@@ -470,13 +469,6 @@ export function renderRunDetail(report, lang = DEFAULT_LANG, skillContext) {
470
469
  return '';
471
470
  return `<details class="audit-fingerprints"><summary>${lang === 'zh' ? '审计指纹(用于复现校验)' : 'Audit fingerprints (for reproducibility)'}</summary><div class="meta-tags">${auditTags}</div></details>`;
472
471
  })();
473
- const blindReveal = m.blind ? `
474
- <div style="margin:12px 0">
475
- <button onclick="document.getElementById('blind-reveal').style.display=document.getElementById('blind-reveal').style.display==='none'?'block':'none'" data-i18n="revealBlind">${t('revealBlind', lang)}</button>
476
- <div id="blind-reveal" style="display:none;margin-top:8px;padding:12px;background:var(--bg-surface);border:1px solid var(--border);border-radius:var(--radius)" role="region" aria-label="Blind variant mapping">
477
- ${Object.entries(m.blindMap || {}).map(([label, real]) => `<div style="font-size:13px;color:var(--text-secondary)"><strong>Variant ${e(label)}</strong> → ${e(real)}</div>`).join('')}
478
- </div>
479
- </div>` : '';
480
472
  // ──────────── 融合单页:Hero → 结论(verdict+六维) → 折叠次要区 → 统一逐用例 ────────────
481
473
  const agentOverview = renderAgentOverview(variants, summary, lang);
482
474
  const varianceSection = renderVarianceComparisons(report.variance, lang, Boolean(report.meta.layeredStats), summary);
@@ -503,7 +495,6 @@ export function renderRunDetail(report, lang = DEFAULT_LANG, skillContext) {
503
495
  : 'Each sample is both a score card and a functional test: composite + layered scores, assertions, diagnostics, and trace combined';
504
496
  const body = `
505
497
  ${conclusionPanel}
506
- ${blindReveal}
507
498
  ${setupFold}
508
499
  ${analysisFold}
509
500
  <section class="ev-samples">
@@ -58,7 +58,7 @@ export const I18N = {
58
58
  score: '分数', cost: '执行成本', time: '时间',
59
59
  deleteBtnText: '删除', deleteConfirm: '确定删除报告', deleteFail: '删除失败',
60
60
  reportTitle: '评测报告', backToList: '← 返回列表',
61
- judge: '评委', executor: '执行器', blindLabel: '盲测', revealBlind: '显示变体对应关系',
61
+ judge: '评委', executor: '执行器',
62
62
  dimFact: '📋 事实', dimFactDesc: '输出的事实声明是否正确(规则可验证:关键词匹配、格式校验等断言)',
63
63
  dimBehavior: '🛠️ 行为', dimBehaviorDesc: '执行过程是否合规(规则可验证:工具调用路径、轮次限制、成本约束等断言)',
64
64
  dimJudge: '💬 LLM 评价', dimJudgeDesc: '请一个 LLM 当评委,读任务执行模型的输出内容,按预先写好的评分规则(英文 rubric)打 1-5 分。主观但能抓到规则断言判不了的"整体好不好"',
@@ -72,7 +72,7 @@ export const I18N = {
72
72
  judgeStddev: '评委波动', judgeStddevDesc: '同一份输出让评委评 N 次 (--judge-repeat) 得到 N 个分数的标准差。值低 = 评委对自己很坚定;值高 = 这个分本身就是噪声',
73
73
  judgeFailures: '评委失败', judgeFailuresDesc: 'N 次评委评分中返回 score=0(解析失败 / 调用错误)的次数。stddev=0 + failureCount>0 不是"完美一致",是"大部分炸了"',
74
74
  judgeReasoning: '评委推理', judgeReasoningExpand: '展开',
75
- ensembleHeader: '多评委评分对比', ensembleDesc: '不同评委模型对同一份输出的独立评分。用于反驳"同模态偏差"',
75
+ ensembleHeader: '多评委评分对比', ensembleDesc: '不同评委模型对同一份输出的独立评分。跨厂商评委时可反驳"同模态偏差";单厂商 ensemble 一致性高只是共有偏置,不构成反驳',
76
76
  agreementHeader: '跨用例评委一致性', agreementDesc: '在所有评测用例上算的多评委一致性',
77
77
  pearsonLabel: '皮尔逊系数 (Pearson)', pearsonDesc: '皮尔逊相关系数:1=完全同向排序,0=无关,-1=完全反向',
78
78
  madLabel: '平均绝对差 (MAD)', madDesc: '平均绝对差。1-5 制下 < 0.5 紧密一致, > 1.5 大分歧',
@@ -171,7 +171,7 @@ export const I18N = {
171
171
  score: 'Score', cost: 'Execution cost', time: 'Time',
172
172
  deleteBtnText: 'Delete', deleteConfirm: 'Delete report', deleteFail: 'Delete failed',
173
173
  reportTitle: 'Evaluation Report', backToList: '← Back to list',
174
- judge: 'Judge', executor: 'Executor', blindLabel: 'BLIND', revealBlind: 'Reveal variant mapping',
174
+ judge: 'Judge', executor: 'Executor',
175
175
  dimFact: '📋 Fact', dimFactDesc: 'Are factual claims correct (rule-verified: keyword matching, schema checks, etc.)',
176
176
  dimBehavior: '🛠️ Behavior', dimBehaviorDesc: 'Is execution compliant (rule-verified: tool paths, turn limits, cost constraints)',
177
177
  dimJudge: '💬 LLM judge', dimJudgeDesc: 'A separate LLM acts as judge: reads the task execution model output, scores 1-5 against a predefined rubric. Subjective, catches "overall feel" rule-based assertions miss',
@@ -185,7 +185,7 @@ export const I18N = {
185
185
  judgeStddev: 'Judge stddev', judgeStddevDesc: 'Stddev across N judge calls (--judge-repeat). Low = judge is consistent; high = this score itself is noisy',
186
186
  judgeFailures: 'Judge failures', judgeFailuresDesc: 'How many of N judge calls returned score=0 (parse / executor failure). stddev=0 + failureCount>0 is NOT "perfect agreement" — it means most calls failed',
187
187
  judgeReasoning: 'CoT reasoning', judgeReasoningExpand: 'expand',
188
- ensembleHeader: 'Per-judge scores', ensembleDesc: 'Independent scores from different judge models for the same output — refutes same-modality bias',
188
+ ensembleHeader: 'Per-judge scores', ensembleDesc: 'Independent scores from different judge models for the same output — refutes same-modality bias only when judges span vendors; a single-vendor ensemble agreement is shared bias, not a rebuttal',
189
189
  agreementHeader: 'Inter-judge agreement', agreementDesc: 'Cross-sample agreement metrics across all judges in this variant',
190
190
  pearsonLabel: 'Pearson', pearsonDesc: 'Pearson correlation: 1=perfect rank agreement, 0=independent, -1=anti-correlated',
191
191
  madLabel: 'MAD', madDesc: 'Mean absolute difference. On 1-5 scale: < 0.5 tight, > 1.5 large disagreement',
@@ -252,7 +252,7 @@ export function renderHumanAgreement(agreement, lang) {
252
252
  <td><strong>Krippendorff α</strong></td>
253
253
  <td style="text-align:center;color:${alphaColor}"><strong>${fmt(a.alpha)}</strong></td>
254
254
  <td style="text-align:center;font-size:11px">[${fmt(a.alphaCI.low)}, ${fmt(a.alphaCI.high)}]</td>
255
- <td style="font-size:12px;color:var(--text-secondary)">${lang === 'zh' ? '主指标,序数加权' : 'primary, ordinal-weighted'}</td>
255
+ <td style="font-size:12px;color:var(--text-secondary)">${lang === 'zh' ? '主指标,区间加权' : 'primary, interval-weighted'}</td>
256
256
  </tr>
257
257
  <tr>
258
258
  <td>${lang === 'zh' ? '加权 κ' : 'weighted κ'}</td>
@@ -1496,6 +1496,20 @@ function localizedInsightMessage(insight, report, lang) {
1496
1496
  ? `${count} 个评测用例成本显著高于平均值`
1497
1497
  : `${count} samples cost materially more than average`;
1498
1498
  }
1499
+ case 'judge_self_preference': {
1500
+ const outv = asArray(details.outputVendors).map(String).join('/') || '同厂商';
1501
+ const calibrated = details.goldCalibrated === true;
1502
+ return lang === 'zh'
1503
+ ? `评委与被测输出同厂商(${outv})${calibrated ? '(已 gold 校准)' : ''},存在自我偏好敞口:评委可能给同家族输出打高分`
1504
+ : `Judge is the same vendor as the executor that produced the outputs (${outv})${calibrated ? ' (gold-calibrated)' : ''}; self-preference exposure — the judge may over-score same-family output`;
1505
+ }
1506
+ case 'single_vendor_ensemble': {
1507
+ const n = asArray(details.judgeVendors).length;
1508
+ const calibrated = details.goldCalibrated === true;
1509
+ return lang === 'zh'
1510
+ ? `${n} 个评委同属一个厂商${calibrated ? '(已 gold 校准)' : ''},ensemble 一致性高只反映共有偏置,不构成对同模型偏置的反驳`
1511
+ : `All ${n} judges are from one vendor${calibrated ? ' (gold-calibrated)' : ''}; high ensemble agreement reflects shared bias, not a rebuttal to same-model bias`;
1512
+ }
1499
1513
  default:
1500
1514
  return lang === 'zh'
1501
1515
  ? `结构化诊断:${type}`
@@ -1544,6 +1558,14 @@ function localizedSuggestion(insight, lang) {
1544
1558
  return lang === 'zh'
1545
1559
  ? '重写 agent 断言时优先约束工具路径、关键文件读取和 turns 上限'
1546
1560
  : 'When rewriting agent assertions, prioritize tool path, key file reads, and turn-limit constraints';
1561
+ case 'judge_self_preference':
1562
+ return lang === 'zh'
1563
+ ? '换跨厂商评委(如 --judge-models openai-api:gpt-4o)消除自我偏好,或挂人工金标校准(omk eval gold compare)量化评委是否可信。注:固定模型的 A/B 差值受影响较小,绝对分 / 跨版本曲线受影响更大'
1564
+ : 'Use a cross-vendor judge (e.g. --judge-models openai-api:gpt-4o) to remove self-preference, or calibrate against human gold (omk eval gold compare). Note: the A/B delta is less affected than absolute scores / cross-version curves';
1565
+ case 'single_vendor_ensemble':
1566
+ return lang === 'zh'
1567
+ ? '把 ensemble 配成跨厂商(如 --judge-models claude:opus,openai-api:gpt-4o),让 agreement 真能反驳同模型偏置'
1568
+ : 'Make the ensemble cross-vendor (e.g. --judge-models claude:opus,openai-api:gpt-4o) so agreement actually rebuts same-model bias';
1547
1569
  default:
1548
1570
  return lang === 'zh'
1549
1571
  ? '查看结构化 details 字段定位原因'
@@ -0,0 +1,4 @@
1
+ export declare const LENGTH_DEBIAS_INSTRUCTION: string;
2
+ export declare const PRESENTATION_NEUTRALITY_INSTRUCTION: string;
3
+ export declare const RAG_LENGTH_DEBIAS: string;
4
+ export declare const RAG_PRESENTATION_NEUTRALITY: string;
@@ -0,0 +1,44 @@
1
+ // 评委 prompt 的共享去偏 / 中性化指令 —— 单一来源。
2
+ //
3
+ // 此前这几段在 grading/judge.ts(rubric 评委)与 grading/assertions.ts(RAG 评委)各写一份,
4
+ // 改一处漏一处(J3 即如此)。统一收口于此,两侧 import。
5
+ //
6
+ // 两套变体是有意的、不要强并成一条:
7
+ // - 全版(LENGTH_DEBIAS_INSTRUCTION / PRESENTATION_NEUTRALITY_INSTRUCTION):给 rubric 评委,
8
+ // 带研究依据句;rubric prompt 较短,容得下。被 getJudgePromptHash 冻结,改字节 = BREAKING-COMPARABILITY。
9
+ // - 短版(RAG_LENGTH_DEBIAS / RAG_PRESENTATION_NEUTRALITY):给 RAG 评委,RAG prompt 本就长,精简表述。
10
+ // 改任一段先想另一段是否同步;改全版还要走冻结 hash 的 bump 流程。
11
+ export const LENGTH_DEBIAS_INSTRUCTION = [
12
+ '## 重要:长度不是质量信号',
13
+ '评分时聚焦内容实质与正确性。回答的篇幅、行文丰富度、结构复杂度本身不是质量指标 ——',
14
+ '简洁正确的回答不应因短而扣分;冗长但偏题或重复的回答不应因长而加分。',
15
+ '研究显示 LLM 评委容易隐性偏向更长的回答,请在打分前先警觉这一点。',
16
+ ].join('\n');
17
+ // 始终开启(不给开关):排版与语气都不是质量信号。措辞严格对称 —— 既不奖励精致 / 自信,也不
18
+ // 因朴素 / 含糊而扣分,避免"抗偏置"本身过度矫正成反向偏置。研究表明 LLM 评委除长度外,还隐性
19
+ // 偏向排版精致(format / markdown bias)与语气自信、自我表扬的回答(谄媚 / 权威偏置)。
20
+ export const PRESENTATION_NEUTRALITY_INSTRUCTION = [
21
+ '## 重要:排版与语气不是质量信号',
22
+ '评分只看内容是否对照评分标准、是否正确。Markdown 排版、标题、列表、加粗、表格等呈现形式',
23
+ '本身不是质量指标 —— 朴素但正确的回答不应因没有排版而扣分;排版精致但偏题或错误的回答不应',
24
+ '因好看而加分。',
25
+ '同样,回答的语气、自信程度、是否自我表扬(如「这是最优方案」)也不是质量信号 —— 不要被笃定',
26
+ '的口吻或自我评价带跑:自信但错误的回答不应高于含糊但正确的回答,一切结论都要回到评分标准',
27
+ '逐条核实。',
28
+ '研究显示 LLM 评委容易隐性偏向排版精致、语气自信的回答,请在打分前先警觉这两点。',
29
+ ].join('\n');
30
+ // RAG 指标(faithfulness / answer_relevancy / context_recall)的去偏**恒开**,
31
+ // 按 default-strict 不提供关闭开关:`--no-debias-length` 只作用于 rubric 评委 prompt 的长度去偏,
32
+ // 不下探到 RAG —— RAG 评的是内容保真 / 切题 / 召回,放任长度 / 排版 / 语气偏置只会污染这些指标,
33
+ // 没有「关掉它」的正当用例。故 runRagJudge 不收 lengthDebias 参数,两段恒注入。
34
+ export const RAG_LENGTH_DEBIAS = [
35
+ '## 重要:长度不是质量信号',
36
+ '评分时聚焦内容实质,不要因输出更长就给更高分。',
37
+ '简洁正确的回答与冗长正确的回答应得相同分数。',
38
+ ].join('\n');
39
+ // 排版 / 语气中性化(format / sycophancy bias),与 RAG_LENGTH_DEBIAS 同 footprint 恒开。
40
+ export const RAG_PRESENTATION_NEUTRALITY = [
41
+ '## 重要:排版与语气不是质量信号',
42
+ '评分只看内容是否符合上述标准。排版是否精致、语气是否自信、有无自我表扬都不是质量信号 ——',
43
+ '不要被笃定的口吻带跑:自信但错误的回答不应高于含糊但正确的回答。',
44
+ ].join('\n');
@@ -0,0 +1,30 @@
1
+ export declare const JUDGE_SYSTEM_PROMPT = "\u4F60\u662F\u4E00\u4E2A\u4E25\u683C\u7684 AI \u8F93\u51FA\u8D28\u91CF\u8BC4\u5BA1\u5458\u3002\u5148\u9010\u6761\u5BF9\u7167\u8BC4\u5206\u6807\u51C6\u505A\u63A8\u7406\uFF0C\u518D\u7ED9\u6700\u7EC8\u5206\u6570\u3002\u53EA\u8FD4\u56DE JSON\uFF0C\u4E0D\u8981\u5176\u4ED6\u5185\u5BB9\u3002";
2
+ export declare function buildJudgePrompt(prompt: string, rubric: string, output: string, traceSummary: string | null, lengthDebias?: boolean): string;
3
+ /**
4
+ * Stable hash of the judge prompt template. Saved into ReportMeta.judgePromptHash so
5
+ * downstream readers can detect "the judge prompt changed between these two reports".
6
+ *
7
+ * `lengthDebias` defaults to true. Pass false (via `--no-debias-length`) to drop the
8
+ * length-debias instruction; that produces the debias-off prompt variant, whose hash
9
+ * differs from the default so the two are never compared blind.
10
+ */
11
+ export declare function getJudgePromptHash(lengthDebias?: boolean): string;
12
+ export declare const SEMANTIC_SIMILARITY_SYSTEM = "\u4F60\u662F\u8BED\u4E49\u76F8\u4F3C\u5EA6\u8BC4\u5BA1\u5458\u3002\u53EA\u8FD4\u56DE JSON\uFF0C\u4E0D\u8981\u5176\u4ED6\u5185\u5BB9\u3002";
13
+ export declare function buildSemanticSimilarityPrompt(reference: string, output: string): string;
14
+ export type RagJudgeType = 'faithfulness' | 'answer_relevancy' | 'context_recall';
15
+ export interface RagJudgeFields {
16
+ output: string;
17
+ /** faithfulness: 参考 context。 */
18
+ context?: string;
19
+ /** answer_relevancy: 用户问题(sample.prompt)。 */
20
+ question?: string;
21
+ /** context_recall: 参考 gold。 */
22
+ reference?: string;
23
+ }
24
+ /** RAG 评委 system + 用户 prompt。缺字段的早退 / 解析逻辑留在 assertions.ts。 */
25
+ export declare function buildRagJudgePrompt(type: RagJudgeType, fields: RagJudgeFields): {
26
+ system: string;
27
+ prompt: string;
28
+ };
29
+ export declare function getSemanticPromptHash(): string;
30
+ export declare function getRagJudgePromptHash(type: RagJudgeType): string;
@@ -0,0 +1,205 @@
1
+ import { createHash } from 'node:crypto';
2
+ import { LENGTH_DEBIAS_INSTRUCTION, PRESENTATION_NEUTRALITY_INSTRUCTION, RAG_LENGTH_DEBIAS, RAG_PRESENTATION_NEUTRALITY, } from './debias-instructions.js';
3
+ // 评分类 prompt 单一来源 —— 直接决定分数的 LLM 评委 prompt(rubric 主评委 + RAG + 语义相似度)。
4
+ // 这些是测量学不变量:文本字节决定可比性,改动须配合 prompt-registry 的冻结 hash bump
5
+ // (BREAKING-COMPARABILITY)。执行器调用 / JSON 解析逻辑留在 grading/judge.ts 与 grading/assertions.ts。
6
+ // ===========================================================================
7
+ // Rubric 主评委 prompt
8
+ // ===========================================================================
9
+ /**
10
+ * Judge prompt template version.
11
+ *
12
+ * The version string encodes WHICH debias / context features the judge sees, so reports
13
+ * tagged with the same hash are score-comparable; a mismatched hash means "we changed how
14
+ * we ask the judge to think" and the reports should not be compared blind. Bump it (and the
15
+ * frozen hashes in `test/shared/prompt-registry-freeze.test.ts`) whenever the template's bytes
16
+ * change — that change is BREAKING-COMPARABILITY.
17
+ *
18
+ * Naming: a single main version (`v5`) + a `-feature` suffix per debias/context capability.
19
+ * The only asymmetry between the two strings is the `-len` suffix, gated by the
20
+ * `--no-debias-length` toggle; every other feature is always-on and appears in both.
21
+ */
22
+ // 版本演进(内嵌"评委看到什么 / 被要求忽略什么"的语义):
23
+ // v2-cot : 仅看 output + rubric + 工具名分布(早期 buildTraceSummary)
24
+ // v3-cot-length : 加 length-debias 指令
25
+ // v3-cot-toolargs(off) / v4-cot-len-args(on)
26
+ // : 加 tool input 预览(让评委识别 wrapper-style skill 在 Bash 命令里的
27
+ // 真实语义调用,如 `Bash: mcporter --tool X`,否则误判"只调了 Bash")
28
+ // v5-cot-toolargs-fmt(off) / v5-cot-toolargs-fmt-len(on)
29
+ // : 加排版 / 语气中性化指令(始终开启,不给开关)。研究表明 LLM 评委除了
30
+ // 偏向更长的回答,还隐性偏向排版精致(标题 / 列表 / 加粗)与语气自信 / 自我
31
+ // 表扬(谄媚 / 权威偏置)的回答;显式指令要求评委只对照评分标准核内容。
32
+ // 同时借此把命名统一成单一主序号(v5)+ feature 后缀,`-len` 仍是开关那条。
33
+ const JUDGE_PROMPT_VERSION_DEBIAS_OFF = 'v5-cot-toolargs-fmt';
34
+ const JUDGE_PROMPT_VERSION_DEBIAS_ON = 'v5-cot-toolargs-fmt-len';
35
+ export const JUDGE_SYSTEM_PROMPT = '你是一个严格的 AI 输出质量评审员。先逐条对照评分标准做推理,再给最终分数。只返回 JSON,不要其他内容。';
36
+ export function buildJudgePrompt(prompt, rubric, output, traceSummary, lengthDebias = true) {
37
+ const version = lengthDebias ? JUDGE_PROMPT_VERSION_DEBIAS_ON : JUDGE_PROMPT_VERSION_DEBIAS_OFF;
38
+ const traceSection = traceSummary
39
+ ? ['', '## Agent 执行过程', traceSummary, '', '请同时考虑执行过程的合理性(工具选择、步骤效率、错误恢复)。']
40
+ : [];
41
+ // 排版 / 语气中性化始终开启(不受 --no-debias-length 影响);length-debias 仍受开关控。
42
+ const neutralitySection = ['', PRESENTATION_NEUTRALITY_INSTRUCTION];
43
+ const debiasSection = lengthDebias ? ['', LENGTH_DEBIAS_INSTRUCTION] : [];
44
+ return [
45
+ `请对以下 AI 输出进行质量评分(template ${version})。`,
46
+ '',
47
+ '## 原始任务',
48
+ prompt,
49
+ '',
50
+ '## 评分标准',
51
+ rubric,
52
+ '',
53
+ '## AI 输出',
54
+ output,
55
+ ...traceSection,
56
+ ...neutralitySection,
57
+ ...debiasSection,
58
+ '',
59
+ '## 评分流程',
60
+ '1. 逐条对照评分标准,先做推理(reasoning):列出 AI 输出哪些点对应哪条标准,哪些缺失,哪些有歧义。',
61
+ '2. 基于推理给出最终分数(1-5 的整数)和简短理由。',
62
+ '',
63
+ '请返回 JSON(不要包含 markdown 代码块标记):',
64
+ '{"reasoning": "<对照标准的逐条推理>", "score": <1-5的整数>, "reason": "<最终结论的简短理由>"}',
65
+ '',
66
+ '评分标准:1=完全不达标, 2=部分涉及, 3=基本达标, 4=较好, 5=优秀',
67
+ ].join('\n');
68
+ }
69
+ /**
70
+ * Stable hash of the judge prompt template. Saved into ReportMeta.judgePromptHash so
71
+ * downstream readers can detect "the judge prompt changed between these two reports".
72
+ *
73
+ * `lengthDebias` defaults to true. Pass false (via `--no-debias-length`) to drop the
74
+ * length-debias instruction; that produces the debias-off prompt variant, whose hash
75
+ * differs from the default so the two are never compared blind.
76
+ */
77
+ export function getJudgePromptHash(lengthDebias = true) {
78
+ const version = lengthDebias ? JUDGE_PROMPT_VERSION_DEBIAS_ON : JUDGE_PROMPT_VERSION_DEBIAS_OFF;
79
+ // Hash the template-shaping function source + the version tag together. We hash a
80
+ // deterministic stringified form of the template (with placeholder inputs) so any
81
+ // structural edit shows up.
82
+ const sample = buildJudgePrompt('<P>', '<R>', '<O>', '<T>', lengthDebias);
83
+ return createHash('sha256').update(version + '\n' + sample).digest('hex').slice(0, 12);
84
+ }
85
+ // ===========================================================================
86
+ // 语义相似度评委 prompt(semantic_similarity 断言)
87
+ // ===========================================================================
88
+ export const SEMANTIC_SIMILARITY_SYSTEM = '你是语义相似度评审员。只返回 JSON,不要其他内容。';
89
+ export function buildSemanticSimilarityPrompt(reference, output) {
90
+ return [
91
+ '请判断以下两段文本的语义相似度。',
92
+ '',
93
+ '## 参考文本',
94
+ reference,
95
+ '',
96
+ '## 待评估文本',
97
+ output,
98
+ '',
99
+ '请返回 JSON(不要包含 markdown 代码块标记):',
100
+ '{"score": <1-5的整数>, "reason": "<简短理由>"}',
101
+ '',
102
+ '评分:1=完全无关, 2=略有关联, 3=部分相似, 4=大致相同, 5=高度一致',
103
+ ].join('\n');
104
+ }
105
+ /** RAG 评委 system + 用户 prompt。缺字段的早退 / 解析逻辑留在 assertions.ts。 */
106
+ export function buildRagJudgePrompt(type, fields) {
107
+ if (type === 'faithfulness') {
108
+ return {
109
+ system: '你是 RAG 评审员,专注判断输出是否被参考 context 支持。只返回 JSON。',
110
+ prompt: [
111
+ '请判断"待评估输出"中的事实性陈述是否被"参考 context"支持。',
112
+ '',
113
+ '## 参考 context',
114
+ fields.context ?? '',
115
+ '',
116
+ '## 待评估输出',
117
+ fields.output,
118
+ '',
119
+ RAG_LENGTH_DEBIAS,
120
+ RAG_PRESENTATION_NEUTRALITY,
121
+ '',
122
+ '## 评分流程',
123
+ '1. 列出待评估输出中所有事实性陈述',
124
+ '2. 逐条判断是否能在 context 中找到支持',
125
+ '3. 给出 1-5 分:',
126
+ ' 5 = 全部陈述都有 context 支持,无编造',
127
+ ' 4 = 多数有支持,有 1-2 处不重要的编造',
128
+ ' 3 = 一半有支持',
129
+ ' 2 = 多数无支持',
130
+ ' 1 = 完全编造或与 context 矛盾',
131
+ '',
132
+ '请返回 JSON(不要 markdown 代码块):',
133
+ '{"score": <1-5的整数>, "reason": "<简短理由>"}',
134
+ ].join('\n'),
135
+ };
136
+ }
137
+ if (type === 'answer_relevancy') {
138
+ return {
139
+ system: '你是答题切题度评审员。只返回 JSON。',
140
+ prompt: [
141
+ '请判断"AI 输出"是否直接、切题地回答了"用户问题"。',
142
+ '',
143
+ '## 用户问题',
144
+ fields.question ?? '',
145
+ '',
146
+ '## AI 输出',
147
+ fields.output,
148
+ '',
149
+ RAG_LENGTH_DEBIAS,
150
+ RAG_PRESENTATION_NEUTRALITY,
151
+ '',
152
+ '## 评分',
153
+ '5 = 完整切题回答,无冗余无遗漏',
154
+ '4 = 切题但有少量冗余或小遗漏',
155
+ '3 = 部分切题,部分跑题或避而不答',
156
+ '2 = 大部分跑题',
157
+ '1 = 完全跑题或拒答',
158
+ '',
159
+ '请返回 JSON(不要 markdown 代码块):',
160
+ '{"score": <1-5的整数>, "reason": "<简短理由>"}',
161
+ ].join('\n'),
162
+ };
163
+ }
164
+ // context_recall
165
+ return {
166
+ system: '你是 context 覆盖率评审员。只返回 JSON。',
167
+ prompt: [
168
+ '请判断"参考 gold"中的关键事实在"AI 输出"中被覆盖的程度。',
169
+ '',
170
+ '## 参考 gold',
171
+ fields.reference ?? '',
172
+ '',
173
+ '## AI 输出',
174
+ fields.output,
175
+ '',
176
+ RAG_LENGTH_DEBIAS,
177
+ RAG_PRESENTATION_NEUTRALITY,
178
+ '',
179
+ '## 评分流程',
180
+ '1. 列出参考中的关键事实(忽略修饰性内容)',
181
+ '2. 检查每条是否在输出中被提及/使用',
182
+ '3. 给出 1-5 分:',
183
+ ' 5 = 全部关键事实被覆盖',
184
+ ' 4 = 大部分覆盖,缺 1-2 条次要事实',
185
+ ' 3 = 一半覆盖',
186
+ ' 2 = 仅覆盖少量',
187
+ ' 1 = 完全未覆盖',
188
+ '',
189
+ '请返回 JSON(不要 markdown 代码块):',
190
+ '{"score": <1-5的整数>, "reason": "<简短理由>"}',
191
+ ].join('\n'),
192
+ };
193
+ }
194
+ /** 评分类断言 prompt 的 drift-detection hash(占位符输入,12 位,仅供 prompt-registry 冻结)。 */
195
+ function hashPromptSample(label, system, prompt) {
196
+ return createHash('sha256').update(`${label}\n${system}\n${prompt}`).digest('hex').slice(0, 12);
197
+ }
198
+ export function getSemanticPromptHash() {
199
+ return hashPromptSample('semantic_similarity', SEMANTIC_SIMILARITY_SYSTEM, buildSemanticSimilarityPrompt('<R>', '<O>'));
200
+ }
201
+ export function getRagJudgePromptHash(type) {
202
+ const fields = { output: '<O>', context: '<C>', question: '<Q>', reference: '<R>' };
203
+ const { system, prompt } = buildRagJudgePrompt(type, fields);
204
+ return hashPromptSample(`rag:${type}`, system, prompt);
205
+ }
@@ -0,0 +1,27 @@
1
+ /**
2
+ * omk 所有 LLM prompt 的单一编目 —— 发现入口 + 测量学冻结入口。
3
+ *
4
+ * 解决「prompt 散在多个目录、难管理」与「只有 2 个 prompt 被冻结」两个问题:
5
+ * - 每条登记 prompt 的位置(module)+ 用途,一处可查全。
6
+ * - measurementInvariant 为 true 的(直接决定分数的评委 prompt)带 getHash,
7
+ * 由 test/shared/prompt-registry-freeze.test.ts 统一冻结,文本漂移即 CI 红。
8
+ *
9
+ * 注意:hash 形状按各自历史口径,不强行统一 ——
10
+ * - rubric 评委:12 位(getJudgePromptHash,持久化进 report.meta.judgePromptHash 的同口径);
11
+ * - RAG / semantic:12 位(getRagJudgePromptHash / getSemanticPromptHash,仅供冻结 drift 检测);
12
+ * - observe 复盘:64 位(readPromptDocument 读 .md 全文 sha256)。
13
+ * 用限定名 `promptId`(非裸 `kind`,见 terminology-spec §5.4)。
14
+ */
15
+ export interface PromptRegistryEntry {
16
+ /** 稳定标识(冻结表的 key)。 */
17
+ promptId: string;
18
+ /** 人类可读用途。 */
19
+ purpose: string;
20
+ /** 定义位置(发现用,相对仓库根的源文件路径)。 */
21
+ module: string;
22
+ /** 是否直接决定分数 → 须冻结防漂移。 */
23
+ measurementInvariant: boolean;
24
+ /** 仅 measurementInvariant 条目提供:当前 prompt 的 hash。 */
25
+ getHash?: () => string;
26
+ }
27
+ export declare const PROMPT_REGISTRY: PromptRegistryEntry[];