oh-my-knowledge 0.40.0 → 0.42.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (71) hide show
  1. package/README.md +8 -4
  2. package/README.zh.md +8 -4
  3. package/dist/analysis/report-diagnostics.d.ts +8 -1
  4. package/dist/analysis/report-diagnostics.js +109 -1
  5. package/dist/assets/agent-skills/omk/references/commands.md +2 -2
  6. package/dist/authoring/evolver.d.ts +3 -14
  7. package/dist/authoring/evolver.js +1 -52
  8. package/dist/authoring/generator.d.ts +24 -0
  9. package/dist/authoring/generator.js +36 -8
  10. package/dist/cli/commands/eval/index.d.ts +1 -1
  11. package/dist/cli/commands/eval/index.js +50 -15
  12. package/dist/cli/commands/init.js +10 -7
  13. package/dist/cli/lib/cmd-flags.d.ts +0 -1
  14. package/dist/cli/lib/i18n-dict/init.d.ts +1 -1
  15. package/dist/cli/lib/i18n-dict/init.js +14 -11
  16. package/dist/cli/lib/i18n-dict/run.d.ts +1 -1
  17. package/dist/cli/lib/i18n-dict/run.js +6 -2
  18. package/dist/cli/lib/parse-run-config.d.ts +3 -1
  19. package/dist/cli/lib/parse-run-config.js +0 -2
  20. package/dist/eval-core/evaluation-job.d.ts +2 -2
  21. package/dist/eval-core/evaluation-job.js +2 -2
  22. package/dist/eval-core/evaluation-reporting.d.ts +0 -1
  23. package/dist/eval-core/evaluation-reporting.js +9 -41
  24. package/dist/eval-core/execution-strategy.js +3 -2
  25. package/dist/eval-core/holdout.d.ts +66 -0
  26. package/dist/eval-core/holdout.js +118 -0
  27. package/dist/eval-core/judge-independence.d.ts +28 -0
  28. package/dist/eval-core/judge-independence.js +29 -0
  29. package/dist/eval-core/verdict.d.ts +53 -2
  30. package/dist/eval-core/verdict.js +216 -16
  31. package/dist/eval-workflows/batch-evaluation-workflow.d.ts +1 -1
  32. package/dist/eval-workflows/batch-evaluation-workflow.js +1 -2
  33. package/dist/eval-workflows/evaluation-pipeline/report-finalize.d.ts +1 -5
  34. package/dist/eval-workflows/evaluation-pipeline/report-finalize.js +19 -8
  35. package/dist/eval-workflows/evaluation-pipeline/run-state.d.ts +2 -2
  36. package/dist/eval-workflows/evaluation-pipeline/run-state.js +2 -2
  37. package/dist/eval-workflows/evaluation-pipeline.d.ts +3 -2
  38. package/dist/eval-workflows/evaluation-pipeline.js +3 -4
  39. package/dist/eval-workflows/run-evaluation.d.ts +6 -4
  40. package/dist/eval-workflows/run-evaluation.js +8 -6
  41. package/dist/executors/claude-cli.js +5 -6
  42. package/dist/executors/claude-sdk.d.ts +5 -2
  43. package/dist/executors/claude-sdk.js +13 -8
  44. package/dist/executors/codex-cli.js +3 -4
  45. package/dist/executors/shared.d.ts +2 -0
  46. package/dist/executors/shared.js +15 -0
  47. package/dist/grading/assertions.js +6 -122
  48. package/dist/grading/gold-cli.js +1 -1
  49. package/dist/grading/human-gold.d.ts +5 -3
  50. package/dist/grading/human-gold.js +5 -3
  51. package/dist/grading/index.d.ts +4 -4
  52. package/dist/grading/judge.d.ts +6 -14
  53. package/dist/grading/judge.js +5 -88
  54. package/dist/inputs/eval-config.js +12 -2
  55. package/dist/managed/evidence.js +1 -2
  56. package/dist/managed/version-scores.js +1 -1
  57. package/dist/renderer/html-renderer.js +0 -9
  58. package/dist/renderer/layout.js +4 -4
  59. package/dist/renderer/summary.js +59 -4
  60. package/dist/shared/llm-prompts/debias-instructions.d.ts +4 -0
  61. package/dist/shared/llm-prompts/debias-instructions.js +44 -0
  62. package/dist/shared/llm-prompts/judge-prompts.d.ts +30 -0
  63. package/dist/shared/llm-prompts/judge-prompts.js +205 -0
  64. package/dist/shared/llm-prompts/registry.d.ts +27 -0
  65. package/dist/shared/llm-prompts/registry.js +69 -0
  66. package/dist/types/eval.d.ts +15 -8
  67. package/dist/types/judge.d.ts +1 -1
  68. package/dist/types/report.d.ts +50 -3
  69. package/package.json +1 -1
  70. package/dist/grading/debias-validate.d.ts +0 -83
  71. package/dist/grading/debias-validate.js +0 -176
@@ -45,10 +45,10 @@ interface GradeOptions {
45
45
  */
46
46
  judgeRepeat?: number;
47
47
  /**
48
- * v0.21 length-debias toggle. Defaults to true — judge prompt includes the
49
- * "length is not a quality signal" instruction, prompt template version is
50
- * v3-cot-length. Set false (via `--no-debias-length`) to revert to the
51
- * legacy v2-cot prompt for reproducing historical reports.
48
+ * length-debias toggle. Defaults to true — judge prompt includes the
49
+ * "length is not a quality signal" instruction. Set false (via `--no-debias-length`)
50
+ * to drop it (the debias-off prompt variant) for reproducing older no-length-debias
51
+ * reports. The presentation/tone neutrality instruction is always on regardless.
52
52
  */
53
53
  lengthDebias?: boolean;
54
54
  }
@@ -1,13 +1,5 @@
1
1
  import type { DimensionResult, ExecutorFn, JudgeAgreement, JudgeConfig, ToolCallInfo, TurnInfo } from '../types/index.js';
2
- export declare function buildJudgePrompt(prompt: string, rubric: string, output: string, traceSummary: string | null, lengthDebias?: boolean): string;
3
- /**
4
- * Stable hash of the judge prompt template. Saved into ReportMeta.judgePromptHash so
5
- * downstream readers can detect "the judge prompt changed between these two reports".
6
- *
7
- * `lengthDebias` defaults to true (v0.21+ default). Pass false when running under
8
- * `--no-debias-length` so the hash matches historical v2-cot reports.
9
- */
10
- export declare function getJudgePromptHash(lengthDebias?: boolean): string;
2
+ export { buildJudgePrompt, getJudgePromptHash } from '../shared/llm-prompts/judge-prompts.js';
11
3
  interface LlmJudgeOptions {
12
4
  output: string;
13
5
  rubric: string;
@@ -17,9 +9,10 @@ interface LlmJudgeOptions {
17
9
  traceSummary?: string | null;
18
10
  /**
19
11
  * When true (default), the judge prompt includes an explicit
20
- * "length is not a quality signal" instruction. Pass false to fall back to
21
- * the legacy v2-cot prompt — only useful for reproducing pre-v0.21 reports
22
- * or running A/B comparisons with alternate length-debias settings.
12
+ * "length is not a quality signal" instruction. Pass false to drop it (the
13
+ * debias-off prompt variant) — only useful for reproducing older no-length-debias
14
+ * reports or running A/B comparisons with alternate length-debias settings.
15
+ * (The presentation/tone neutrality instruction is always on, independent of this.)
23
16
  */
24
17
  lengthDebias?: boolean;
25
18
  }
@@ -60,7 +53,7 @@ export declare function judgeId(config: JudgeConfig): string;
60
53
  export declare function computeJudgeAgreement(judgeScores: number[][]): JudgeAgreement;
61
54
  /**
62
55
  * Judge a single (output, rubric) pair with N judge models in parallel. Each judge
63
- * may use a different executor (e.g. claude:opus + openai:gpt-4o + gemini:pro). Each
56
+ * may use a different executor (e.g. claude:opus + openai-api:gpt-4o + gemini:pro). Each
64
57
  * judge can also be repeated `judgeRepeat` times — final per-judge score is its mean.
65
58
  *
66
59
  * Returns: aggregate DimensionResult (score = mean across judges; this is the "consensus"
@@ -72,4 +65,3 @@ export declare function computeJudgeAgreement(judgeScores: number[][]): JudgeAgr
72
65
  * provide the raw ensemble so downstream can recompute.
73
66
  */
74
67
  export declare function llmJudgeEnsemble(options: LlmJudgeOptions, judges: JudgeConfig[], executorByName: (name: string) => ExecutorFn, judgeRepeat?: number): Promise<DimensionResult>;
75
- export {};
@@ -1,4 +1,7 @@
1
- import { createHash } from 'node:crypto';
1
+ import { buildJudgePrompt, JUDGE_SYSTEM_PROMPT } from '../shared/llm-prompts/judge-prompts.js';
2
+ // 评分类 prompt 已收口到 shared/llm-prompts/judge-prompts.ts(单一来源 + prompt-registry 冻结)。
3
+ // 这里 re-export 保对外 API 不破:既有消费方仍从 grading/judge.js import 这两个符号。
4
+ export { buildJudgePrompt, getJudgePromptHash } from '../shared/llm-prompts/judge-prompts.js';
2
5
  function extractFirstJsonObject(text) {
3
6
  const start = text.indexOf('{');
4
7
  if (start === -1)
@@ -46,92 +49,6 @@ function salvageJudgeResponse(text) {
46
49
  reasoning: text.trim().slice(0, 2000),
47
50
  };
48
51
  }
49
- /**
50
- * Judge prompt template version.
51
- *
52
- * - 'v2-cot' — legacy; CoT scoring without explicit length-debias instruction.
53
- * Kept for `--no-debias-length` so users can reproduce historical
54
- * reports byte-for-byte.
55
- * - 'v3-cot-length' — adds a paragraph telling the judge that length is not a quality
56
- * signal. Default on (research consistently shows
57
- * LLM judges over-weight verbosity; explicit instruction mitigates).
58
- *
59
- * Bump when the prompt's intent or structure changes meaningfully — reports tagged
60
- * with the same hash are score-comparable; mismatched hashes mean "we changed how we
61
- * ask the judge to think" and should not be compared blind.
62
- */
63
- // 版本字符串内嵌"判官能看到什么"的语义。bump 时机:
64
- // v2-cot : 仅看 output + rubric + 工具名分布(早期 buildTraceSummary)
65
- // v3-cot-length : 加了 length-debias 指令(v0.21)
66
- // v3-cot-toolargs : 加了 tool input 预览(本次,BREAKING-COMPARABILITY)
67
- // v4-cot-len-args : v3-cot-length 的同步升级(BREAKING-COMPARABILITY)
68
- // bump 原因:之前 trace 只给 tool 名 + 分布,wrapper-style skill(mcporter / code-host CLI 等)
69
- // 被判官当成"只调了 Bash,没用 skill 指定的 MCP 工具"——结论事实错误。
70
- // tool input 预览让判官能识别 `Bash: mcporter --tool skylark_xxx` 内的真实语义调用。
71
- //
72
- // ⚠️ 命名约定(下次 bump 时执行):
73
- // 当前 debias OFF: v3-* / debias ON: v4-* 编号不对称(历史上 debias OFF 跑 v2→v3,
74
- // debias ON 跑 v3→v4)。这次再 bump 会让两条线持续偏移 v4/v5、v5/v6、…
75
- //
76
- // 下次 bump 时统一为单一主序号 + features 后缀,如:
77
- // JUDGE_PROMPT_VERSION_DEBIAS_OFF = 'v5-cot-<features>'
78
- // JUDGE_PROMPT_VERSION_DEBIAS_ON = 'v5-cot-<features>-len'
79
- // 这样 hash 测试自然分两条,主序号清晰对齐,features 字段独立描述差异。
80
- // 本 PR 不在意命名,只是把约定写下来,避免下次又跟着错误偏移走。
81
- const JUDGE_PROMPT_VERSION_DEBIAS_OFF = 'v3-cot-toolargs';
82
- const JUDGE_PROMPT_VERSION_DEBIAS_ON = 'v4-cot-len-args';
83
- const JUDGE_SYSTEM_PROMPT = '你是一个严格的 AI 输出质量评审员。先逐条对照评分标准做推理,再给最终分数。只返回 JSON,不要其他内容。';
84
- const LENGTH_DEBIAS_INSTRUCTION = [
85
- '## 重要:长度不是质量信号',
86
- '评分时聚焦内容实质与正确性。回答的篇幅、行文丰富度、结构复杂度本身不是质量指标 ——',
87
- '简洁正确的回答不应因短而扣分;冗长但偏题或重复的回答不应因长而加分。',
88
- '研究显示 LLM 评委容易隐性偏向更长的回答,请在打分前先警觉这一点。',
89
- ].join('\n');
90
- export function buildJudgePrompt(prompt, rubric, output, traceSummary, lengthDebias = true) {
91
- const version = lengthDebias ? JUDGE_PROMPT_VERSION_DEBIAS_ON : JUDGE_PROMPT_VERSION_DEBIAS_OFF;
92
- const traceSection = traceSummary
93
- ? ['', '## Agent 执行过程', traceSummary, '', '请同时考虑执行过程的合理性(工具选择、步骤效率、错误恢复)。']
94
- : [];
95
- const debiasSection = lengthDebias ? ['', LENGTH_DEBIAS_INSTRUCTION] : [];
96
- return [
97
- `请对以下 AI 输出进行质量评分(template ${version})。`,
98
- '',
99
- '## 原始任务',
100
- prompt,
101
- '',
102
- '## 评分标准',
103
- rubric,
104
- '',
105
- '## AI 输出',
106
- output,
107
- ...traceSection,
108
- ...debiasSection,
109
- '',
110
- '## 评分流程',
111
- '1. 逐条对照评分标准,先做推理(reasoning):列出 AI 输出哪些点对应哪条标准,哪些缺失,哪些有歧义。',
112
- '2. 基于推理给出最终分数(1-5 的整数)和简短理由。',
113
- '',
114
- '请返回 JSON(不要包含 markdown 代码块标记):',
115
- '{"reasoning": "<对照标准的逐条推理>", "score": <1-5的整数>, "reason": "<最终结论的简短理由>"}',
116
- '',
117
- '评分标准:1=完全不达标, 2=部分涉及, 3=基本达标, 4=较好, 5=优秀',
118
- ].join('\n');
119
- }
120
- /**
121
- * Stable hash of the judge prompt template. Saved into ReportMeta.judgePromptHash so
122
- * downstream readers can detect "the judge prompt changed between these two reports".
123
- *
124
- * `lengthDebias` defaults to true (v0.21+ default). Pass false when running under
125
- * `--no-debias-length` so the hash matches historical v2-cot reports.
126
- */
127
- export function getJudgePromptHash(lengthDebias = true) {
128
- const version = lengthDebias ? JUDGE_PROMPT_VERSION_DEBIAS_ON : JUDGE_PROMPT_VERSION_DEBIAS_OFF;
129
- // Hash the template-shaping function source + the version tag together. We hash a
130
- // deterministic stringified form of the template (with placeholder inputs) so any
131
- // structural edit shows up.
132
- const sample = buildJudgePrompt('<P>', '<R>', '<O>', '<T>', lengthDebias);
133
- return createHash('sha256').update(version + '\n' + sample).digest('hex').slice(0, 12);
134
- }
135
52
  function getErrorMessage(err) {
136
53
  return err instanceof Error ? err.message : String(err);
137
54
  }
@@ -412,7 +329,7 @@ export function computeJudgeAgreement(judgeScores) {
412
329
  }
413
330
  /**
414
331
  * Judge a single (output, rubric) pair with N judge models in parallel. Each judge
415
- * may use a different executor (e.g. claude:opus + openai:gpt-4o + gemini:pro). Each
332
+ * may use a different executor (e.g. claude:opus + openai-api:gpt-4o + gemini:pro). Each
416
333
  * judge can also be repeated `judgeRepeat` times — final per-judge score is its mean.
417
334
  *
418
335
  * Returns: aggregate DimensionResult (score = mean across judges; this is the "consensus"
@@ -107,6 +107,9 @@ function validateEvalConfig(parsed, configPath) {
107
107
  throw new Error(`${configPath}: variants[${i}].allowedSkills[${j}] must be a non-empty string`);
108
108
  }
109
109
  }
110
+ if (v.allowedSkills.length > 0) {
111
+ throw new Error(`${configPath}: variants[${i}].allowedSkills must be [] (strict isolation: no skills) or omitted (no isolation) — a non-empty skill whitelist is no longer supported because it could not be fully isolated (the subagent Skill tool and cwd filesystem channels leaked). For a multi-skill experiment, control the eval environment instead.`);
112
+ }
110
113
  allowedSkills = v.allowedSkills;
111
114
  }
112
115
  if (seen.has(v.name)) {
@@ -146,11 +149,13 @@ function validateEvalConfig(parsed, configPath) {
146
149
  if (obj.judgeModel !== undefined || obj.judgeExecutor !== undefined) {
147
150
  throw new Error(`${configPath}: \`judgeModel\` and \`judgeExecutor\` were removed in v0.25 — use \`judgeModels: [{executor, model}]\` instead (single judge is the 1-entry case). See README.`);
148
151
  }
152
+ if (obj.blind !== undefined) {
153
+ throw new Error(`${configPath}: \`blind\` (judge blind mode) was removed — delete it from your eval.yaml; reports are no longer blinded.`);
154
+ }
149
155
  assertNumberOpt('concurrency');
150
156
  assertNumberOpt('timeoutMs');
151
157
  assertBoolOpt('noCache');
152
158
  assertBoolOpt('noJudge');
153
- assertBoolOpt('blind');
154
159
  assertStringOpt('mcpConfig');
155
160
  assertStringOpt('goldDir');
156
161
  assertBoolOpt('bootstrap');
@@ -166,6 +171,11 @@ function validateEvalConfig(parsed, configPath) {
166
171
  };
167
172
  assertPositiveIntOpt('repeat');
168
173
  assertPositiveIntOpt('judgeRepeat');
174
+ if (obj.holdoutRatio !== undefined) {
175
+ if (typeof obj.holdoutRatio !== 'number' || !Number.isFinite(obj.holdoutRatio) || obj.holdoutRatio <= 0 || obj.holdoutRatio >= 1) {
176
+ throw new Error(`${configPath}: holdoutRatio must be a number in (0, 1)`);
177
+ }
178
+ }
169
179
  if (obj.bootstrapSamples !== undefined) {
170
180
  if (typeof obj.bootstrapSamples !== 'number' || !Number.isFinite(obj.bootstrapSamples) || obj.bootstrapSamples < 100) {
171
181
  throw new Error(`${configPath}: bootstrapSamples must be a number ≥ 100`);
@@ -247,11 +257,11 @@ function validateEvalConfig(parsed, configPath) {
247
257
  timeoutMs: obj.timeoutMs,
248
258
  noCache: obj.noCache,
249
259
  noJudge: obj.noJudge,
250
- blind: obj.blind,
251
260
  mcpConfig: obj.mcpConfig,
252
261
  variants,
253
262
  budget,
254
263
  repeat: obj.repeat,
264
+ holdoutRatio: obj.holdoutRatio,
255
265
  judgeRepeat: obj.judgeRepeat,
256
266
  bootstrap: obj.bootstrap,
257
267
  bootstrapSamples: obj.bootstrapSamples,
@@ -13,8 +13,7 @@
13
13
  * `deriveManagedState` 按 contentHash 匹配裁定(重装新内容后旧证据留存供回滚,却不让新内容显得已测)。
14
14
  * - **跨源**:用三级身份消歧把被测 variant 对到受管记录(install 与 eval 命名不一致:记录名是
15
15
  * skill 短名 `review`,而 eval 报告 variant key 可能是整串表达式 `git:HEAD:skills/review`、
16
- * eval.yaml 别名 `candidate`、blind 模式的 `A`/`B`)。`applyBlindMode` 盲化 `variants` 但**不**动
17
- * `artifactHashes` / `variantConfigs` 的键面,故三级都按真实键工作:
16
+ * eval.yaml 别名 `candidate`)。三级都按 `artifactHashes` / `variantConfigs` 的真实键工作:
18
17
  * (a) 显式**同名** variant —— 最强身份(本地 `--treatment <name>` 与 drift);
19
18
  * (b) **结构化源匹配** —— `variantConfigs[].locator(+ref)` 与 `record.source` 对齐(git / 远端 /
20
19
  * 别名),即便内容撞哈也能精确消歧;
@@ -33,7 +33,7 @@ export function buildVersionScores(record, reportsById) {
33
33
  continue;
34
34
  const variant = Object.keys(hashes).find((v) => hashes[v] === ev.contentHash);
35
35
  if (!variant)
36
- continue; // 匹配不到变体(旧 schema 不绑 / blind 未带哈)→ 跳过
36
+ continue; // 匹配不到变体(旧 schema 不绑)→ 跳过
37
37
  const s = report.summary?.[variant];
38
38
  const composite = s?.avgCompositeScore;
39
39
  if (typeof composite !== 'number' || Number.isNaN(composite))
@@ -459,7 +459,6 @@ export function renderRunDetail(report, lang = DEFAULT_LANG, skillContext) {
459
459
  ${m.sampleHashes ? `<span class="meta-tag" title="${t('sampleHashCountDesc', lang)}">${t('sampleHashCount', lang)}: ${Object.keys(m.sampleHashes).length}/${m.sampleCount}</span>` : ''}
460
460
  ${m.evaluationFramework ? `<span class="meta-tag" title="${t('evalFrameworkDesc', lang)}">${t('evalFrameworkLabel', lang)}: ${m.evaluationFramework === 'bootstrap' ? t('evalFrameworkBootstrap', lang) : m.evaluationFramework === 'both' ? t('evalFrameworkBoth', lang) : t('evalFrameworkTTest', lang)}</span>` : ''}
461
461
  ${renderDebiasModeTag(m.debiasMode, lang)}
462
- ${m.blind ? `<span class="meta-tag" style="color:var(--green)" data-i18n="blindLabel">${t('blindLabel', lang)}</span>` : ''}
463
462
  </div>`;
464
463
  const auditFingerprints = (() => {
465
464
  const auditTags = [
@@ -470,13 +469,6 @@ export function renderRunDetail(report, lang = DEFAULT_LANG, skillContext) {
470
469
  return '';
471
470
  return `<details class="audit-fingerprints"><summary>${lang === 'zh' ? '审计指纹(用于复现校验)' : 'Audit fingerprints (for reproducibility)'}</summary><div class="meta-tags">${auditTags}</div></details>`;
472
471
  })();
473
- const blindReveal = m.blind ? `
474
- <div style="margin:12px 0">
475
- <button onclick="document.getElementById('blind-reveal').style.display=document.getElementById('blind-reveal').style.display==='none'?'block':'none'" data-i18n="revealBlind">${t('revealBlind', lang)}</button>
476
- <div id="blind-reveal" style="display:none;margin-top:8px;padding:12px;background:var(--bg-surface);border:1px solid var(--border);border-radius:var(--radius)" role="region" aria-label="Blind variant mapping">
477
- ${Object.entries(m.blindMap || {}).map(([label, real]) => `<div style="font-size:13px;color:var(--text-secondary)"><strong>Variant ${e(label)}</strong> → ${e(real)}</div>`).join('')}
478
- </div>
479
- </div>` : '';
480
472
  // ──────────── 融合单页:Hero → 结论(verdict+六维) → 折叠次要区 → 统一逐用例 ────────────
481
473
  const agentOverview = renderAgentOverview(variants, summary, lang);
482
474
  const varianceSection = renderVarianceComparisons(report.variance, lang, Boolean(report.meta.layeredStats), summary);
@@ -503,7 +495,6 @@ export function renderRunDetail(report, lang = DEFAULT_LANG, skillContext) {
503
495
  : 'Each sample is both a score card and a functional test: composite + layered scores, assertions, diagnostics, and trace combined';
504
496
  const body = `
505
497
  ${conclusionPanel}
506
- ${blindReveal}
507
498
  ${setupFold}
508
499
  ${analysisFold}
509
500
  <section class="ev-samples">
@@ -58,7 +58,7 @@ export const I18N = {
58
58
  score: '分数', cost: '执行成本', time: '时间',
59
59
  deleteBtnText: '删除', deleteConfirm: '确定删除报告', deleteFail: '删除失败',
60
60
  reportTitle: '评测报告', backToList: '← 返回列表',
61
- judge: '评委', executor: '执行器', blindLabel: '盲测', revealBlind: '显示变体对应关系',
61
+ judge: '评委', executor: '执行器',
62
62
  dimFact: '📋 事实', dimFactDesc: '输出的事实声明是否正确(规则可验证:关键词匹配、格式校验等断言)',
63
63
  dimBehavior: '🛠️ 行为', dimBehaviorDesc: '执行过程是否合规(规则可验证:工具调用路径、轮次限制、成本约束等断言)',
64
64
  dimJudge: '💬 LLM 评价', dimJudgeDesc: '请一个 LLM 当评委,读任务执行模型的输出内容,按预先写好的评分规则(英文 rubric)打 1-5 分。主观但能抓到规则断言判不了的"整体好不好"',
@@ -72,7 +72,7 @@ export const I18N = {
72
72
  judgeStddev: '评委波动', judgeStddevDesc: '同一份输出让评委评 N 次 (--judge-repeat) 得到 N 个分数的标准差。值低 = 评委对自己很坚定;值高 = 这个分本身就是噪声',
73
73
  judgeFailures: '评委失败', judgeFailuresDesc: 'N 次评委评分中返回 score=0(解析失败 / 调用错误)的次数。stddev=0 + failureCount>0 不是"完美一致",是"大部分炸了"',
74
74
  judgeReasoning: '评委推理', judgeReasoningExpand: '展开',
75
- ensembleHeader: '多评委评分对比', ensembleDesc: '不同评委模型对同一份输出的独立评分。用于反驳"同模态偏差"',
75
+ ensembleHeader: '多评委评分对比', ensembleDesc: '不同评委模型对同一份输出的独立评分。跨厂商评委时可反驳"同模态偏差";单厂商 ensemble 一致性高只是共有偏置,不构成反驳',
76
76
  agreementHeader: '跨用例评委一致性', agreementDesc: '在所有评测用例上算的多评委一致性',
77
77
  pearsonLabel: '皮尔逊系数 (Pearson)', pearsonDesc: '皮尔逊相关系数:1=完全同向排序,0=无关,-1=完全反向',
78
78
  madLabel: '平均绝对差 (MAD)', madDesc: '平均绝对差。1-5 制下 < 0.5 紧密一致, > 1.5 大分歧',
@@ -171,7 +171,7 @@ export const I18N = {
171
171
  score: 'Score', cost: 'Execution cost', time: 'Time',
172
172
  deleteBtnText: 'Delete', deleteConfirm: 'Delete report', deleteFail: 'Delete failed',
173
173
  reportTitle: 'Evaluation Report', backToList: '← Back to list',
174
- judge: 'Judge', executor: 'Executor', blindLabel: 'BLIND', revealBlind: 'Reveal variant mapping',
174
+ judge: 'Judge', executor: 'Executor',
175
175
  dimFact: '📋 Fact', dimFactDesc: 'Are factual claims correct (rule-verified: keyword matching, schema checks, etc.)',
176
176
  dimBehavior: '🛠️ Behavior', dimBehaviorDesc: 'Is execution compliant (rule-verified: tool paths, turn limits, cost constraints)',
177
177
  dimJudge: '💬 LLM judge', dimJudgeDesc: 'A separate LLM acts as judge: reads the task execution model output, scores 1-5 against a predefined rubric. Subjective, catches "overall feel" rule-based assertions miss',
@@ -185,7 +185,7 @@ export const I18N = {
185
185
  judgeStddev: 'Judge stddev', judgeStddevDesc: 'Stddev across N judge calls (--judge-repeat). Low = judge is consistent; high = this score itself is noisy',
186
186
  judgeFailures: 'Judge failures', judgeFailuresDesc: 'How many of N judge calls returned score=0 (parse / executor failure). stddev=0 + failureCount>0 is NOT "perfect agreement" — it means most calls failed',
187
187
  judgeReasoning: 'CoT reasoning', judgeReasoningExpand: 'expand',
188
- ensembleHeader: 'Per-judge scores', ensembleDesc: 'Independent scores from different judge models for the same output — refutes same-modality bias',
188
+ ensembleHeader: 'Per-judge scores', ensembleDesc: 'Independent scores from different judge models for the same output — refutes same-modality bias only when judges span vendors; a single-vendor ensemble agreement is shared bias, not a rebuttal',
189
189
  agreementHeader: 'Inter-judge agreement', agreementDesc: 'Cross-sample agreement metrics across all judges in this variant',
190
190
  pearsonLabel: 'Pearson', pearsonDesc: 'Pearson correlation: 1=perfect rank agreement, 0=independent, -1=anti-correlated',
191
191
  madLabel: 'MAD', madDesc: 'Mean absolute difference. On 1-5 scale: < 0.5 tight, > 1.5 large disagreement',
@@ -101,6 +101,30 @@ function computeMedianCVPercent(report) {
101
101
  const stab = medianStabilityCV(report);
102
102
  return stab ? stab.cv * 100 : null;
103
103
  }
104
+ // Verdict caveats(过拟合 / 知识缺口)渲染进 pill —— 让 HTML 报告和 CLI 说同一件事:
105
+ // CLI 在 verbose rationale 里给这两条,HTML 之前只剩一个泛化后的 level、看不到触发原因。
106
+ // 用 result.caveats 的结构化数据 i18n,而不是重解析 zh rationale 串。
107
+ function renderVerdictCaveats(caveats, lang) {
108
+ if (!caveats)
109
+ return '';
110
+ const lines = [];
111
+ if (caveats.overfitting) {
112
+ const c = caveats.overfitting;
113
+ lines.push(lang === 'zh'
114
+ ? `⚠ 过拟合敞口:${c.variant} 训练 ${c.trainScore.toFixed(2)} / 留出 ${c.holdoutScore.toFixed(2)}(差 ${c.gap.toFixed(2)}),提升可能不泛化`
115
+ : `⚠ Overfitting: ${c.variant} train ${c.trainScore.toFixed(2)} / holdout ${c.holdoutScore.toFixed(2)} (gap ${c.gap.toFixed(2)}) — gain may not generalize`);
116
+ }
117
+ if (caveats.gapSignal) {
118
+ const g = caveats.gapSignal;
119
+ const wm = g.testSetHash ? g.testSetHash.slice(0, 8) : (g.testSetPath ?? '');
120
+ lines.push(lang === 'zh'
121
+ ? `知识缺口率 ${g.gapRatePct}%(test set ${wm},informational)`
122
+ : `Knowledge gap ${g.gapRatePct}% (test set ${wm}, informational)`);
123
+ }
124
+ if (lines.length === 0)
125
+ return '';
126
+ return `<div class="page-verdict-caveats">${lines.map((l) => `<span class="page-verdict-caveat">${e(l)}</span>`).join('')}</div>`;
127
+ }
104
128
  export function renderVerdictPill(report, lang) {
105
129
  let result;
106
130
  try {
@@ -110,8 +134,16 @@ export function renderVerdictPill(report, lang) {
110
134
  return '';
111
135
  }
112
136
  const level = result.level;
113
- const pair = result.perPair?.[0];
137
+ // representative = top-level worst pair(与 CLI 同口径),不是第一对。多 treatment 报告里
138
+ // worst pair 不一定是 perPair[0],用它才不会把错的 treatment 名写进结论。fallback 兼容旧路径。
139
+ const pair = result.representative ?? result.perPair?.[0];
114
140
  const oneLine = verdictOneLine(level, lang, pair?.treatment, pair?.control);
141
+ // Δ/CI 证据必须跟文案指同一对:按 representative 匹配对应的 pairComparison(alpha 也走这对),
142
+ // 否则多 treatment 报告会出现「文案 t2、数字 t1」的混搭。匹配不到 / 无 representative 时 fallback [0]。
143
+ const pairComparisons = report.meta?.pairComparisons;
144
+ const activeComparison = (pair
145
+ ? pairComparisons?.find((p) => p.treatment === pair.treatment && p.control === pair.control)
146
+ : undefined) ?? pairComparisons?.[0];
115
147
  const tooltip = levelTooltip(level, lang);
116
148
  const prefix = lang === 'zh' ? '测评结论' : 'Verdict';
117
149
  // 机器可读 enum 永远是 level token; 显示给用户的文字按 lang i18n.
@@ -120,7 +152,7 @@ export function renderVerdictPill(report, lang) {
120
152
  // hero 只放「答案」: 分差是 verdict 的核心证据数字, 单独一枚 chip。
121
153
  // 评测规模 (用例数 × 轮次) 走「实验配置」section 的 subtitle 那条 canonical 路径,
122
154
  // 不在 hero 里重复; CV / CI 走 chip tooltip + 方法学审计 / 波动表。
123
- const ci = report.meta?.pairComparisons?.[0]?.diffBootstrapCI;
155
+ const ci = activeComparison?.diffBootstrapCI;
124
156
  const cvPct = computeMedianCVPercent(report);
125
157
  const metrics = [];
126
158
  if (ci) {
@@ -128,7 +160,7 @@ export function renderVerdictPill(report, lang) {
128
160
  const cvSuffix = cvPct != null
129
161
  ? (lang === 'zh' ? `;多轮稳定性 CV=${cvPct.toFixed(1)}% (${cvPct < 5 ? '稳' : cvPct <= STABILITY_UNSTABLE_CV * 100 ? '中' : '不稳'})` : `; CV=${cvPct.toFixed(1)}% (${cvPct < 5 ? 'stable' : cvPct <= STABILITY_UNSTABLE_CV * 100 ? 'moderate' : 'unstable'})`)
130
162
  : '';
131
- const pctLabel = ciLevelLabel(report.meta?.pairComparisons?.[0]?.alpha);
163
+ const pctLabel = ciLevelLabel(activeComparison?.alpha);
132
164
  const ciTipBase = lang === 'zh'
133
165
  ? `实验组与对照组综合分均值差(Δ)。bootstrap ${pctLabel} 可信区间 [${ci.low}, ${ci.high}],${ci.significant ? '不含 0 = 差异显著' : '跨过 0 = 差异不显著'}${cvSuffix}`
134
166
  : `Treatment minus control mean composite score (Δ). Bootstrap ${pctLabel} CI [${ci.low}, ${ci.high}], ${ci.significant ? 'excludes 0 ⇒ significant' : 'spans 0 ⇒ not significant'}${cvSuffix}`;
@@ -146,6 +178,7 @@ export function renderVerdictPill(report, lang) {
146
178
  <span class="page-verdict-badge"><span class="page-verdict-badge-dot" aria-hidden="true">●</span>${e(levelDisplay)}</span>
147
179
  <span class="page-verdict-text">${e(oneLine)}</span>
148
180
  </div>
181
+ ${renderVerdictCaveats(result.caveats, lang)}
149
182
  ${metricChips ? `<div class="page-verdict-metrics">${metricChips}</div>` : ''}
150
183
  </section>`;
151
184
  }
@@ -252,7 +285,7 @@ export function renderHumanAgreement(agreement, lang) {
252
285
  <td><strong>Krippendorff α</strong></td>
253
286
  <td style="text-align:center;color:${alphaColor}"><strong>${fmt(a.alpha)}</strong></td>
254
287
  <td style="text-align:center;font-size:11px">[${fmt(a.alphaCI.low)}, ${fmt(a.alphaCI.high)}]</td>
255
- <td style="font-size:12px;color:var(--text-secondary)">${lang === 'zh' ? '主指标,序数加权' : 'primary, ordinal-weighted'}</td>
288
+ <td style="font-size:12px;color:var(--text-secondary)">${lang === 'zh' ? '主指标,区间加权' : 'primary, interval-weighted'}</td>
256
289
  </tr>
257
290
  <tr>
258
291
  <td>${lang === 'zh' ? '加权 κ' : 'weighted κ'}</td>
@@ -1496,6 +1529,20 @@ function localizedInsightMessage(insight, report, lang) {
1496
1529
  ? `${count} 个评测用例成本显著高于平均值`
1497
1530
  : `${count} samples cost materially more than average`;
1498
1531
  }
1532
+ case 'judge_self_preference': {
1533
+ const outv = asArray(details.outputVendors).map(String).join('/') || '同厂商';
1534
+ const calibrated = details.goldCalibrated === true;
1535
+ return lang === 'zh'
1536
+ ? `评委与被测输出同厂商(${outv})${calibrated ? '(已 gold 校准)' : ''},存在自我偏好敞口:评委可能给同家族输出打高分`
1537
+ : `Judge is the same vendor as the executor that produced the outputs (${outv})${calibrated ? ' (gold-calibrated)' : ''}; self-preference exposure — the judge may over-score same-family output`;
1538
+ }
1539
+ case 'single_vendor_ensemble': {
1540
+ const n = asArray(details.judgeVendors).length;
1541
+ const calibrated = details.goldCalibrated === true;
1542
+ return lang === 'zh'
1543
+ ? `${n} 个评委同属一个厂商${calibrated ? '(已 gold 校准)' : ''},ensemble 一致性高只反映共有偏置,不构成对同模型偏置的反驳`
1544
+ : `All ${n} judges are from one vendor${calibrated ? ' (gold-calibrated)' : ''}; high ensemble agreement reflects shared bias, not a rebuttal to same-model bias`;
1545
+ }
1499
1546
  default:
1500
1547
  return lang === 'zh'
1501
1548
  ? `结构化诊断:${type}`
@@ -1544,6 +1591,14 @@ function localizedSuggestion(insight, lang) {
1544
1591
  return lang === 'zh'
1545
1592
  ? '重写 agent 断言时优先约束工具路径、关键文件读取和 turns 上限'
1546
1593
  : 'When rewriting agent assertions, prioritize tool path, key file reads, and turn-limit constraints';
1594
+ case 'judge_self_preference':
1595
+ return lang === 'zh'
1596
+ ? '换跨厂商评委(如 --judge-models openai-api:gpt-4o)消除自我偏好,或挂人工金标校准(omk eval gold compare)量化评委是否可信。注:固定模型的 A/B 差值受影响较小,绝对分 / 跨版本曲线受影响更大'
1597
+ : 'Use a cross-vendor judge (e.g. --judge-models openai-api:gpt-4o) to remove self-preference, or calibrate against human gold (omk eval gold compare). Note: the A/B delta is less affected than absolute scores / cross-version curves';
1598
+ case 'single_vendor_ensemble':
1599
+ return lang === 'zh'
1600
+ ? '把 ensemble 配成跨厂商(如 --judge-models claude:opus,openai-api:gpt-4o),让 agreement 真能反驳同模型偏置'
1601
+ : 'Make the ensemble cross-vendor (e.g. --judge-models claude:opus,openai-api:gpt-4o) so agreement actually rebuts same-model bias';
1547
1602
  default:
1548
1603
  return lang === 'zh'
1549
1604
  ? '查看结构化 details 字段定位原因'
@@ -0,0 +1,4 @@
1
+ export declare const LENGTH_DEBIAS_INSTRUCTION: string;
2
+ export declare const PRESENTATION_NEUTRALITY_INSTRUCTION: string;
3
+ export declare const RAG_LENGTH_DEBIAS: string;
4
+ export declare const RAG_PRESENTATION_NEUTRALITY: string;
@@ -0,0 +1,44 @@
1
+ // 评委 prompt 的共享去偏 / 中性化指令 —— 单一来源。
2
+ //
3
+ // 此前这几段在 grading/judge.ts(rubric 评委)与 grading/assertions.ts(RAG 评委)各写一份,
4
+ // 改一处漏一处(J3 即如此)。统一收口于此,两侧 import。
5
+ //
6
+ // 两套变体是有意的、不要强并成一条:
7
+ // - 全版(LENGTH_DEBIAS_INSTRUCTION / PRESENTATION_NEUTRALITY_INSTRUCTION):给 rubric 评委,
8
+ // 带研究依据句;rubric prompt 较短,容得下。被 getJudgePromptHash 冻结,改字节 = BREAKING-COMPARABILITY。
9
+ // - 短版(RAG_LENGTH_DEBIAS / RAG_PRESENTATION_NEUTRALITY):给 RAG 评委,RAG prompt 本就长,精简表述。
10
+ // 改任一段先想另一段是否同步;改全版还要走冻结 hash 的 bump 流程。
11
+ export const LENGTH_DEBIAS_INSTRUCTION = [
12
+ '## 重要:长度不是质量信号',
13
+ '评分时聚焦内容实质与正确性。回答的篇幅、行文丰富度、结构复杂度本身不是质量指标 ——',
14
+ '简洁正确的回答不应因短而扣分;冗长但偏题或重复的回答不应因长而加分。',
15
+ '研究显示 LLM 评委容易隐性偏向更长的回答,请在打分前先警觉这一点。',
16
+ ].join('\n');
17
+ // 始终开启(不给开关):排版与语气都不是质量信号。措辞严格对称 —— 既不奖励精致 / 自信,也不
18
+ // 因朴素 / 含糊而扣分,避免"抗偏置"本身过度矫正成反向偏置。研究表明 LLM 评委除长度外,还隐性
19
+ // 偏向排版精致(format / markdown bias)与语气自信、自我表扬的回答(谄媚 / 权威偏置)。
20
+ export const PRESENTATION_NEUTRALITY_INSTRUCTION = [
21
+ '## 重要:排版与语气不是质量信号',
22
+ '评分只看内容是否对照评分标准、是否正确。Markdown 排版、标题、列表、加粗、表格等呈现形式',
23
+ '本身不是质量指标 —— 朴素但正确的回答不应因没有排版而扣分;排版精致但偏题或错误的回答不应',
24
+ '因好看而加分。',
25
+ '同样,回答的语气、自信程度、是否自我表扬(如「这是最优方案」)也不是质量信号 —— 不要被笃定',
26
+ '的口吻或自我评价带跑:自信但错误的回答不应高于含糊但正确的回答,一切结论都要回到评分标准',
27
+ '逐条核实。',
28
+ '研究显示 LLM 评委容易隐性偏向排版精致、语气自信的回答,请在打分前先警觉这两点。',
29
+ ].join('\n');
30
+ // RAG 指标(faithfulness / answer_relevancy / context_recall)的去偏**恒开**,
31
+ // 按 default-strict 不提供关闭开关:`--no-debias-length` 只作用于 rubric 评委 prompt 的长度去偏,
32
+ // 不下探到 RAG —— RAG 评的是内容保真 / 切题 / 召回,放任长度 / 排版 / 语气偏置只会污染这些指标,
33
+ // 没有「关掉它」的正当用例。故 runRagJudge 不收 lengthDebias 参数,两段恒注入。
34
+ export const RAG_LENGTH_DEBIAS = [
35
+ '## 重要:长度不是质量信号',
36
+ '评分时聚焦内容实质,不要因输出更长就给更高分。',
37
+ '简洁正确的回答与冗长正确的回答应得相同分数。',
38
+ ].join('\n');
39
+ // 排版 / 语气中性化(format / sycophancy bias),与 RAG_LENGTH_DEBIAS 同 footprint 恒开。
40
+ export const RAG_PRESENTATION_NEUTRALITY = [
41
+ '## 重要:排版与语气不是质量信号',
42
+ '评分只看内容是否符合上述标准。排版是否精致、语气是否自信、有无自我表扬都不是质量信号 ——',
43
+ '不要被笃定的口吻带跑:自信但错误的回答不应高于含糊但正确的回答。',
44
+ ].join('\n');
@@ -0,0 +1,30 @@
1
+ export declare const JUDGE_SYSTEM_PROMPT = "\u4F60\u662F\u4E00\u4E2A\u4E25\u683C\u7684 AI \u8F93\u51FA\u8D28\u91CF\u8BC4\u5BA1\u5458\u3002\u5148\u9010\u6761\u5BF9\u7167\u8BC4\u5206\u6807\u51C6\u505A\u63A8\u7406\uFF0C\u518D\u7ED9\u6700\u7EC8\u5206\u6570\u3002\u53EA\u8FD4\u56DE JSON\uFF0C\u4E0D\u8981\u5176\u4ED6\u5185\u5BB9\u3002";
2
+ export declare function buildJudgePrompt(prompt: string, rubric: string, output: string, traceSummary: string | null, lengthDebias?: boolean): string;
3
+ /**
4
+ * Stable hash of the judge prompt template. Saved into ReportMeta.judgePromptHash so
5
+ * downstream readers can detect "the judge prompt changed between these two reports".
6
+ *
7
+ * `lengthDebias` defaults to true. Pass false (via `--no-debias-length`) to drop the
8
+ * length-debias instruction; that produces the debias-off prompt variant, whose hash
9
+ * differs from the default so the two are never compared blind.
10
+ */
11
+ export declare function getJudgePromptHash(lengthDebias?: boolean): string;
12
+ export declare const SEMANTIC_SIMILARITY_SYSTEM = "\u4F60\u662F\u8BED\u4E49\u76F8\u4F3C\u5EA6\u8BC4\u5BA1\u5458\u3002\u53EA\u8FD4\u56DE JSON\uFF0C\u4E0D\u8981\u5176\u4ED6\u5185\u5BB9\u3002";
13
+ export declare function buildSemanticSimilarityPrompt(reference: string, output: string): string;
14
+ export type RagJudgeType = 'faithfulness' | 'answer_relevancy' | 'context_recall';
15
+ export interface RagJudgeFields {
16
+ output: string;
17
+ /** faithfulness: 参考 context。 */
18
+ context?: string;
19
+ /** answer_relevancy: 用户问题(sample.prompt)。 */
20
+ question?: string;
21
+ /** context_recall: 参考 gold。 */
22
+ reference?: string;
23
+ }
24
+ /** RAG 评委 system + 用户 prompt。缺字段的早退 / 解析逻辑留在 assertions.ts。 */
25
+ export declare function buildRagJudgePrompt(type: RagJudgeType, fields: RagJudgeFields): {
26
+ system: string;
27
+ prompt: string;
28
+ };
29
+ export declare function getSemanticPromptHash(): string;
30
+ export declare function getRagJudgePromptHash(type: RagJudgeType): string;