oh-my-knowledge 0.40.0 → 0.42.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (71) hide show
  1. package/README.md +8 -4
  2. package/README.zh.md +8 -4
  3. package/dist/analysis/report-diagnostics.d.ts +8 -1
  4. package/dist/analysis/report-diagnostics.js +109 -1
  5. package/dist/assets/agent-skills/omk/references/commands.md +2 -2
  6. package/dist/authoring/evolver.d.ts +3 -14
  7. package/dist/authoring/evolver.js +1 -52
  8. package/dist/authoring/generator.d.ts +24 -0
  9. package/dist/authoring/generator.js +36 -8
  10. package/dist/cli/commands/eval/index.d.ts +1 -1
  11. package/dist/cli/commands/eval/index.js +50 -15
  12. package/dist/cli/commands/init.js +10 -7
  13. package/dist/cli/lib/cmd-flags.d.ts +0 -1
  14. package/dist/cli/lib/i18n-dict/init.d.ts +1 -1
  15. package/dist/cli/lib/i18n-dict/init.js +14 -11
  16. package/dist/cli/lib/i18n-dict/run.d.ts +1 -1
  17. package/dist/cli/lib/i18n-dict/run.js +6 -2
  18. package/dist/cli/lib/parse-run-config.d.ts +3 -1
  19. package/dist/cli/lib/parse-run-config.js +0 -2
  20. package/dist/eval-core/evaluation-job.d.ts +2 -2
  21. package/dist/eval-core/evaluation-job.js +2 -2
  22. package/dist/eval-core/evaluation-reporting.d.ts +0 -1
  23. package/dist/eval-core/evaluation-reporting.js +9 -41
  24. package/dist/eval-core/execution-strategy.js +3 -2
  25. package/dist/eval-core/holdout.d.ts +66 -0
  26. package/dist/eval-core/holdout.js +118 -0
  27. package/dist/eval-core/judge-independence.d.ts +28 -0
  28. package/dist/eval-core/judge-independence.js +29 -0
  29. package/dist/eval-core/verdict.d.ts +53 -2
  30. package/dist/eval-core/verdict.js +216 -16
  31. package/dist/eval-workflows/batch-evaluation-workflow.d.ts +1 -1
  32. package/dist/eval-workflows/batch-evaluation-workflow.js +1 -2
  33. package/dist/eval-workflows/evaluation-pipeline/report-finalize.d.ts +1 -5
  34. package/dist/eval-workflows/evaluation-pipeline/report-finalize.js +19 -8
  35. package/dist/eval-workflows/evaluation-pipeline/run-state.d.ts +2 -2
  36. package/dist/eval-workflows/evaluation-pipeline/run-state.js +2 -2
  37. package/dist/eval-workflows/evaluation-pipeline.d.ts +3 -2
  38. package/dist/eval-workflows/evaluation-pipeline.js +3 -4
  39. package/dist/eval-workflows/run-evaluation.d.ts +6 -4
  40. package/dist/eval-workflows/run-evaluation.js +8 -6
  41. package/dist/executors/claude-cli.js +5 -6
  42. package/dist/executors/claude-sdk.d.ts +5 -2
  43. package/dist/executors/claude-sdk.js +13 -8
  44. package/dist/executors/codex-cli.js +3 -4
  45. package/dist/executors/shared.d.ts +2 -0
  46. package/dist/executors/shared.js +15 -0
  47. package/dist/grading/assertions.js +6 -122
  48. package/dist/grading/gold-cli.js +1 -1
  49. package/dist/grading/human-gold.d.ts +5 -3
  50. package/dist/grading/human-gold.js +5 -3
  51. package/dist/grading/index.d.ts +4 -4
  52. package/dist/grading/judge.d.ts +6 -14
  53. package/dist/grading/judge.js +5 -88
  54. package/dist/inputs/eval-config.js +12 -2
  55. package/dist/managed/evidence.js +1 -2
  56. package/dist/managed/version-scores.js +1 -1
  57. package/dist/renderer/html-renderer.js +0 -9
  58. package/dist/renderer/layout.js +4 -4
  59. package/dist/renderer/summary.js +59 -4
  60. package/dist/shared/llm-prompts/debias-instructions.d.ts +4 -0
  61. package/dist/shared/llm-prompts/debias-instructions.js +44 -0
  62. package/dist/shared/llm-prompts/judge-prompts.d.ts +30 -0
  63. package/dist/shared/llm-prompts/judge-prompts.js +205 -0
  64. package/dist/shared/llm-prompts/registry.d.ts +27 -0
  65. package/dist/shared/llm-prompts/registry.js +69 -0
  66. package/dist/types/eval.d.ts +15 -8
  67. package/dist/types/judge.d.ts +1 -1
  68. package/dist/types/report.d.ts +50 -3
  69. package/package.json +1 -1
  70. package/dist/grading/debias-validate.d.ts +0 -83
  71. package/dist/grading/debias-validate.js +0 -176
@@ -1,176 +0,0 @@
1
- /**
2
- * Measure how much the judge's scores shift when the length-debias instruction
3
- * is toggled.
4
- *
5
- * What it actually measures
6
- * -------------------------
7
- * Given a finished report, this command re-judges every (sample, variant) pair
8
- * using the OPPOSITE length-debias setting from the original run. It then
9
- * compares the two score distributions with a bootstrap CI on the mean
10
- * difference.
11
- *
12
- * If the original report ran with debias-on (v3-cot-length), we re-judge with
13
- * v2-cot (legacy). If the original ran with debias-off, we re-judge with
14
- * v3-cot-length. Significant difference → the prompt change moves scores → the
15
- * judge is sensitive to the length-debias instruction. That's *consistent with*
16
- * length bias being present, but it doesn't prove it directly — a perfectly
17
- * length-neutral judge could in principle also be sensitive to the wording for
18
- * other reasons. We label the verdict accordingly.
19
- *
20
- * Cost
21
- * ----
22
- * Re-judging is a full second pass over all (sample, variant) cells. Cost
23
- * doubles vs the original run. The CLI surfaces this so users running on
24
- * large/expensive evaluations can opt in deliberately.
25
- */
26
- import { llmJudge } from './judge.js';
27
- import { bootstrapPairedDiffCI } from '../eval-core/bootstrap.js';
28
- /**
29
- * Map a bootstrap diff CI to a verdict bucket. The ranges are deliberately
30
- * conservative: we only label "strong" when the CI fully sits >= |0.5| away
31
- * from zero (about half a point on a 1-5 scale).
32
- */
33
- function classifyVerdict(diff) {
34
- if (!diff.significant) {
35
- return { zh: '未检测到显著差异', en: 'no significant shift', level: 'none' };
36
- }
37
- const mag = Math.min(Math.abs(diff.low), Math.abs(diff.high));
38
- if (mag >= 0.5) {
39
- return { zh: '强差异——prompt 改动对评分影响大', en: 'strong shift', level: 'strong' };
40
- }
41
- if (mag >= 0.2) {
42
- return { zh: '中等差异——校正对结论有实质影响', en: 'medium shift', level: 'medium' };
43
- }
44
- return { zh: '弱差异——显著但幅度小', en: 'weak shift', level: 'weak' };
45
- }
46
- /**
47
- * Re-judge every sample of `variant` in the given report with the OPPOSITE
48
- * lengthDebias setting and compute the bootstrap CI on the mean difference.
49
- *
50
- * The judge call uses the rubric from the samples file and the output stored
51
- * in the report — we do NOT re-execute the model. Only judging is repeated.
52
- *
53
- * Multi-dimensional samples currently use the rubric as fallback when there's
54
- * no top-level rubric. Per-dimension validation can be added later if needed.
55
- */
56
- export async function validateLengthDebias(input) {
57
- const { report, samples, judgeExecutor, judgeModel, bootstrapSamples = 1000, seed, onProgress } = input;
58
- const variant = input.variant ?? report.meta.variants?.[0];
59
- if (!variant)
60
- throw new Error('report has no variants — nothing to validate');
61
- // Detect original lengthDebias setting from meta. Default to true (v0.21+).
62
- // Older reports without debiasMode set are treated as legacy (debias-off).
63
- const debiasModeList = report.meta.debiasMode ?? [];
64
- const originalLengthDebias = debiasModeList.includes('length');
65
- const alternateLengthDebias = !originalLengthDebias;
66
- const sampleById = new Map();
67
- for (const s of samples)
68
- sampleById.set(s.sample_id, s);
69
- // Pre-pass: collect (sample_id, output, originalScore) tuples.
70
- const tasks = [];
71
- const unscored = [];
72
- const missing = [];
73
- for (const entry of report.results ?? []) {
74
- const v = entry.variants?.[variant];
75
- if (!v || !v.fullOutput)
76
- continue;
77
- const sample = sampleById.get(entry.sample_id);
78
- if (!sample) {
79
- missing.push(entry.sample_id);
80
- continue;
81
- }
82
- if (typeof v.llmScore !== 'number' || v.llmScore <= 0) {
83
- unscored.push(entry.sample_id);
84
- continue;
85
- }
86
- tasks.push({ sample, output: v.fullOutput, originalScore: v.llmScore });
87
- }
88
- // Re-judge with the opposite debias setting. We use the simplest path:
89
- // single-rubric judge. Multi-dim samples fall back to the explicit rubric
90
- // string if available. Samples without a rubric are skipped — we have
91
- // nothing to feed the judge.
92
- const pairs = [];
93
- let alternateJudgeCostUSD = 0;
94
- let completed = 0;
95
- for (const t of tasks) {
96
- completed++;
97
- onProgress?.({ sample_id: t.sample.sample_id, completed, total: tasks.length });
98
- const rubric = t.sample.rubric
99
- ?? (t.sample.dimensions ? Object.values(t.sample.dimensions).join('\n') : '');
100
- if (!rubric)
101
- continue;
102
- const altResult = await llmJudge({
103
- output: t.output,
104
- rubric,
105
- prompt: t.sample.prompt,
106
- executor: judgeExecutor,
107
- model: judgeModel,
108
- lengthDebias: alternateLengthDebias,
109
- });
110
- if (altResult.judgeCostUSD)
111
- alternateJudgeCostUSD += altResult.judgeCostUSD;
112
- if (altResult.score > 0) {
113
- pairs.push({
114
- sample_id: t.sample.sample_id,
115
- originalScore: t.originalScore,
116
- alternateScore: altResult.score,
117
- });
118
- }
119
- }
120
- const meanOriginal = avg(pairs.map((p) => p.originalScore));
121
- const meanAlternate = avg(pairs.map((p) => p.alternateScore));
122
- // **配对** diff CI:每个 sample 同时有 original 与 alternate prompt 两个分数(同一回答、两个 judge prompt
123
- // = 配对设计)。原先拆成两数组喂独立重采样,丢弃了配对、高估方差、CI 偏宽 —— 对一个**检测**长度偏置敏感性
124
- // 的工具,保守方向恰好是错的(更难检出真实偏置)。改配对:重采样 sample 下标、按 (alternate − original) 算,
125
- // 保留 within-sample 相关、收紧 CI,提升对偏置的检出力。diff = b − a = alternate − original(同原约定)。
126
- const diffCI = bootstrapPairedDiffCI(pairs.map((p) => ({ a: p.originalScore, b: p.alternateScore })), 0.05, bootstrapSamples, seed);
127
- const verdict = classifyVerdict(diffCI);
128
- return {
129
- variant,
130
- originalLengthDebias,
131
- pairs,
132
- meanOriginal: Number(meanOriginal.toFixed(3)),
133
- meanAlternate: Number(meanAlternate.toFixed(3)),
134
- diffCI,
135
- verdict,
136
- alternateJudgeCostUSD: Number(alternateJudgeCostUSD.toFixed(6)),
137
- unscored,
138
- missing,
139
- };
140
- }
141
- function avg(arr) {
142
- if (arr.length === 0)
143
- return 0;
144
- return arr.reduce((s, x) => s + x, 0) / arr.length;
145
- }
146
- export function formatDebiasValidate(result) {
147
- const lines = [];
148
- const dirOrig = result.originalLengthDebias ? 'on (v3-cot-length)' : 'off (v2-cot)';
149
- const dirAlt = result.originalLengthDebias ? 'off (v2-cot)' : 'on (v3-cot-length)';
150
- lines.push(`\n Length-debias 灵敏度验证 (variant: ${result.variant})\n`);
151
- lines.push(` 原始 prompt: ${dirOrig}`);
152
- lines.push(` 对照 prompt: ${dirAlt}`);
153
- lines.push(` 用例数: ${result.pairs.length}`);
154
- if (result.pairs.length === 0) {
155
- lines.push(' 无可比对用例——检查报告是否含 fullOutput / rubric。');
156
- return lines.join('\n');
157
- }
158
- lines.push(` 原均值: ${result.meanOriginal.toFixed(3)}`);
159
- lines.push(` 对照均值: ${result.meanAlternate.toFixed(3)}`);
160
- const ci = result.diffCI;
161
- lines.push(` 差值 (alt-orig): ${ci.estimate >= 0 ? '+' : ''}${ci.estimate}`);
162
- lines.push(` 95% CI: [${ci.low}, ${ci.high}] (${ci.significant ? '显著' : '不显著'})`);
163
- lines.push('');
164
- lines.push(` 结论: ${result.verdict.zh}`);
165
- lines.push('');
166
- lines.push(` 注: 该结论反映"prompt 切换是否改变评分"。差异显著 = 评分对 length-debias 指令敏感,`);
167
- lines.push(` 间接支持 length bias 存在;但 prompt 文本变化也可能因其他原因影响评分。`);
168
- lines.push(` 重判 cost: ${result.alternateJudgeCostUSD.toFixed(6)} USD`);
169
- if (result.missing.length) {
170
- lines.push(` 缺用例 ID: ${result.missing.length} 条`);
171
- }
172
- if (result.unscored.length) {
173
- lines.push(` 无 LLM 分: ${result.unscored.length} 条`);
174
- }
175
- return lines.join('\n');
176
- }