oh-my-knowledge 0.40.0 → 0.41.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +1 -2
- package/README.zh.md +1 -2
- package/dist/analysis/report-diagnostics.js +27 -0
- package/dist/assets/agent-skills/omk/references/commands.md +1 -2
- package/dist/authoring/generator.js +3 -2
- package/dist/cli/commands/eval/index.d.ts +0 -1
- package/dist/cli/commands/eval/index.js +10 -10
- package/dist/cli/lib/cmd-flags.d.ts +0 -1
- package/dist/cli/lib/i18n-dict/run.js +2 -2
- package/dist/cli/lib/parse-run-config.d.ts +0 -1
- package/dist/cli/lib/parse-run-config.js +0 -2
- package/dist/eval-core/evaluation-job.d.ts +1 -2
- package/dist/eval-core/evaluation-job.js +1 -2
- package/dist/eval-core/evaluation-reporting.d.ts +0 -1
- package/dist/eval-core/evaluation-reporting.js +2 -38
- package/dist/eval-core/execution-strategy.js +3 -2
- package/dist/eval-core/judge-independence.d.ts +28 -0
- package/dist/eval-core/judge-independence.js +29 -0
- package/dist/eval-core/verdict.d.ts +9 -1
- package/dist/eval-core/verdict.js +43 -5
- package/dist/eval-workflows/batch-evaluation-workflow.d.ts +1 -1
- package/dist/eval-workflows/batch-evaluation-workflow.js +1 -2
- package/dist/eval-workflows/evaluation-pipeline/report-finalize.d.ts +1 -5
- package/dist/eval-workflows/evaluation-pipeline/report-finalize.js +1 -8
- package/dist/eval-workflows/evaluation-pipeline/run-state.d.ts +1 -2
- package/dist/eval-workflows/evaluation-pipeline/run-state.js +1 -2
- package/dist/eval-workflows/evaluation-pipeline.d.ts +1 -2
- package/dist/eval-workflows/evaluation-pipeline.js +1 -3
- package/dist/eval-workflows/run-evaluation.d.ts +2 -3
- package/dist/eval-workflows/run-evaluation.js +1 -2
- package/dist/executors/claude-cli.js +5 -6
- package/dist/executors/claude-sdk.d.ts +5 -2
- package/dist/executors/claude-sdk.js +13 -8
- package/dist/executors/codex-cli.js +3 -4
- package/dist/executors/shared.d.ts +2 -0
- package/dist/executors/shared.js +15 -0
- package/dist/grading/assertions.js +6 -122
- package/dist/grading/gold-cli.js +1 -1
- package/dist/grading/human-gold.d.ts +5 -3
- package/dist/grading/human-gold.js +5 -3
- package/dist/grading/index.d.ts +4 -4
- package/dist/grading/judge.d.ts +6 -14
- package/dist/grading/judge.js +5 -88
- package/dist/inputs/eval-config.js +6 -2
- package/dist/managed/evidence.js +1 -2
- package/dist/managed/version-scores.js +1 -1
- package/dist/renderer/html-renderer.js +0 -9
- package/dist/renderer/layout.js +4 -4
- package/dist/renderer/summary.js +23 -1
- package/dist/shared/llm-prompts/debias-instructions.d.ts +4 -0
- package/dist/shared/llm-prompts/debias-instructions.js +44 -0
- package/dist/shared/llm-prompts/judge-prompts.d.ts +30 -0
- package/dist/shared/llm-prompts/judge-prompts.js +205 -0
- package/dist/shared/llm-prompts/registry.d.ts +27 -0
- package/dist/shared/llm-prompts/registry.js +69 -0
- package/dist/types/eval.d.ts +8 -8
- package/dist/types/judge.d.ts +1 -1
- package/dist/types/report.d.ts +1 -3
- package/package.json +1 -1
- package/dist/grading/debias-validate.d.ts +0 -83
- package/dist/grading/debias-validate.js +0 -176
|
@@ -0,0 +1,69 @@
|
|
|
1
|
+
import { getJudgePromptHash, getSemanticPromptHash, getRagJudgePromptHash } from './judge-prompts.js';
|
|
2
|
+
import { readPromptDocument } from './index.js';
|
|
3
|
+
import { PROMPTS_DIR, SOFT_STANDARD_PROMPT_ID, SOFT_STANDARD_PROMPT_VERSION } from '../../observability/soft-standards/constants.js';
|
|
4
|
+
export const PROMPT_REGISTRY = [
|
|
5
|
+
// —— 测量学不变量:直接决定分数的评委 prompt,统一冻结 ——
|
|
6
|
+
{
|
|
7
|
+
promptId: 'rubric-judge-debias-on',
|
|
8
|
+
purpose: 'rubric 主评委(length-debias 开,默认)',
|
|
9
|
+
module: 'src/shared/llm-prompts/judge-prompts.ts',
|
|
10
|
+
measurementInvariant: true,
|
|
11
|
+
getHash: () => getJudgePromptHash(true),
|
|
12
|
+
},
|
|
13
|
+
{
|
|
14
|
+
promptId: 'rubric-judge-debias-off',
|
|
15
|
+
purpose: 'rubric 主评委(--no-debias-length,length-debias 关)',
|
|
16
|
+
module: 'src/shared/llm-prompts/judge-prompts.ts',
|
|
17
|
+
measurementInvariant: true,
|
|
18
|
+
getHash: () => getJudgePromptHash(false),
|
|
19
|
+
},
|
|
20
|
+
{
|
|
21
|
+
promptId: 'semantic-similarity',
|
|
22
|
+
purpose: 'semantic_similarity 断言评委',
|
|
23
|
+
module: 'src/shared/llm-prompts/judge-prompts.ts',
|
|
24
|
+
measurementInvariant: true,
|
|
25
|
+
getHash: () => getSemanticPromptHash(),
|
|
26
|
+
},
|
|
27
|
+
{
|
|
28
|
+
promptId: 'rag-faithfulness',
|
|
29
|
+
purpose: 'RAG faithfulness(输出是否被 context 支持)',
|
|
30
|
+
module: 'src/shared/llm-prompts/judge-prompts.ts',
|
|
31
|
+
measurementInvariant: true,
|
|
32
|
+
getHash: () => getRagJudgePromptHash('faithfulness'),
|
|
33
|
+
},
|
|
34
|
+
{
|
|
35
|
+
promptId: 'rag-answer-relevancy',
|
|
36
|
+
purpose: 'RAG answer_relevancy(是否切题)',
|
|
37
|
+
module: 'src/shared/llm-prompts/judge-prompts.ts',
|
|
38
|
+
measurementInvariant: true,
|
|
39
|
+
getHash: () => getRagJudgePromptHash('answer_relevancy'),
|
|
40
|
+
},
|
|
41
|
+
{
|
|
42
|
+
promptId: 'rag-context-recall',
|
|
43
|
+
purpose: 'RAG context_recall(关键事实覆盖率)',
|
|
44
|
+
module: 'src/shared/llm-prompts/judge-prompts.ts',
|
|
45
|
+
measurementInvariant: true,
|
|
46
|
+
getHash: () => getRagJudgePromptHash('context_recall'),
|
|
47
|
+
},
|
|
48
|
+
{
|
|
49
|
+
promptId: 'observe-llm-enhanced-review',
|
|
50
|
+
purpose: 'observe 运行期软标准抽取 / LLM 增强复盘',
|
|
51
|
+
module: 'src/observability/prompts/llm-enhanced-review.prompt.md',
|
|
52
|
+
measurementInvariant: true,
|
|
53
|
+
getHash: () => readPromptDocument({
|
|
54
|
+
dir: PROMPTS_DIR,
|
|
55
|
+
fileName: `${SOFT_STANDARD_PROMPT_ID}.prompt.md`,
|
|
56
|
+
id: SOFT_STANDARD_PROMPT_ID,
|
|
57
|
+
version: SOFT_STANDARD_PROMPT_VERSION,
|
|
58
|
+
}).hash,
|
|
59
|
+
},
|
|
60
|
+
// —— 非评分类:不决定 composite / verdict / assertion 分数,不冻结,仅登记供发现 ——
|
|
61
|
+
{ promptId: 'failure-diagnostic', purpose: '失败诊断(root cause / 修复建议)', module: 'src/grading/diagnostic.ts', measurementInvariant: false },
|
|
62
|
+
{ promptId: 'failure-clusterer', purpose: '失败案例聚类', module: 'src/analysis/failure-clusterer.ts', measurementInvariant: false },
|
|
63
|
+
{ promptId: 'hedging-classifier', purpose: 'hedging 判定(喂 gap-signal,非评分)', module: 'src/analysis/hedging-classifier.ts', measurementInvariant: false },
|
|
64
|
+
{ promptId: 'sample-generator', purpose: '用例生成(skill / trace → samples)', module: 'src/authoring/generator.ts', measurementInvariant: false },
|
|
65
|
+
{ promptId: 'sample-fixer', purpose: '坏用例修复', module: 'src/authoring/sample-fixer.ts', measurementInvariant: false },
|
|
66
|
+
{ promptId: 'skill-improve', purpose: 'skill 迭代改进(evolve)', module: 'src/authoring/evolver.ts', measurementInvariant: false },
|
|
67
|
+
{ promptId: 'doctor-fixer', purpose: 'doctor 健康项修复向导', module: 'src/doctor/fixer.ts', measurementInvariant: false },
|
|
68
|
+
{ promptId: 'skill-health', purpose: 'skill 健康检查打分', module: 'src/shared/llm-prompts/skill-health.ts', measurementInvariant: false },
|
|
69
|
+
];
|
package/dist/types/eval.d.ts
CHANGED
|
@@ -211,7 +211,6 @@ export interface EvalConfig {
|
|
|
211
211
|
timeoutMs?: number;
|
|
212
212
|
noCache?: boolean;
|
|
213
213
|
noJudge?: boolean;
|
|
214
|
-
blind?: boolean;
|
|
215
214
|
mcpConfig?: string;
|
|
216
215
|
variants: EvalConfigVariant[];
|
|
217
216
|
/** hard budget caps. When any limit is hit during a run, remaining
|
|
@@ -229,7 +228,7 @@ export interface EvalConfig {
|
|
|
229
228
|
bootstrapSamples?: number;
|
|
230
229
|
/** --gold-dir. After-run automatic comparison against a human-anchor dataset. */
|
|
231
230
|
goldDir?: string;
|
|
232
|
-
/** --no-debias-length flips this to false. Default true (
|
|
231
|
+
/** --no-debias-length flips this to false. Default true (length-debias instruction on). */
|
|
233
232
|
lengthDebias?: boolean;
|
|
234
233
|
/** --no-strict-baseline flips this to false. Default true (baseline-kind allowedSkills=[]). */
|
|
235
234
|
strictBaseline?: boolean;
|
|
@@ -256,7 +255,6 @@ export interface EvaluationRequest {
|
|
|
256
255
|
timeoutMs?: number;
|
|
257
256
|
noCache: boolean;
|
|
258
257
|
dryRun: boolean;
|
|
259
|
-
blind: boolean;
|
|
260
258
|
/** --repeat N; 1 表示单次跑,> 1 走 runMultiple 做 variance 分析 */
|
|
261
259
|
repeat?: number;
|
|
262
260
|
/** --batch; default absent/false. True means skill-batch mode. */
|
|
@@ -266,8 +264,9 @@ export interface EvaluationRequest {
|
|
|
266
264
|
/** Unified judge config — always non-empty.
|
|
267
265
|
* - length === 1: single judge (degenerate ensemble of size 1).
|
|
268
266
|
* - length >= 2: multi-judge ensemble. Each (sample × dimension) is scored by every judge,
|
|
269
|
-
* inter-judge agreement (Pearson + mean absolute difference) reported as
|
|
270
|
-
*
|
|
267
|
+
* inter-judge agreement (Pearson + mean absolute difference) reported as a rebuttal to
|
|
268
|
+
* "judge same-model bias" — but only when judges span vendors; a single-vendor ensemble's
|
|
269
|
+
* high agreement reflects shared bias, not independence (flagged by `single_vendor_ensemble`).
|
|
271
270
|
* When `noJudge: true` the entry is preserved for audit but no judge call actually runs. */
|
|
272
271
|
judgeModels: JudgeConfig[];
|
|
273
272
|
/** --bootstrap; true 时 aggregateReport 加跑 bootstrap mean/diff CI, 写入 VariantSummary.
|
|
@@ -275,9 +274,10 @@ export interface EvaluationRequest {
|
|
|
275
274
|
bootstrap?: boolean;
|
|
276
275
|
/** --bootstrap-samples N; bootstrap 重采样次数, 默认 1000. > 10000 时 stderr 警告. */
|
|
277
276
|
bootstrapSamples?: number;
|
|
278
|
-
/** length-debias toggle. Default true
|
|
279
|
-
* CLI flag --no-debias-length flips to false (
|
|
280
|
-
* value is reflected in
|
|
277
|
+
/** length-debias toggle. Default true — judge prompt carries the length-debias
|
|
278
|
+
* instruction. CLI flag --no-debias-length flips to false (drops that instruction,
|
|
279
|
+
* the debias-off prompt variant). The active value is reflected in
|
|
280
|
+
* ReportMeta.judgePromptHash and ReportMeta.debiasMode. */
|
|
281
281
|
lengthDebias?: boolean;
|
|
282
282
|
/** hard budget caps. See EvalBudget. */
|
|
283
283
|
budget?: EvalBudget;
|
package/dist/types/judge.d.ts
CHANGED
|
@@ -18,7 +18,7 @@ export interface JudgeRuntimeEntry {
|
|
|
18
18
|
}
|
|
19
19
|
/** Per-judge ensemble entry: which judge gave what score (mean over judge-repeat if N>1). */
|
|
20
20
|
export interface EnsembleJudgeResult {
|
|
21
|
-
/** "executor:model" identifier — e.g. "claude:opus" or "openai:gpt-4o". */
|
|
21
|
+
/** "executor:model" identifier — e.g. "claude:opus" or "openai-api:gpt-4o". */
|
|
22
22
|
judge: string;
|
|
23
23
|
/** Mean score from this judge over judge-repeat calls (or single score if repeat=1). */
|
|
24
24
|
score: number;
|
package/dist/types/report.d.ts
CHANGED
|
@@ -317,7 +317,7 @@ export interface ReportMeta {
|
|
|
317
317
|
humanAgreement?: ReportHumanAgreement;
|
|
318
318
|
variantConfigs?: VariantConfig[];
|
|
319
319
|
/** Skill isolation 快照(per-variant)。
|
|
320
|
-
* key = variant name;value = allowedSkills(undefined → null,SDK 默认全发现 / [] →
|
|
320
|
+
* key = variant name;value = allowedSkills(undefined → null,SDK 默认全发现 / [] → 完全隔离;非空白名单已移除)。
|
|
321
321
|
* 跨报告对比 verdict / Δ 时,isolation 状态不一致会被 stderr warn 标"不可比"。
|
|
322
322
|
* 字段缺失意味着报告产自 之前(默认全发现,construct validity 不保证)。 */
|
|
323
323
|
skillIsolation?: Record<string, string[] | null>;
|
|
@@ -325,8 +325,6 @@ export interface ReportMeta {
|
|
|
325
325
|
run?: EvaluationRun;
|
|
326
326
|
job?: EvaluationJob;
|
|
327
327
|
gitInfo?: GitInfo | null;
|
|
328
|
-
blind?: boolean;
|
|
329
|
-
blindMap?: Record<string, string>;
|
|
330
328
|
layeredStats?: boolean;
|
|
331
329
|
/** Evolve 合并报告的原始 skill 归属。variants 会被 relabel 为 round-0/round-1,
|
|
332
330
|
* Studio skill 索引用该字段把报告归回 skill 卡片。 */
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "oh-my-knowledge",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.41.0",
|
|
4
4
|
"packageManager": "yarn@4.16.0",
|
|
5
5
|
"description": "Evaluation framework for LLM knowledge inputs — prompts, RAG corpora, skills, agent workflows. Fix the model, vary the artifact. Built-in statistical rigor: bootstrap CI, Krippendorff α, length-debias, saturation curves.",
|
|
6
6
|
"type": "module",
|
|
@@ -1,83 +0,0 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* Measure how much the judge's scores shift when the length-debias instruction
|
|
3
|
-
* is toggled.
|
|
4
|
-
*
|
|
5
|
-
* What it actually measures
|
|
6
|
-
* -------------------------
|
|
7
|
-
* Given a finished report, this command re-judges every (sample, variant) pair
|
|
8
|
-
* using the OPPOSITE length-debias setting from the original run. It then
|
|
9
|
-
* compares the two score distributions with a bootstrap CI on the mean
|
|
10
|
-
* difference.
|
|
11
|
-
*
|
|
12
|
-
* If the original report ran with debias-on (v3-cot-length), we re-judge with
|
|
13
|
-
* v2-cot (legacy). If the original ran with debias-off, we re-judge with
|
|
14
|
-
* v3-cot-length. Significant difference → the prompt change moves scores → the
|
|
15
|
-
* judge is sensitive to the length-debias instruction. That's *consistent with*
|
|
16
|
-
* length bias being present, but it doesn't prove it directly — a perfectly
|
|
17
|
-
* length-neutral judge could in principle also be sensitive to the wording for
|
|
18
|
-
* other reasons. We label the verdict accordingly.
|
|
19
|
-
*
|
|
20
|
-
* Cost
|
|
21
|
-
* ----
|
|
22
|
-
* Re-judging is a full second pass over all (sample, variant) cells. Cost
|
|
23
|
-
* doubles vs the original run. The CLI surfaces this so users running on
|
|
24
|
-
* large/expensive evaluations can opt in deliberately.
|
|
25
|
-
*/
|
|
26
|
-
import type { ExecutorFn, Report, Sample } from '../types/index.js';
|
|
27
|
-
import { type BootstrapDiffCI } from '../eval-core/bootstrap.js';
|
|
28
|
-
export interface DebiasValidateInput {
|
|
29
|
-
report: Report;
|
|
30
|
-
samples: Sample[];
|
|
31
|
-
judgeExecutor: ExecutorFn;
|
|
32
|
-
judgeModel: string;
|
|
33
|
-
/** Variant to validate. Defaults to first variant. */
|
|
34
|
-
variant?: string;
|
|
35
|
-
/** Bootstrap iterations for the diff CI. Default 1000. */
|
|
36
|
-
bootstrapSamples?: number;
|
|
37
|
-
seed?: number;
|
|
38
|
-
/** Progress hook. */
|
|
39
|
-
onProgress?: (info: {
|
|
40
|
-
sample_id: string;
|
|
41
|
-
completed: number;
|
|
42
|
-
total: number;
|
|
43
|
-
}) => void;
|
|
44
|
-
}
|
|
45
|
-
export interface DebiasValidateResult {
|
|
46
|
-
variant: string;
|
|
47
|
-
/** Original lengthDebias setting (true if debias-on at run time). */
|
|
48
|
-
originalLengthDebias: boolean;
|
|
49
|
-
/** Pairs of (originalScore, alternateScore) per sample. */
|
|
50
|
-
pairs: Array<{
|
|
51
|
-
sample_id: string;
|
|
52
|
-
originalScore: number;
|
|
53
|
-
alternateScore: number;
|
|
54
|
-
}>;
|
|
55
|
-
meanOriginal: number;
|
|
56
|
-
meanAlternate: number;
|
|
57
|
-
/** Mean of (alternate - original). Positive = alternate prompt scored higher. */
|
|
58
|
-
diffCI: BootstrapDiffCI;
|
|
59
|
-
/** Verdict in the {未检测, 弱, 中, 强} bucket plus an English shadow. */
|
|
60
|
-
verdict: {
|
|
61
|
-
zh: string;
|
|
62
|
-
en: string;
|
|
63
|
-
level: 'none' | 'weak' | 'medium' | 'strong';
|
|
64
|
-
};
|
|
65
|
-
/** Total cost burned re-judging. */
|
|
66
|
-
alternateJudgeCostUSD: number;
|
|
67
|
-
/** Sample_ids that the report had but lacked judge scores. */
|
|
68
|
-
unscored: string[];
|
|
69
|
-
/** Sample_ids in the samples file that are missing from the report. */
|
|
70
|
-
missing: string[];
|
|
71
|
-
}
|
|
72
|
-
/**
|
|
73
|
-
* Re-judge every sample of `variant` in the given report with the OPPOSITE
|
|
74
|
-
* lengthDebias setting and compute the bootstrap CI on the mean difference.
|
|
75
|
-
*
|
|
76
|
-
* The judge call uses the rubric from the samples file and the output stored
|
|
77
|
-
* in the report — we do NOT re-execute the model. Only judging is repeated.
|
|
78
|
-
*
|
|
79
|
-
* Multi-dimensional samples currently use the rubric as fallback when there's
|
|
80
|
-
* no top-level rubric. Per-dimension validation can be added later if needed.
|
|
81
|
-
*/
|
|
82
|
-
export declare function validateLengthDebias(input: DebiasValidateInput): Promise<DebiasValidateResult>;
|
|
83
|
-
export declare function formatDebiasValidate(result: DebiasValidateResult): string;
|
|
@@ -1,176 +0,0 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* Measure how much the judge's scores shift when the length-debias instruction
|
|
3
|
-
* is toggled.
|
|
4
|
-
*
|
|
5
|
-
* What it actually measures
|
|
6
|
-
* -------------------------
|
|
7
|
-
* Given a finished report, this command re-judges every (sample, variant) pair
|
|
8
|
-
* using the OPPOSITE length-debias setting from the original run. It then
|
|
9
|
-
* compares the two score distributions with a bootstrap CI on the mean
|
|
10
|
-
* difference.
|
|
11
|
-
*
|
|
12
|
-
* If the original report ran with debias-on (v3-cot-length), we re-judge with
|
|
13
|
-
* v2-cot (legacy). If the original ran with debias-off, we re-judge with
|
|
14
|
-
* v3-cot-length. Significant difference → the prompt change moves scores → the
|
|
15
|
-
* judge is sensitive to the length-debias instruction. That's *consistent with*
|
|
16
|
-
* length bias being present, but it doesn't prove it directly — a perfectly
|
|
17
|
-
* length-neutral judge could in principle also be sensitive to the wording for
|
|
18
|
-
* other reasons. We label the verdict accordingly.
|
|
19
|
-
*
|
|
20
|
-
* Cost
|
|
21
|
-
* ----
|
|
22
|
-
* Re-judging is a full second pass over all (sample, variant) cells. Cost
|
|
23
|
-
* doubles vs the original run. The CLI surfaces this so users running on
|
|
24
|
-
* large/expensive evaluations can opt in deliberately.
|
|
25
|
-
*/
|
|
26
|
-
import { llmJudge } from './judge.js';
|
|
27
|
-
import { bootstrapPairedDiffCI } from '../eval-core/bootstrap.js';
|
|
28
|
-
/**
|
|
29
|
-
* Map a bootstrap diff CI to a verdict bucket. The ranges are deliberately
|
|
30
|
-
* conservative: we only label "strong" when the CI fully sits >= |0.5| away
|
|
31
|
-
* from zero (about half a point on a 1-5 scale).
|
|
32
|
-
*/
|
|
33
|
-
function classifyVerdict(diff) {
|
|
34
|
-
if (!diff.significant) {
|
|
35
|
-
return { zh: '未检测到显著差异', en: 'no significant shift', level: 'none' };
|
|
36
|
-
}
|
|
37
|
-
const mag = Math.min(Math.abs(diff.low), Math.abs(diff.high));
|
|
38
|
-
if (mag >= 0.5) {
|
|
39
|
-
return { zh: '强差异——prompt 改动对评分影响大', en: 'strong shift', level: 'strong' };
|
|
40
|
-
}
|
|
41
|
-
if (mag >= 0.2) {
|
|
42
|
-
return { zh: '中等差异——校正对结论有实质影响', en: 'medium shift', level: 'medium' };
|
|
43
|
-
}
|
|
44
|
-
return { zh: '弱差异——显著但幅度小', en: 'weak shift', level: 'weak' };
|
|
45
|
-
}
|
|
46
|
-
/**
|
|
47
|
-
* Re-judge every sample of `variant` in the given report with the OPPOSITE
|
|
48
|
-
* lengthDebias setting and compute the bootstrap CI on the mean difference.
|
|
49
|
-
*
|
|
50
|
-
* The judge call uses the rubric from the samples file and the output stored
|
|
51
|
-
* in the report — we do NOT re-execute the model. Only judging is repeated.
|
|
52
|
-
*
|
|
53
|
-
* Multi-dimensional samples currently use the rubric as fallback when there's
|
|
54
|
-
* no top-level rubric. Per-dimension validation can be added later if needed.
|
|
55
|
-
*/
|
|
56
|
-
export async function validateLengthDebias(input) {
|
|
57
|
-
const { report, samples, judgeExecutor, judgeModel, bootstrapSamples = 1000, seed, onProgress } = input;
|
|
58
|
-
const variant = input.variant ?? report.meta.variants?.[0];
|
|
59
|
-
if (!variant)
|
|
60
|
-
throw new Error('report has no variants — nothing to validate');
|
|
61
|
-
// Detect original lengthDebias setting from meta. Default to true (v0.21+).
|
|
62
|
-
// Older reports without debiasMode set are treated as legacy (debias-off).
|
|
63
|
-
const debiasModeList = report.meta.debiasMode ?? [];
|
|
64
|
-
const originalLengthDebias = debiasModeList.includes('length');
|
|
65
|
-
const alternateLengthDebias = !originalLengthDebias;
|
|
66
|
-
const sampleById = new Map();
|
|
67
|
-
for (const s of samples)
|
|
68
|
-
sampleById.set(s.sample_id, s);
|
|
69
|
-
// Pre-pass: collect (sample_id, output, originalScore) tuples.
|
|
70
|
-
const tasks = [];
|
|
71
|
-
const unscored = [];
|
|
72
|
-
const missing = [];
|
|
73
|
-
for (const entry of report.results ?? []) {
|
|
74
|
-
const v = entry.variants?.[variant];
|
|
75
|
-
if (!v || !v.fullOutput)
|
|
76
|
-
continue;
|
|
77
|
-
const sample = sampleById.get(entry.sample_id);
|
|
78
|
-
if (!sample) {
|
|
79
|
-
missing.push(entry.sample_id);
|
|
80
|
-
continue;
|
|
81
|
-
}
|
|
82
|
-
if (typeof v.llmScore !== 'number' || v.llmScore <= 0) {
|
|
83
|
-
unscored.push(entry.sample_id);
|
|
84
|
-
continue;
|
|
85
|
-
}
|
|
86
|
-
tasks.push({ sample, output: v.fullOutput, originalScore: v.llmScore });
|
|
87
|
-
}
|
|
88
|
-
// Re-judge with the opposite debias setting. We use the simplest path:
|
|
89
|
-
// single-rubric judge. Multi-dim samples fall back to the explicit rubric
|
|
90
|
-
// string if available. Samples without a rubric are skipped — we have
|
|
91
|
-
// nothing to feed the judge.
|
|
92
|
-
const pairs = [];
|
|
93
|
-
let alternateJudgeCostUSD = 0;
|
|
94
|
-
let completed = 0;
|
|
95
|
-
for (const t of tasks) {
|
|
96
|
-
completed++;
|
|
97
|
-
onProgress?.({ sample_id: t.sample.sample_id, completed, total: tasks.length });
|
|
98
|
-
const rubric = t.sample.rubric
|
|
99
|
-
?? (t.sample.dimensions ? Object.values(t.sample.dimensions).join('\n') : '');
|
|
100
|
-
if (!rubric)
|
|
101
|
-
continue;
|
|
102
|
-
const altResult = await llmJudge({
|
|
103
|
-
output: t.output,
|
|
104
|
-
rubric,
|
|
105
|
-
prompt: t.sample.prompt,
|
|
106
|
-
executor: judgeExecutor,
|
|
107
|
-
model: judgeModel,
|
|
108
|
-
lengthDebias: alternateLengthDebias,
|
|
109
|
-
});
|
|
110
|
-
if (altResult.judgeCostUSD)
|
|
111
|
-
alternateJudgeCostUSD += altResult.judgeCostUSD;
|
|
112
|
-
if (altResult.score > 0) {
|
|
113
|
-
pairs.push({
|
|
114
|
-
sample_id: t.sample.sample_id,
|
|
115
|
-
originalScore: t.originalScore,
|
|
116
|
-
alternateScore: altResult.score,
|
|
117
|
-
});
|
|
118
|
-
}
|
|
119
|
-
}
|
|
120
|
-
const meanOriginal = avg(pairs.map((p) => p.originalScore));
|
|
121
|
-
const meanAlternate = avg(pairs.map((p) => p.alternateScore));
|
|
122
|
-
// **配对** diff CI:每个 sample 同时有 original 与 alternate prompt 两个分数(同一回答、两个 judge prompt
|
|
123
|
-
// = 配对设计)。原先拆成两数组喂独立重采样,丢弃了配对、高估方差、CI 偏宽 —— 对一个**检测**长度偏置敏感性
|
|
124
|
-
// 的工具,保守方向恰好是错的(更难检出真实偏置)。改配对:重采样 sample 下标、按 (alternate − original) 算,
|
|
125
|
-
// 保留 within-sample 相关、收紧 CI,提升对偏置的检出力。diff = b − a = alternate − original(同原约定)。
|
|
126
|
-
const diffCI = bootstrapPairedDiffCI(pairs.map((p) => ({ a: p.originalScore, b: p.alternateScore })), 0.05, bootstrapSamples, seed);
|
|
127
|
-
const verdict = classifyVerdict(diffCI);
|
|
128
|
-
return {
|
|
129
|
-
variant,
|
|
130
|
-
originalLengthDebias,
|
|
131
|
-
pairs,
|
|
132
|
-
meanOriginal: Number(meanOriginal.toFixed(3)),
|
|
133
|
-
meanAlternate: Number(meanAlternate.toFixed(3)),
|
|
134
|
-
diffCI,
|
|
135
|
-
verdict,
|
|
136
|
-
alternateJudgeCostUSD: Number(alternateJudgeCostUSD.toFixed(6)),
|
|
137
|
-
unscored,
|
|
138
|
-
missing,
|
|
139
|
-
};
|
|
140
|
-
}
|
|
141
|
-
function avg(arr) {
|
|
142
|
-
if (arr.length === 0)
|
|
143
|
-
return 0;
|
|
144
|
-
return arr.reduce((s, x) => s + x, 0) / arr.length;
|
|
145
|
-
}
|
|
146
|
-
export function formatDebiasValidate(result) {
|
|
147
|
-
const lines = [];
|
|
148
|
-
const dirOrig = result.originalLengthDebias ? 'on (v3-cot-length)' : 'off (v2-cot)';
|
|
149
|
-
const dirAlt = result.originalLengthDebias ? 'off (v2-cot)' : 'on (v3-cot-length)';
|
|
150
|
-
lines.push(`\n Length-debias 灵敏度验证 (variant: ${result.variant})\n`);
|
|
151
|
-
lines.push(` 原始 prompt: ${dirOrig}`);
|
|
152
|
-
lines.push(` 对照 prompt: ${dirAlt}`);
|
|
153
|
-
lines.push(` 用例数: ${result.pairs.length}`);
|
|
154
|
-
if (result.pairs.length === 0) {
|
|
155
|
-
lines.push(' 无可比对用例——检查报告是否含 fullOutput / rubric。');
|
|
156
|
-
return lines.join('\n');
|
|
157
|
-
}
|
|
158
|
-
lines.push(` 原均值: ${result.meanOriginal.toFixed(3)}`);
|
|
159
|
-
lines.push(` 对照均值: ${result.meanAlternate.toFixed(3)}`);
|
|
160
|
-
const ci = result.diffCI;
|
|
161
|
-
lines.push(` 差值 (alt-orig): ${ci.estimate >= 0 ? '+' : ''}${ci.estimate}`);
|
|
162
|
-
lines.push(` 95% CI: [${ci.low}, ${ci.high}] (${ci.significant ? '显著' : '不显著'})`);
|
|
163
|
-
lines.push('');
|
|
164
|
-
lines.push(` 结论: ${result.verdict.zh}`);
|
|
165
|
-
lines.push('');
|
|
166
|
-
lines.push(` 注: 该结论反映"prompt 切换是否改变评分"。差异显著 = 评分对 length-debias 指令敏感,`);
|
|
167
|
-
lines.push(` 间接支持 length bias 存在;但 prompt 文本变化也可能因其他原因影响评分。`);
|
|
168
|
-
lines.push(` 重判 cost: ${result.alternateJudgeCostUSD.toFixed(6)} USD`);
|
|
169
|
-
if (result.missing.length) {
|
|
170
|
-
lines.push(` 缺用例 ID: ${result.missing.length} 条`);
|
|
171
|
-
}
|
|
172
|
-
if (result.unscored.length) {
|
|
173
|
-
lines.push(` 无 LLM 分: ${result.unscored.length} 条`);
|
|
174
|
-
}
|
|
175
|
-
return lines.join('\n');
|
|
176
|
-
}
|