oh-my-knowledge 0.40.0 → 0.42.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +8 -4
- package/README.zh.md +8 -4
- package/dist/analysis/report-diagnostics.d.ts +8 -1
- package/dist/analysis/report-diagnostics.js +109 -1
- package/dist/assets/agent-skills/omk/references/commands.md +2 -2
- package/dist/authoring/evolver.d.ts +3 -14
- package/dist/authoring/evolver.js +1 -52
- package/dist/authoring/generator.d.ts +24 -0
- package/dist/authoring/generator.js +36 -8
- package/dist/cli/commands/eval/index.d.ts +1 -1
- package/dist/cli/commands/eval/index.js +50 -15
- package/dist/cli/commands/init.js +10 -7
- package/dist/cli/lib/cmd-flags.d.ts +0 -1
- package/dist/cli/lib/i18n-dict/init.d.ts +1 -1
- package/dist/cli/lib/i18n-dict/init.js +14 -11
- package/dist/cli/lib/i18n-dict/run.d.ts +1 -1
- package/dist/cli/lib/i18n-dict/run.js +6 -2
- package/dist/cli/lib/parse-run-config.d.ts +3 -1
- package/dist/cli/lib/parse-run-config.js +0 -2
- package/dist/eval-core/evaluation-job.d.ts +2 -2
- package/dist/eval-core/evaluation-job.js +2 -2
- package/dist/eval-core/evaluation-reporting.d.ts +0 -1
- package/dist/eval-core/evaluation-reporting.js +9 -41
- package/dist/eval-core/execution-strategy.js +3 -2
- package/dist/eval-core/holdout.d.ts +66 -0
- package/dist/eval-core/holdout.js +118 -0
- package/dist/eval-core/judge-independence.d.ts +28 -0
- package/dist/eval-core/judge-independence.js +29 -0
- package/dist/eval-core/verdict.d.ts +53 -2
- package/dist/eval-core/verdict.js +216 -16
- package/dist/eval-workflows/batch-evaluation-workflow.d.ts +1 -1
- package/dist/eval-workflows/batch-evaluation-workflow.js +1 -2
- package/dist/eval-workflows/evaluation-pipeline/report-finalize.d.ts +1 -5
- package/dist/eval-workflows/evaluation-pipeline/report-finalize.js +19 -8
- package/dist/eval-workflows/evaluation-pipeline/run-state.d.ts +2 -2
- package/dist/eval-workflows/evaluation-pipeline/run-state.js +2 -2
- package/dist/eval-workflows/evaluation-pipeline.d.ts +3 -2
- package/dist/eval-workflows/evaluation-pipeline.js +3 -4
- package/dist/eval-workflows/run-evaluation.d.ts +6 -4
- package/dist/eval-workflows/run-evaluation.js +8 -6
- package/dist/executors/claude-cli.js +5 -6
- package/dist/executors/claude-sdk.d.ts +5 -2
- package/dist/executors/claude-sdk.js +13 -8
- package/dist/executors/codex-cli.js +3 -4
- package/dist/executors/shared.d.ts +2 -0
- package/dist/executors/shared.js +15 -0
- package/dist/grading/assertions.js +6 -122
- package/dist/grading/gold-cli.js +1 -1
- package/dist/grading/human-gold.d.ts +5 -3
- package/dist/grading/human-gold.js +5 -3
- package/dist/grading/index.d.ts +4 -4
- package/dist/grading/judge.d.ts +6 -14
- package/dist/grading/judge.js +5 -88
- package/dist/inputs/eval-config.js +12 -2
- package/dist/managed/evidence.js +1 -2
- package/dist/managed/version-scores.js +1 -1
- package/dist/renderer/html-renderer.js +0 -9
- package/dist/renderer/layout.js +4 -4
- package/dist/renderer/summary.js +59 -4
- package/dist/shared/llm-prompts/debias-instructions.d.ts +4 -0
- package/dist/shared/llm-prompts/debias-instructions.js +44 -0
- package/dist/shared/llm-prompts/judge-prompts.d.ts +30 -0
- package/dist/shared/llm-prompts/judge-prompts.js +205 -0
- package/dist/shared/llm-prompts/registry.d.ts +27 -0
- package/dist/shared/llm-prompts/registry.js +69 -0
- package/dist/types/eval.d.ts +15 -8
- package/dist/types/judge.d.ts +1 -1
- package/dist/types/report.d.ts +50 -3
- package/package.json +1 -1
- package/dist/grading/debias-validate.d.ts +0 -83
- package/dist/grading/debias-validate.js +0 -176
|
@@ -9,18 +9,15 @@
|
|
|
9
9
|
* - 可选 gapReports: 文本信号(markers / hedging)始终适用,与 tool trace 无关;
|
|
10
10
|
* 带上 testSetHash 水印(spec §7.1 强制要求)
|
|
11
11
|
*
|
|
12
|
-
* 最后一步 `applyBlindMode` 是 blind 模式下的字段脱敏,放在所有 analysis 之后,
|
|
13
|
-
* 避免脱敏后字段被分析逻辑读取。
|
|
14
|
-
*
|
|
15
12
|
* 仅被 orchestrator 调用;独立拆出主要为让 orchestrator 的 try-finally 主干
|
|
16
13
|
* 看起来纯粹是「执行→收尾」时序。
|
|
17
14
|
*/
|
|
18
15
|
import { analyzeResults } from '../../analysis/report-diagnostics.js';
|
|
19
16
|
import { computeReportCoverage } from '../../analysis/coverage-analyzer.js';
|
|
20
17
|
import { computeReportGapRates } from '../../analysis/gap-analyzer.js';
|
|
21
|
-
import {
|
|
18
|
+
import { computeHoldoutBreakdown } from '../../eval-core/holdout.js';
|
|
22
19
|
import { computeTestSetHash } from './test-set-hash.js';
|
|
23
|
-
export function finalizeEvaluationReport({ report, results, artifacts, variantNames,
|
|
20
|
+
export function finalizeEvaluationReport({ report, results, artifacts, variantNames, samplesPath, samplesSourceFiles, samples, }) {
|
|
24
21
|
// pass samples so analyzeResults can populate analysis.sampleQuality
|
|
25
22
|
// (capability/difficulty/construct/provenance coverage aggregate). Without
|
|
26
23
|
// samples, analysis.sampleQuality is omitted (老报告读取仍可工作).
|
|
@@ -37,9 +34,13 @@ export function finalizeEvaluationReport({ report, results, artifacts, variantNa
|
|
|
37
34
|
// Gap rate computation runs on every successful report regardless of whether
|
|
38
35
|
// tool trace data is present — text-based signals (markers, hedging) still
|
|
39
36
|
// apply. The samples-file SHA is the mandatory watermark required by spec §7.1.
|
|
37
|
+
// The same hash watermarks the opt-in holdout breakdown below, so it is computed
|
|
38
|
+
// once and shared (both coverage-class numbers must carry the same test-set id).
|
|
39
|
+
const holdoutRatio = report.meta?.request?.holdoutRatio ?? 0;
|
|
40
40
|
const gapReports = computeReportGapRates(report.results, variantNames);
|
|
41
|
+
const needsWatermark = Object.keys(gapReports).length > 0 || holdoutRatio > 0;
|
|
42
|
+
const testSetHash = needsWatermark ? computeTestSetHash(samplesPath, samplesSourceFiles) : null;
|
|
41
43
|
if (Object.keys(gapReports).length > 0) {
|
|
42
|
-
const testSetHash = computeTestSetHash(samplesPath, samplesSourceFiles);
|
|
43
44
|
for (const variant of variantNames) {
|
|
44
45
|
const gr = gapReports[variant];
|
|
45
46
|
if (!gr)
|
|
@@ -49,8 +50,18 @@ export function finalizeEvaluationReport({ report, results, artifacts, variantNa
|
|
|
49
50
|
}
|
|
50
51
|
report.analysis.gapReports = gapReports;
|
|
51
52
|
}
|
|
52
|
-
|
|
53
|
-
|
|
53
|
+
// Opt-in train/holdout generalization breakdown (`omk eval --holdout-ratio`).
|
|
54
|
+
// Post-hoc over report.results — never perturbs the headline composite or the
|
|
55
|
+
// bootstrap CI. A large train − holdout gap is the overfitting signal the verdict
|
|
56
|
+
// overfitting gate reads (src/eval-core/verdict.ts); absent on default runs.
|
|
57
|
+
if (holdoutRatio > 0) {
|
|
58
|
+
// 切分按 samples 的稳定原始顺序(文件顺序),不依赖 report.results 的并发完成落盘顺序,
|
|
59
|
+
// 否则同批样本在不同并发/时序下 holdout 子集会漂、verdict 过拟合门控跟着漂。
|
|
60
|
+
const sampleIdOrder = samples.map((s) => s.sample_id);
|
|
61
|
+
const holdout = computeHoldoutBreakdown(report, variantNames, holdoutRatio, sampleIdOrder);
|
|
62
|
+
holdout.testSetPath = samplesPath;
|
|
63
|
+
holdout.testSetHash = testSetHash;
|
|
64
|
+
report.analysis.holdout = holdout;
|
|
54
65
|
}
|
|
55
66
|
return report;
|
|
56
67
|
}
|
|
@@ -22,7 +22,7 @@ export interface EvaluationRunState {
|
|
|
22
22
|
runningJob: EvaluationJob;
|
|
23
23
|
resolvedJobStore: JobStore | null;
|
|
24
24
|
}
|
|
25
|
-
export declare function initializeEvaluationRunState({ samplesPath, skillDir, artifacts, model, judgeModel, noJudge, executorName, judgeExecutorName, concurrency, timeoutMs, noCache,
|
|
25
|
+
export declare function initializeEvaluationRunState({ samplesPath, skillDir, artifacts, model, judgeModel, noJudge, executorName, judgeExecutorName, concurrency, timeoutMs, noCache, project, owner, tags, runId, jobStore, persistJob, repeat, holdoutRatio, batch, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget, effort, }: {
|
|
26
26
|
samplesPath: string;
|
|
27
27
|
skillDir: string;
|
|
28
28
|
artifacts: Artifact[];
|
|
@@ -34,7 +34,6 @@ export declare function initializeEvaluationRunState({ samplesPath, skillDir, ar
|
|
|
34
34
|
concurrency: number;
|
|
35
35
|
timeoutMs?: number;
|
|
36
36
|
noCache: boolean;
|
|
37
|
-
blind: boolean;
|
|
38
37
|
project?: string;
|
|
39
38
|
owner?: string;
|
|
40
39
|
tags?: string[];
|
|
@@ -42,6 +41,7 @@ export declare function initializeEvaluationRunState({ samplesPath, skillDir, ar
|
|
|
42
41
|
jobStore?: JobStore | null;
|
|
43
42
|
persistJob?: boolean;
|
|
44
43
|
repeat?: number;
|
|
44
|
+
holdoutRatio?: number;
|
|
45
45
|
batch?: boolean;
|
|
46
46
|
judgeRepeat?: number;
|
|
47
47
|
judgeModels?: import('../../types/index.js').JudgeConfig[];
|
|
@@ -14,7 +14,7 @@
|
|
|
14
14
|
import { buildEvaluationRequest, createFailedJob, createEvaluationRun, createQueuedJob, createSucceededJob, finalizeEvaluationRun, markJobRunning, failEvaluationRun, } from '../../eval-core/evaluation-job.js';
|
|
15
15
|
import { createFileJobStore } from '../../server/job-store.js';
|
|
16
16
|
import { DEFAULT_JOBS_DIR } from '../../eval-core/default-dirs.js';
|
|
17
|
-
export async function initializeEvaluationRunState({ samplesPath, skillDir, artifacts, model, judgeModel, noJudge, executorName, judgeExecutorName, concurrency, timeoutMs, noCache,
|
|
17
|
+
export async function initializeEvaluationRunState({ samplesPath, skillDir, artifacts, model, judgeModel, noJudge, executorName, judgeExecutorName, concurrency, timeoutMs, noCache, project, owner, tags, runId, jobStore, persistJob, repeat, holdoutRatio, batch, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget, effort, }) {
|
|
18
18
|
const effectiveJudges = judgeModels && judgeModels.length > 0
|
|
19
19
|
? judgeModels
|
|
20
20
|
: [{ executor: judgeExecutorName, model: judgeModel }];
|
|
@@ -29,11 +29,11 @@ export async function initializeEvaluationRunState({ samplesPath, skillDir, arti
|
|
|
29
29
|
timeoutMs,
|
|
30
30
|
noCache,
|
|
31
31
|
dryRun: false,
|
|
32
|
-
blind,
|
|
33
32
|
project,
|
|
34
33
|
owner,
|
|
35
34
|
tags,
|
|
36
35
|
repeat,
|
|
36
|
+
holdoutRatio,
|
|
37
37
|
batch,
|
|
38
38
|
judgeRepeat,
|
|
39
39
|
judgeModels: effectiveJudges,
|
|
@@ -47,7 +47,6 @@ export interface EvaluationPipelineOptions {
|
|
|
47
47
|
project?: string;
|
|
48
48
|
owner?: string;
|
|
49
49
|
tags?: string[];
|
|
50
|
-
blind?: boolean;
|
|
51
50
|
concurrency?: number;
|
|
52
51
|
timeoutMs?: number;
|
|
53
52
|
noCache?: boolean;
|
|
@@ -62,6 +61,8 @@ export interface EvaluationPipelineOptions {
|
|
|
62
61
|
layeredStats?: boolean;
|
|
63
62
|
/** 透传到 meta.request.repeat */
|
|
64
63
|
repeat?: number;
|
|
64
|
+
/** 透传到 meta.request.holdoutRatio;> 0 时 report-finalize 算 train/holdout 子集综合分。 */
|
|
65
|
+
holdoutRatio?: number;
|
|
65
66
|
/** 透传到 meta.request.batch */
|
|
66
67
|
batch?: boolean;
|
|
67
68
|
/** 透传到 meta.request.judgeRepeat 与 grade(),每条 sample × dimension judge N 次 */
|
|
@@ -90,7 +91,7 @@ export interface EvaluationPipelineOptions {
|
|
|
90
91
|
noDiagnostic?: boolean;
|
|
91
92
|
}
|
|
92
93
|
type VariantResult = import('../types/index.js').VariantResult;
|
|
93
|
-
export declare function executeEvaluationPipeline({ samplesPath, samplesBaseDir, samplesSourceFiles, skillDir, samples, tasks, artifacts, model, judgeModel, noJudge, executorName, judgeExecutorName, executor, judgeExecutor, outputDir, project, owner, tags,
|
|
94
|
+
export declare function executeEvaluationPipeline({ samplesPath, samplesBaseDir, samplesSourceFiles, skillDir, samples, tasks, artifacts, model, judgeModel, noJudge, executorName, judgeExecutorName, executor, judgeExecutor, outputDir, project, owner, tags, concurrency, timeoutMs, noCache, jobStore, persistJob, onProgress, skipConnectivity, verbose, retry, existingResults, requires: _requires, layeredStats, repeat, holdoutRatio, batch, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget, strictBaseline, runId, lang, effort, noDiagnostic, }: EvaluationPipelineOptions): Promise<{
|
|
94
95
|
report: Report;
|
|
95
96
|
filePath: string | null;
|
|
96
97
|
}>;
|
|
@@ -30,10 +30,10 @@ import { emitIsolationWarnings, emitPowerWarnings } from './evaluation-pipeline/
|
|
|
30
30
|
// 兼容 re-export:测试与 run-evaluation.ts 动态 import 仍打 evaluation-pipeline.js
|
|
31
31
|
export { buildPowerWarnings, buildIsolationWarnings } from './evaluation-pipeline/preflight-warnings.js';
|
|
32
32
|
export { _computeTestSetHashForTest } from './evaluation-pipeline/test-set-hash.js';
|
|
33
|
-
export async function executeEvaluationPipeline({ samplesPath, samplesBaseDir, samplesSourceFiles, skillDir, samples, tasks, artifacts, model, judgeModel, noJudge, executorName, judgeExecutorName, executor, judgeExecutor, outputDir = DEFAULT_OUTPUT_DIR, project, owner, tags,
|
|
33
|
+
export async function executeEvaluationPipeline({ samplesPath, samplesBaseDir, samplesSourceFiles, skillDir, samples, tasks, artifacts, model, judgeModel, noJudge, executorName, judgeExecutorName, executor, judgeExecutor, outputDir = DEFAULT_OUTPUT_DIR, project, owner, tags, concurrency = 1, timeoutMs, noCache = false, jobStore = null, persistJob = true, onProgress = null, skipConnectivity = false, verbose = false, retry = 0, existingResults,
|
|
34
34
|
// requires 现在由 runEvaluation 上游传给 doctor 处理; eval-pipeline 不再用
|
|
35
35
|
// 但保留接口字段,避免破 programmatic API (类型层面接收, 内部忽略)
|
|
36
|
-
requires: _requires, layeredStats = false, repeat, batch, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias = true, budget, strictBaseline, runId, lang = 'zh', effort, noDiagnostic, }) {
|
|
36
|
+
requires: _requires, layeredStats = false, repeat, holdoutRatio, batch, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias = true, budget, strictBaseline, runId, lang = 'zh', effort, noDiagnostic, }) {
|
|
37
37
|
const variantNames = artifacts.map((artifact) => artifact.name);
|
|
38
38
|
const runState = await initializeEvaluationRunState({
|
|
39
39
|
samplesPath,
|
|
@@ -47,7 +47,6 @@ requires: _requires, layeredStats = false, repeat, batch, judgeRepeat, judgeMode
|
|
|
47
47
|
concurrency,
|
|
48
48
|
timeoutMs,
|
|
49
49
|
noCache,
|
|
50
|
-
blind,
|
|
51
50
|
project,
|
|
52
51
|
owner,
|
|
53
52
|
tags,
|
|
@@ -55,6 +54,7 @@ requires: _requires, layeredStats = false, repeat, batch, judgeRepeat, judgeMode
|
|
|
55
54
|
jobStore,
|
|
56
55
|
persistJob,
|
|
57
56
|
repeat,
|
|
57
|
+
holdoutRatio,
|
|
58
58
|
batch,
|
|
59
59
|
judgeRepeat,
|
|
60
60
|
judgeModels,
|
|
@@ -153,7 +153,6 @@ requires: _requires, layeredStats = false, repeat, batch, judgeRepeat, judgeMode
|
|
|
153
153
|
results,
|
|
154
154
|
artifacts,
|
|
155
155
|
variantNames,
|
|
156
|
-
blind,
|
|
157
156
|
samplesPath,
|
|
158
157
|
samplesSourceFiles,
|
|
159
158
|
samples,
|
|
@@ -32,6 +32,9 @@ interface CommonEvaluationOptions {
|
|
|
32
32
|
/** --repeat N. 1 表示单次(默认); > 1 时在 runMultiple 层聚合 variance。
|
|
33
33
|
* 记入 report.meta.request.repeat 让 meta 如实反映用户输入。 */
|
|
34
34
|
repeat?: number;
|
|
35
|
+
/** --holdout-ratio R. 0 / 缺省 = 不切分(默认)。> 0 时 report-finalize 算 train/holdout
|
|
36
|
+
* 子集综合分(report.analysis.holdout),供 verdict 过拟合门控读取。 */
|
|
37
|
+
holdoutRatio?: number;
|
|
35
38
|
/** --batch 模式标记, true 表示当前评测是 skill batch 流程。
|
|
36
39
|
* 记入 report.meta.request.batch。 */
|
|
37
40
|
batch?: boolean;
|
|
@@ -46,7 +49,7 @@ interface CommonEvaluationOptions {
|
|
|
46
49
|
bootstrap?: boolean;
|
|
47
50
|
/** --bootstrap-samples N. Default 1000. */
|
|
48
51
|
bootstrapSamples?: number;
|
|
49
|
-
/** length-debias toggle. Default true (
|
|
52
|
+
/** length-debias toggle. Default true (length-debias instruction on).
|
|
50
53
|
* CLI passes false when --no-debias-length is set. */
|
|
51
54
|
lengthDebias?: boolean;
|
|
52
55
|
/** hard budget caps. */
|
|
@@ -64,7 +67,6 @@ export interface RunEvaluationOptions extends CommonEvaluationOptions {
|
|
|
64
67
|
skillDir: string;
|
|
65
68
|
variantSpecs?: VariantSpec[];
|
|
66
69
|
dryRun?: boolean;
|
|
67
|
-
blind?: boolean;
|
|
68
70
|
retry?: number;
|
|
69
71
|
resume?: string;
|
|
70
72
|
/** Explicit persisted run id. Used by batch workflows that need stable child ids. */
|
|
@@ -118,7 +120,7 @@ export interface DryRunReport extends DryRunBase {
|
|
|
118
120
|
samplesPath: string;
|
|
119
121
|
tasks: DryRunTask[];
|
|
120
122
|
}
|
|
121
|
-
export declare function runEvaluation({ samplesPath, skillDir, variantSpecs, model, outputDir, project, owner, tags, noJudge, dryRun,
|
|
123
|
+
export declare function runEvaluation({ samplesPath, skillDir, variantSpecs, model, outputDir, project, owner, tags, noJudge, dryRun, concurrency, timeoutMs, noCache, executorName, jobStore, persistJob, onProgress, skipConnectivity, skipDoctor, lang, mcpConfig, verbose, retry, resume, runId, layeredStats, repeat, holdoutRatio, batch, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget, strictBaseline, effort, noDiagnostic, }: RunEvaluationOptions): Promise<{
|
|
122
124
|
report: Report | DryRunReport;
|
|
123
125
|
filePath: string | null;
|
|
124
126
|
}>;
|
|
@@ -134,7 +136,7 @@ export interface DryRunBatchReport extends DryRunBase {
|
|
|
134
136
|
artifacts: DryRunBatchSkill[];
|
|
135
137
|
}
|
|
136
138
|
export declare function buildVarianceData(runs: Report[], bootstrapSamples?: number, seed?: number): VarianceData | null;
|
|
137
|
-
export declare function runBatchEvaluation({ skillDir, model, outputDir, project, owner, tags, noJudge, dryRun, concurrency, timeoutMs, executorName, jobStore, persistJob, onProgress, onSkillProgress, skipConnectivity, skipDoctor, lang, mcpConfig, verbose, repeat, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget, noCache, strictBaseline, variantAllowedSkills, }: RunBatchEvaluationOptions): Promise<{
|
|
139
|
+
export declare function runBatchEvaluation({ skillDir, model, outputDir, project, owner, tags, noJudge, dryRun, concurrency, timeoutMs, executorName, jobStore, persistJob, onProgress, onSkillProgress, skipConnectivity, skipDoctor, lang, mcpConfig, verbose, repeat, holdoutRatio, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget, noCache, strictBaseline, variantAllowedSkills, }: RunBatchEvaluationOptions): Promise<{
|
|
138
140
|
report: BatchEvaluationReport | DryRunBatchReport;
|
|
139
141
|
filePath: string | null;
|
|
140
142
|
}>;
|
|
@@ -8,7 +8,7 @@ import { buildDryRunTaskReport, prepareEvaluationRun, } from './evaluation-prepa
|
|
|
8
8
|
import { executeEvaluationPipeline } from './evaluation-pipeline.js';
|
|
9
9
|
import { findSaturationPoint } from '../analysis/saturation.js';
|
|
10
10
|
import { bootstrapMeanCI, DEFAULT_BOOTSTRAP_ALPHA, DEFAULT_BOOTSTRAP_SAMPLES } from '../eval-core/bootstrap.js';
|
|
11
|
-
export async function runEvaluation({ samplesPath, skillDir, variantSpecs = [], model = DEFAULT_MODEL, outputDir = DEFAULT_OUTPUT_DIR, project, owner, tags, noJudge = false, dryRun = false,
|
|
11
|
+
export async function runEvaluation({ samplesPath, skillDir, variantSpecs = [], model = DEFAULT_MODEL, outputDir = DEFAULT_OUTPUT_DIR, project, owner, tags, noJudge = false, dryRun = false, concurrency = 1, timeoutMs, noCache = false, executorName = 'claude', jobStore = null, persistJob = true, onProgress = null, skipConnectivity = false, skipDoctor = false, lang = 'zh', mcpConfig, verbose = false, retry = 0, resume, runId, layeredStats = false, repeat, holdoutRatio, batch, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget, strictBaseline, effort, noDiagnostic, }) {
|
|
12
12
|
// Unified judgeModels → derive single-judge fields for downstream pipeline / grading
|
|
13
13
|
// (which still operate on string `judgeModel` + `judgeExecutorName` fields per call).
|
|
14
14
|
const effectiveJudgeModels = judgeModels && judgeModels.length > 0
|
|
@@ -152,7 +152,6 @@ export async function runEvaluation({ samplesPath, skillDir, variantSpecs = [],
|
|
|
152
152
|
project,
|
|
153
153
|
owner,
|
|
154
154
|
tags,
|
|
155
|
-
blind,
|
|
156
155
|
concurrency,
|
|
157
156
|
timeoutMs,
|
|
158
157
|
noCache,
|
|
@@ -166,6 +165,7 @@ export async function runEvaluation({ samplesPath, skillDir, variantSpecs = [],
|
|
|
166
165
|
requires,
|
|
167
166
|
layeredStats,
|
|
168
167
|
repeat,
|
|
168
|
+
holdoutRatio,
|
|
169
169
|
batch,
|
|
170
170
|
judgeRepeat,
|
|
171
171
|
judgeModels,
|
|
@@ -367,7 +367,7 @@ export function buildVarianceData(runs, bootstrapSamples = DEFAULT_BOOTSTRAP_SAM
|
|
|
367
367
|
...(saturation ? { saturation } : {}),
|
|
368
368
|
};
|
|
369
369
|
}
|
|
370
|
-
export async function runBatchEvaluation({ skillDir, model = DEFAULT_MODEL, outputDir = DEFAULT_OUTPUT_DIR, project, owner, tags, noJudge = false, dryRun = false, concurrency = 1, timeoutMs, executorName = 'claude', jobStore = null, persistJob = true, onProgress = null, onSkillProgress = null, skipConnectivity = false, skipDoctor = false, lang = 'zh', mcpConfig, verbose = false, repeat, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget, noCache = false, strictBaseline, variantAllowedSkills, }) {
|
|
370
|
+
export async function runBatchEvaluation({ skillDir, model = DEFAULT_MODEL, outputDir = DEFAULT_OUTPUT_DIR, project, owner, tags, noJudge = false, dryRun = false, concurrency = 1, timeoutMs, executorName = 'claude', jobStore = null, persistJob = true, onProgress = null, onSkillProgress = null, skipConnectivity = false, skipDoctor = false, lang = 'zh', mcpConfig, verbose = false, repeat, holdoutRatio, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget, noCache = false, strictBaseline, variantAllowedSkills, }) {
|
|
371
371
|
// Same unified judge derivation as runEvaluation (downstream pipeline / report build
|
|
372
372
|
// still uses single judgeModel + judgeExecutorName per call).
|
|
373
373
|
const effectiveJudgeModels = judgeModels && judgeModels.length > 0
|
|
@@ -486,12 +486,14 @@ export async function runBatchEvaluation({ skillDir, model = DEFAULT_MODEL, outp
|
|
|
486
486
|
strictBaseline,
|
|
487
487
|
variantAllowedSkills,
|
|
488
488
|
runSingleEvaluation: async (options) => {
|
|
489
|
-
// repeat > 1 时走 runMultiple 做 variance; batch=true 标记让 meta.request
|
|
489
|
+
// repeat > 1 时走 runMultiple 做 variance; batch=true 标记让 meta.request 如实反映。
|
|
490
|
+
// holdoutRatio 显式转发(executeBatchEvaluationRuns 不一定把它塞进 options),否则
|
|
491
|
+
// batch 子报告 meta.request.holdoutRatio 丢失、过拟合门控对 batch 静默失效。
|
|
490
492
|
if (repeat && repeat > 1) {
|
|
491
|
-
const multi = await runMultiple({ ...options, repeat, batch: true, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget });
|
|
493
|
+
const multi = await runMultiple({ ...options, repeat, holdoutRatio, batch: true, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget });
|
|
492
494
|
return { report: multi.report, filePath: multi.filePath };
|
|
493
495
|
}
|
|
494
|
-
const result = await runEvaluation({ ...options, batch: true, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget });
|
|
496
|
+
const result = await runEvaluation({ ...options, batch: true, holdoutRatio, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget });
|
|
495
497
|
return { report: result.report, filePath: result.filePath };
|
|
496
498
|
},
|
|
497
499
|
});
|
|
@@ -2,20 +2,19 @@ import { extractAgentTrace, isClaudeSdkResultMessage } from './claude-sdk-trace.
|
|
|
2
2
|
import { buildExecEnv, DEFAULT_TIMEOUT_MS, errorMessage, interruptedExecResult, MAX_BUFFER, spawnWithSigintPropagation, timeoutExecResult, } from './shared.js';
|
|
3
3
|
import { materializeForCliConfigDir } from '../eval-core/mocks-runtime.js';
|
|
4
4
|
// claude CLI 用 `--disable-slash-commands` (文档:"Disable all skills") +
|
|
5
|
-
// `--disallowedTools Skill` 实现与 SDK
|
|
6
|
-
//
|
|
5
|
+
// `--disallowedTools Skill` 实现与 SDK 等价的完全隔离。非空 skill 白名单不再支持
|
|
6
|
+
// (它无法真正隔离,见 claude-sdk.ts buildSdkIsolationOptions),所以 [name1, ...] 必须 throw。
|
|
7
7
|
//
|
|
8
8
|
// undefined → 不传任何 flag(原行为,全发现)
|
|
9
9
|
// [] → --disable-slash-commands + --disallowedTools Skill
|
|
10
10
|
// (main session skill discovery + subagent Skill 工具调用都堵)
|
|
11
|
-
// [...] (length > 0) → throw
|
|
11
|
+
// [...] (length > 0) → throw,非空白名单已移除
|
|
12
12
|
function applySkillIsolationToCliArgs(args, allowedSkills) {
|
|
13
13
|
if (allowedSkills === undefined)
|
|
14
14
|
return;
|
|
15
15
|
if (allowedSkills.length > 0) {
|
|
16
|
-
throw new Error(`
|
|
17
|
-
+ ` 仅支持 [](映射为 --disable-slash-commands + --disallowedTools Skill)或 undefined(
|
|
18
|
-
+ ` 精确白名单请改用 --executor claude-sdk(SDK skills option pass-through)。`);
|
|
16
|
+
throw new Error(`skill 白名单(allowedSkills=${JSON.stringify(allowedSkills)})不再支持:非空白名单无法真正隔离。\n`
|
|
17
|
+
+ ` 仅支持 [](映射为 --disable-slash-commands + --disallowedTools Skill,全封死)或 undefined(不隔离)。`);
|
|
19
18
|
}
|
|
20
19
|
// 完全隔离:双堵 main session skill 发现 + subagent Skill 工具
|
|
21
20
|
args.push('--disable-slash-commands', '--disallowedTools', 'Skill');
|
|
@@ -3,8 +3,11 @@ import type { ExecResult, ExecutorInput } from '../types/index.js';
|
|
|
3
3
|
* Map ExecutorInput.allowedSkills to SDK query options for skill isolation.
|
|
4
4
|
* undefined → {} (SDK default: full ~/.claude/skills/ discovery)
|
|
5
5
|
* [] → { skills: [], disallowedTools: ['Skill'] } (main session + subagent 双堵)
|
|
6
|
-
* [...] →
|
|
7
|
-
*
|
|
6
|
+
* [...] → throw(非空 skill 白名单不再支持:它从不能真隔离 —— 主会话 skill 发现虽被
|
|
7
|
+
* `skills:[...]` 收窄,但子代理 Skill 工具与 cwd 文件系统两条 channel 封不住,
|
|
8
|
+
* 会产出看着干净、实则被白名单外 skill 污染的报告。隔离只留两档:undefined
|
|
9
|
+
* (不隔离)与 [](全封死),与 claude-cli / codex-cli 一致;多 skill 组合实验
|
|
10
|
+
* 请控制评测环境而非靠白名单。)
|
|
8
11
|
*
|
|
9
12
|
* Exported for unit tests to lock the option-shape contract.
|
|
10
13
|
*/
|
|
@@ -9,8 +9,11 @@ let sdkQuery = null;
|
|
|
9
9
|
* Map ExecutorInput.allowedSkills to SDK query options for skill isolation.
|
|
10
10
|
* undefined → {} (SDK default: full ~/.claude/skills/ discovery)
|
|
11
11
|
* [] → { skills: [], disallowedTools: ['Skill'] } (main session + subagent 双堵)
|
|
12
|
-
* [...] →
|
|
13
|
-
*
|
|
12
|
+
* [...] → throw(非空 skill 白名单不再支持:它从不能真隔离 —— 主会话 skill 发现虽被
|
|
13
|
+
* `skills:[...]` 收窄,但子代理 Skill 工具与 cwd 文件系统两条 channel 封不住,
|
|
14
|
+
* 会产出看着干净、实则被白名单外 skill 污染的报告。隔离只留两档:undefined
|
|
15
|
+
* (不隔离)与 [](全封死),与 claude-cli / codex-cli 一致;多 skill 组合实验
|
|
16
|
+
* 请控制评测环境而非靠白名单。)
|
|
14
17
|
*
|
|
15
18
|
* Exported for unit tests to lock the option-shape contract.
|
|
16
19
|
*/
|
|
@@ -19,7 +22,8 @@ export function buildSdkIsolationOptions(allowedSkills) {
|
|
|
19
22
|
return {};
|
|
20
23
|
if (allowedSkills.length === 0)
|
|
21
24
|
return { skills: allowedSkills, disallowedTools: ['Skill'] };
|
|
22
|
-
|
|
25
|
+
throw new Error(`skill 白名单(allowedSkills=${JSON.stringify(allowedSkills)})不再支持:非空白名单无法真正隔离`
|
|
26
|
+
+ `(子代理 Skill 工具 + cwd 文件系统两条 channel 封不住)。仅支持 [](全封死)或 undefined(不隔离)。`);
|
|
23
27
|
}
|
|
24
28
|
async function getSdkQuery() {
|
|
25
29
|
if (!sdkQuery) {
|
|
@@ -29,6 +33,12 @@ async function getSdkQuery() {
|
|
|
29
33
|
return sdkQuery;
|
|
30
34
|
}
|
|
31
35
|
export async function claudeSdkExecutor({ model, system, prompt, cwd, skillDir, timeoutMs = DEFAULT_TIMEOUT_MS, verbose = false, allowedSkills, mocks, mocksBaseDir, mocksStrict, lean, effort }) {
|
|
36
|
+
// 隔离选项在 timer / try 之前解析:非空 allowedSkills(不再支持的 skill 白名单)必须在这里
|
|
37
|
+
// fail-fast 抛错,而不是被下面的 catch 吞成 ok:false 的 ExecResult(那会把配置错误伪装成
|
|
38
|
+
// 每个 sample 执行失败、产出全失败报告,与 claude-cli / codex 的硬抛口径不一致)。lean 仍
|
|
39
|
+
// 覆盖为硬堵,但 buildSdkIsolationOptions 先跑一遍,故 lean 也无法绕过非空校验。
|
|
40
|
+
const baseIsolationOpts = buildSdkIsolationOptions(allowedSkills);
|
|
41
|
+
const isolationOpts = lean ? { skills: [], disallowedTools: ['*'] } : baseIsolationOpts;
|
|
32
42
|
const start = Date.now();
|
|
33
43
|
const abortController = new AbortController();
|
|
34
44
|
const timer = setTimeout(() => abortController.abort(), timeoutMs);
|
|
@@ -45,11 +55,6 @@ export async function claudeSdkExecutor({ model, system, prompt, cwd, skillDir,
|
|
|
45
55
|
const mockStatsOf = () => hookHandle ? { ...hookHandle.stats } : undefined;
|
|
46
56
|
try {
|
|
47
57
|
const query = await getSdkQuery();
|
|
48
|
-
// lean 模式:纯文本生成路径,不需要工具循环 / skill 发现,
|
|
49
|
-
// 用 disallowedTools:['*'] + skills:[] 直接堵住,优先级高于 isolationOpts。
|
|
50
|
-
const isolationOpts = lean
|
|
51
|
-
? { skills: [], disallowedTools: ['*'] }
|
|
52
|
-
: buildSdkIsolationOptions(allowedSkills);
|
|
53
58
|
// effort:lean 强制 'low'(生成路径不需要思考),否则透传调用方传入。
|
|
54
59
|
// SDK 暴露 EffortLevel = 'low' | 'medium' | 'high' | 'xhigh' | 'max',直接对应。
|
|
55
60
|
const effectiveEffort = lean ? 'low' : effort;
|
|
@@ -11,14 +11,13 @@ import { buildExecEnv, DEFAULT_TIMEOUT_MS, errorMessage, interruptedExecResult,
|
|
|
11
11
|
// undefined → 不传 -C(原行为,看 cwd 里有什么 codex 自己决定)
|
|
12
12
|
// [] → 必须提供 cwd 非空(否则 throw),caller 应传一个
|
|
13
13
|
// isolated 空目录(如 ~/.oh-my-knowledge/state/isolated-cwd/)
|
|
14
|
-
// [...] (length>0) → throw
|
|
14
|
+
// [...] (length>0) → throw,非空白名单已移除(无法真正隔离)
|
|
15
15
|
export function isolateCodexCwd(allowedSkills, cwd, executorName = 'codex-cli') {
|
|
16
16
|
if (allowedSkills === undefined)
|
|
17
17
|
return;
|
|
18
18
|
if (allowedSkills.length > 0) {
|
|
19
|
-
throw new Error(
|
|
20
|
-
+ ` 仅支持 [](强制 cwd 隔离,需提供 cwd
|
|
21
|
-
+ ` codex CLI 无 partial 白名单 flag,请改用其他 executor 或显式 cwd 隔离。`);
|
|
19
|
+
throw new Error(`skill 白名单(allowedSkills=${JSON.stringify(allowedSkills)})不再支持:非空白名单无法真正隔离。\n`
|
|
20
|
+
+ ` 仅支持 [](强制 cwd 隔离,需提供 cwd 非空,全封死)或 undefined(不隔离)。`);
|
|
22
21
|
}
|
|
23
22
|
// allowedSkills === [] 时必须有 cwd(channel 3 cwd 隔离是 codex 唯一 channel)
|
|
24
23
|
if (!cwd) {
|
|
@@ -5,6 +5,8 @@ export declare const DEFAULT_MODEL = "sonnet";
|
|
|
5
5
|
export declare const JUDGE_MODEL = "haiku";
|
|
6
6
|
export declare const DEFAULT_TIMEOUT_MS = 600000;
|
|
7
7
|
export declare const MAX_BUFFER: number;
|
|
8
|
+
export type ExecutorVendor = 'anthropic' | 'openai' | 'google' | 'unknown';
|
|
9
|
+
export declare function executorVendor(executor: string): ExecutorVendor;
|
|
8
10
|
export interface TokenUsage {
|
|
9
11
|
input_tokens?: number;
|
|
10
12
|
output_tokens?: number;
|
package/dist/executors/shared.js
CHANGED
|
@@ -9,6 +9,21 @@ export const DEFAULT_MODEL = 'sonnet';
|
|
|
9
9
|
export const JUDGE_MODEL = 'haiku';
|
|
10
10
|
export const DEFAULT_TIMEOUT_MS = 600_000;
|
|
11
11
|
export const MAX_BUFFER = 10 * 1024 * 1024;
|
|
12
|
+
// 把 executor 名归到模型厂商家族,用于评委独立性判断(同厂商评委 = 自我偏好敞口)。
|
|
13
|
+
// codex 是 OpenAI 的代码模型,故 codex* 归 OpenAI。自定义 script executor 无法归类 → unknown
|
|
14
|
+
// (调用方据此跳过判断,不误报)。
|
|
15
|
+
const EXECUTOR_VENDOR = {
|
|
16
|
+
claude: 'anthropic',
|
|
17
|
+
'claude-sdk': 'anthropic',
|
|
18
|
+
'anthropic-api': 'anthropic',
|
|
19
|
+
codex: 'openai',
|
|
20
|
+
'codex-sdk': 'openai',
|
|
21
|
+
'openai-api': 'openai',
|
|
22
|
+
gemini: 'google',
|
|
23
|
+
};
|
|
24
|
+
export function executorVendor(executor) {
|
|
25
|
+
return EXECUTOR_VENDOR[executor] ?? 'unknown';
|
|
26
|
+
}
|
|
12
27
|
export function asErrorLike(err) {
|
|
13
28
|
return typeof err === 'object' && err !== null ? err : {};
|
|
14
29
|
}
|
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
import { resolve } from 'node:path';
|
|
2
2
|
import _Ajv from 'ajv';
|
|
3
3
|
import { ASSERTION_LAYER } from './layered-scores.js';
|
|
4
|
+
import { buildSemanticSimilarityPrompt, SEMANTIC_SIMILARITY_SYSTEM, buildRagJudgePrompt } from '../shared/llm-prompts/judge-prompts.js';
|
|
4
5
|
const Ajv = _Ajv.default ?? _Ajv;
|
|
5
6
|
const ajv = new Ajv();
|
|
6
7
|
const CUSTOM_ASSERTION_TIMEOUT_MS = 30_000;
|
|
@@ -332,24 +333,10 @@ export async function runAsyncAssertions(output, assertions, { executor, judgeMo
|
|
|
332
333
|
let message = '';
|
|
333
334
|
if (assertion.type === 'semantic_similarity') {
|
|
334
335
|
const reference = assertion.reference || '';
|
|
335
|
-
const judgePrompt = [
|
|
336
|
-
'请判断以下两段文本的语义相似度。',
|
|
337
|
-
'',
|
|
338
|
-
'## 参考文本',
|
|
339
|
-
reference,
|
|
340
|
-
'',
|
|
341
|
-
'## 待评估文本',
|
|
342
|
-
output,
|
|
343
|
-
'',
|
|
344
|
-
'请返回 JSON(不要包含 markdown 代码块标记):',
|
|
345
|
-
'{"score": <1-5的整数>, "reason": "<简短理由>"}',
|
|
346
|
-
'',
|
|
347
|
-
'评分:1=完全无关, 2=略有关联, 3=部分相似, 4=大致相同, 5=高度一致',
|
|
348
|
-
].join('\n');
|
|
349
336
|
const result = await executor({
|
|
350
337
|
model: judgeModel,
|
|
351
|
-
system:
|
|
352
|
-
prompt:
|
|
338
|
+
system: SEMANTIC_SIMILARITY_SYSTEM,
|
|
339
|
+
prompt: buildSemanticSimilarityPrompt(reference, output),
|
|
353
340
|
});
|
|
354
341
|
asyncCostUSD += result.costUSD || 0;
|
|
355
342
|
if (result.costReportedByExecutor === false)
|
|
@@ -423,40 +410,6 @@ export async function runAsyncAssertions(output, assertions, { executor, judgeMo
|
|
|
423
410
|
...(anyCostUnreported && { judgeCostReportedByExecutor: false }),
|
|
424
411
|
};
|
|
425
412
|
}
|
|
426
|
-
// ===========================================================================
|
|
427
|
-
// RAG-specific judge metrics
|
|
428
|
-
// ===========================================================================
|
|
429
|
-
//
|
|
430
|
-
// Three metrics, all running through the LLM judge:
|
|
431
|
-
//
|
|
432
|
-
// - faithfulness: does the output's content stay grounded in the
|
|
433
|
-
// reference context? Anti-hallucination check.
|
|
434
|
-
// - answer_relevancy: does the output directly answer the user's question?
|
|
435
|
-
// Catches verbose dodges and topic drift.
|
|
436
|
-
// - context_recall: are the key facts from the gold context actually
|
|
437
|
-
// used in the output? Catches retrieved-but-ignored
|
|
438
|
-
// context (a common RAG bug).
|
|
439
|
-
//
|
|
440
|
-
// Implementation notes:
|
|
441
|
-
//
|
|
442
|
-
// 1. Each prompt is a SINGLE-CALL judge (1-5 score) rather than the multi-step
|
|
443
|
-
// statement-decomposition that RAGAS uses. This is honest tradeoff: simpler,
|
|
444
|
-
// faster, less rigorous than RAGAS but consistent with omk's other LLM-judge
|
|
445
|
-
// assertions. Users who need RAGAS-grade decomposition can drop down to a
|
|
446
|
-
// custom assertion.
|
|
447
|
-
// 2. The prompt includes the SAME length-debias paragraph as the main judge
|
|
448
|
-
// prompt (v3-cot-length) — output verbosity is not a quality signal here
|
|
449
|
-
// either. This is the "auto-inherit length-debias" claim from the plan.
|
|
450
|
-
// 3. Reference resolution:
|
|
451
|
-
// faithfulness: sample.context (or assertion.reference override)
|
|
452
|
-
// context_recall: assertion.reference (or sample.context fallback)
|
|
453
|
-
// answer_relevancy: sample.prompt — no reference needed
|
|
454
|
-
// 4. Threshold defaults to 3 (same as semantic_similarity). User can override.
|
|
455
|
-
const RAG_LENGTH_DEBIAS = [
|
|
456
|
-
'## 重要:长度不是质量信号',
|
|
457
|
-
'评分时聚焦内容实质,不要因输出更长就给更高分。',
|
|
458
|
-
'简洁正确的回答与冗长正确的回答应得相同分数。',
|
|
459
|
-
].join('\n');
|
|
460
413
|
async function runRagJudge(assertion, output, sample, executor, judgeModel) {
|
|
461
414
|
const threshold = assertion.threshold ?? 3;
|
|
462
415
|
let prompt;
|
|
@@ -466,55 +419,10 @@ async function runRagJudge(assertion, output, sample, executor, judgeModel) {
|
|
|
466
419
|
if (!context) {
|
|
467
420
|
return { passed: false, message: 'faithfulness: 缺少 sample.context 或 assertion.reference', costUSD: 0 };
|
|
468
421
|
}
|
|
469
|
-
system = '
|
|
470
|
-
prompt = [
|
|
471
|
-
'请判断"待评估输出"中的事实性陈述是否被"参考 context"支持。',
|
|
472
|
-
'',
|
|
473
|
-
'## 参考 context',
|
|
474
|
-
context,
|
|
475
|
-
'',
|
|
476
|
-
'## 待评估输出',
|
|
477
|
-
output,
|
|
478
|
-
'',
|
|
479
|
-
RAG_LENGTH_DEBIAS,
|
|
480
|
-
'',
|
|
481
|
-
'## 评分流程',
|
|
482
|
-
'1. 列出待评估输出中所有事实性陈述',
|
|
483
|
-
'2. 逐条判断是否能在 context 中找到支持',
|
|
484
|
-
'3. 给出 1-5 分:',
|
|
485
|
-
' 5 = 全部陈述都有 context 支持,无编造',
|
|
486
|
-
' 4 = 多数有支持,有 1-2 处不重要的编造',
|
|
487
|
-
' 3 = 一半有支持',
|
|
488
|
-
' 2 = 多数无支持',
|
|
489
|
-
' 1 = 完全编造或与 context 矛盾',
|
|
490
|
-
'',
|
|
491
|
-
'请返回 JSON(不要 markdown 代码块):',
|
|
492
|
-
'{"score": <1-5的整数>, "reason": "<简短理由>"}',
|
|
493
|
-
].join('\n');
|
|
422
|
+
({ system, prompt } = buildRagJudgePrompt('faithfulness', { output, context }));
|
|
494
423
|
}
|
|
495
424
|
else if (assertion.type === 'answer_relevancy') {
|
|
496
|
-
system = '
|
|
497
|
-
prompt = [
|
|
498
|
-
'请判断"AI 输出"是否直接、切题地回答了"用户问题"。',
|
|
499
|
-
'',
|
|
500
|
-
'## 用户问题',
|
|
501
|
-
sample.prompt,
|
|
502
|
-
'',
|
|
503
|
-
'## AI 输出',
|
|
504
|
-
output,
|
|
505
|
-
'',
|
|
506
|
-
RAG_LENGTH_DEBIAS,
|
|
507
|
-
'',
|
|
508
|
-
'## 评分',
|
|
509
|
-
'5 = 完整切题回答,无冗余无遗漏',
|
|
510
|
-
'4 = 切题但有少量冗余或小遗漏',
|
|
511
|
-
'3 = 部分切题,部分跑题或避而不答',
|
|
512
|
-
'2 = 大部分跑题',
|
|
513
|
-
'1 = 完全跑题或拒答',
|
|
514
|
-
'',
|
|
515
|
-
'请返回 JSON(不要 markdown 代码块):',
|
|
516
|
-
'{"score": <1-5的整数>, "reason": "<简短理由>"}',
|
|
517
|
-
].join('\n');
|
|
425
|
+
({ system, prompt } = buildRagJudgePrompt('answer_relevancy', { output, question: sample.prompt }));
|
|
518
426
|
}
|
|
519
427
|
else {
|
|
520
428
|
// context_recall
|
|
@@ -522,31 +430,7 @@ async function runRagJudge(assertion, output, sample, executor, judgeModel) {
|
|
|
522
430
|
if (!reference) {
|
|
523
431
|
return { passed: false, message: 'context_recall: 缺少 assertion.reference 或 sample.context', costUSD: 0 };
|
|
524
432
|
}
|
|
525
|
-
system = '
|
|
526
|
-
prompt = [
|
|
527
|
-
'请判断"参考 gold"中的关键事实在"AI 输出"中被覆盖的程度。',
|
|
528
|
-
'',
|
|
529
|
-
'## 参考 gold',
|
|
530
|
-
reference,
|
|
531
|
-
'',
|
|
532
|
-
'## AI 输出',
|
|
533
|
-
output,
|
|
534
|
-
'',
|
|
535
|
-
RAG_LENGTH_DEBIAS,
|
|
536
|
-
'',
|
|
537
|
-
'## 评分流程',
|
|
538
|
-
'1. 列出参考中的关键事实(忽略修饰性内容)',
|
|
539
|
-
'2. 检查每条是否在输出中被提及/使用',
|
|
540
|
-
'3. 给出 1-5 分:',
|
|
541
|
-
' 5 = 全部关键事实被覆盖',
|
|
542
|
-
' 4 = 大部分覆盖,缺 1-2 条次要事实',
|
|
543
|
-
' 3 = 一半覆盖',
|
|
544
|
-
' 2 = 仅覆盖少量',
|
|
545
|
-
' 1 = 完全未覆盖',
|
|
546
|
-
'',
|
|
547
|
-
'请返回 JSON(不要 markdown 代码块):',
|
|
548
|
-
'{"score": <1-5的整数>, "reason": "<简短理由>"}',
|
|
549
|
-
].join('\n');
|
|
433
|
+
({ system, prompt } = buildRagJudgePrompt('context_recall', { output, reference }));
|
|
550
434
|
}
|
|
551
435
|
const result = await executor({ model: judgeModel, system, prompt });
|
|
552
436
|
const reported = result.costReportedByExecutor === false ? { costReportedByExecutor: false } : {};
|
package/dist/grading/gold-cli.js
CHANGED
|
@@ -93,7 +93,7 @@ export function formatGoldCompare(result, gold) {
|
|
|
93
93
|
lines.push(` 报告缺失: ${result.missing.join(', ')}`);
|
|
94
94
|
return lines.join('\n');
|
|
95
95
|
}
|
|
96
|
-
lines.push(` Krippendorff α: ${fmt(a.alpha)} (
|
|
96
|
+
lines.push(` Krippendorff α: ${fmt(a.alpha)} (主指标,区间加权)`);
|
|
97
97
|
lines.push(` α 95% CI: [${fmt(a.alphaCI.low)}, ${fmt(a.alphaCI.high)}]`);
|
|
98
98
|
lines.push(` 加权 κ: ${fmt(a.weightedKappa)} (副指标)`);
|
|
99
99
|
lines.push(` Pearson r: ${fmt(a.pearson)} (仅查 rank order)`);
|
|
@@ -15,9 +15,11 @@
|
|
|
15
15
|
*
|
|
16
16
|
* Three metrics are exported:
|
|
17
17
|
*
|
|
18
|
-
* - **Krippendorff's α (interval)** — primary. Distribution-free,
|
|
19
|
-
*
|
|
20
|
-
*
|
|
18
|
+
* - **Krippendorff's α (interval weights)** — primary. Distribution-free,
|
|
19
|
+
* doesn't assume coders are exchangeable (good fit when one "coder" is a model
|
|
20
|
+
* and the other a human annotator). Uses interval distance δ²=(c−k)² — a
|
|
21
|
+
* defensible choice for 1-5 Likert; an ordinal-distance variant would change α
|
|
22
|
+
* (BREAKING-COMPARABILITY) and is not implemented here.
|
|
21
23
|
* - **Quadratic-weighted Cohen's κ** — secondary. Familiar to many readers,
|
|
22
24
|
* useful as a sanity check. Reports lower than α when marginals diverge.
|
|
23
25
|
* - **Pearson r** — tertiary. Captures rank-order agreement only; doesn't
|
|
@@ -15,9 +15,11 @@
|
|
|
15
15
|
*
|
|
16
16
|
* Three metrics are exported:
|
|
17
17
|
*
|
|
18
|
-
* - **Krippendorff's α (interval)** — primary. Distribution-free,
|
|
19
|
-
*
|
|
20
|
-
*
|
|
18
|
+
* - **Krippendorff's α (interval weights)** — primary. Distribution-free,
|
|
19
|
+
* doesn't assume coders are exchangeable (good fit when one "coder" is a model
|
|
20
|
+
* and the other a human annotator). Uses interval distance δ²=(c−k)² — a
|
|
21
|
+
* defensible choice for 1-5 Likert; an ordinal-distance variant would change α
|
|
22
|
+
* (BREAKING-COMPARABILITY) and is not implemented here.
|
|
21
23
|
* - **Quadratic-weighted Cohen's κ** — secondary. Familiar to many readers,
|
|
22
24
|
* useful as a sanity check. Reports lower than α when marginals diverge.
|
|
23
25
|
* - **Pearson r** — tertiary. Captures rank-order agreement only; doesn't
|