oh-my-knowledge 0.22.0 → 0.24.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +63 -63
- package/README.zh.md +64 -62
- package/dist/src/analysis/report-diagnostics.d.ts +28 -3
- package/dist/src/analysis/report-diagnostics.d.ts.map +1 -1
- package/dist/src/analysis/report-diagnostics.js +201 -81
- package/dist/src/analysis/report-diagnostics.js.map +1 -1
- package/dist/src/analysis/sample-diagnostics.d.ts +9 -2
- package/dist/src/analysis/sample-diagnostics.d.ts.map +1 -1
- package/dist/src/analysis/sample-diagnostics.js +224 -25
- package/dist/src/analysis/sample-diagnostics.js.map +1 -1
- package/dist/src/authoring/evolver.d.ts +7 -2
- package/dist/src/authoring/evolver.d.ts.map +1 -1
- package/dist/src/authoring/evolver.js +43 -9
- package/dist/src/authoring/evolver.js.map +1 -1
- package/dist/src/authoring/generator.d.ts +24 -0
- package/dist/src/authoring/generator.d.ts.map +1 -1
- package/dist/src/authoring/generator.js +67 -5
- package/dist/src/authoring/generator.js.map +1 -1
- package/dist/src/cli/coverage-renderer.d.ts +15 -0
- package/dist/src/cli/coverage-renderer.d.ts.map +1 -0
- package/dist/src/cli/coverage-renderer.js +74 -0
- package/dist/src/cli/coverage-renderer.js.map +1 -0
- package/dist/src/cli/i18n-dict.d.ts +1 -1
- package/dist/src/cli/i18n-dict.d.ts.map +1 -1
- package/dist/src/cli/i18n-dict.js +62 -40
- package/dist/src/cli/i18n-dict.js.map +1 -1
- package/dist/src/cli/index.d.ts +3 -0
- package/dist/src/cli/index.d.ts.map +1 -0
- package/dist/src/{cli.js → cli/index.js} +123 -384
- package/dist/src/cli/index.js.map +1 -0
- package/dist/src/cli/parse-run-config.d.ts +56 -0
- package/dist/src/cli/parse-run-config.d.ts.map +1 -0
- package/dist/src/cli/parse-run-config.js +195 -0
- package/dist/src/cli/parse-run-config.js.map +1 -0
- package/dist/src/cli/progress.d.ts +25 -0
- package/dist/src/cli/progress.d.ts.map +1 -0
- package/dist/src/cli/progress.js +62 -0
- package/dist/src/cli/progress.js.map +1 -0
- package/dist/src/cli/update-check.d.ts +3 -0
- package/dist/src/cli/update-check.d.ts.map +1 -0
- package/dist/src/cli/update-check.js +37 -0
- package/dist/src/cli/update-check.js.map +1 -0
- package/dist/src/eval-core/cache.d.ts +7 -5
- package/dist/src/eval-core/cache.d.ts.map +1 -1
- package/dist/src/eval-core/cache.js +11 -7
- package/dist/src/eval-core/cache.js.map +1 -1
- package/dist/src/eval-core/comparability.d.ts +11 -0
- package/dist/src/eval-core/comparability.d.ts.map +1 -0
- package/dist/src/eval-core/comparability.js +271 -0
- package/dist/src/eval-core/comparability.js.map +1 -0
- package/dist/src/eval-core/dependency-checker.d.ts +1 -1
- package/dist/src/eval-core/dependency-checker.js +1 -1
- package/dist/src/eval-core/evaluation-execution.d.ts +5 -2
- package/dist/src/eval-core/evaluation-execution.d.ts.map +1 -1
- package/dist/src/eval-core/evaluation-execution.js +29 -21
- package/dist/src/eval-core/evaluation-execution.js.map +1 -1
- package/dist/src/eval-core/evaluation-job.d.ts +2 -2
- package/dist/src/eval-core/evaluation-job.d.ts.map +1 -1
- package/dist/src/eval-core/evaluation-job.js +2 -2
- package/dist/src/eval-core/evaluation-job.js.map +1 -1
- package/dist/src/eval-core/evaluation-reporting.d.ts +3 -1
- package/dist/src/eval-core/evaluation-reporting.d.ts.map +1 -1
- package/dist/src/eval-core/evaluation-reporting.js +65 -5
- package/dist/src/eval-core/evaluation-reporting.js.map +1 -1
- package/dist/src/eval-core/execution-strategy.js +6 -6
- package/dist/src/eval-core/execution-strategy.js.map +1 -1
- package/dist/src/eval-core/schema.d.ts.map +1 -1
- package/dist/src/eval-core/schema.js +14 -1
- package/dist/src/eval-core/schema.js.map +1 -1
- package/dist/src/eval-core/verdict.d.ts +3 -3
- package/dist/src/eval-core/verdict.js +3 -3
- package/dist/src/eval-workflows/batch-evaluation-workflow.d.ts +111 -0
- package/dist/src/eval-workflows/batch-evaluation-workflow.d.ts.map +1 -0
- package/dist/src/eval-workflows/batch-evaluation-workflow.js +215 -0
- package/dist/src/eval-workflows/batch-evaluation-workflow.js.map +1 -0
- package/dist/src/eval-workflows/evaluation-pipeline.d.ts +7 -5
- package/dist/src/eval-workflows/evaluation-pipeline.d.ts.map +1 -1
- package/dist/src/eval-workflows/evaluation-pipeline.js +13 -8
- package/dist/src/eval-workflows/evaluation-pipeline.js.map +1 -1
- package/dist/src/eval-workflows/evaluation-preparation.d.ts +10 -8
- package/dist/src/eval-workflows/evaluation-preparation.d.ts.map +1 -1
- package/dist/src/eval-workflows/evaluation-preparation.js +8 -4
- package/dist/src/eval-workflows/evaluation-preparation.js.map +1 -1
- package/dist/src/eval-workflows/run-evaluation.d.ts +23 -20
- package/dist/src/eval-workflows/run-evaluation.d.ts.map +1 -1
- package/dist/src/eval-workflows/run-evaluation.js +34 -25
- package/dist/src/eval-workflows/run-evaluation.js.map +1 -1
- package/dist/src/executors/claude-cli.d.ts.map +1 -1
- package/dist/src/executors/claude-cli.js +12 -7
- package/dist/src/executors/claude-cli.js.map +1 -1
- package/dist/src/executors/claude-sdk.d.ts +1 -1
- package/dist/src/executors/claude-sdk.js +1 -1
- package/dist/src/executors/codex-cli-trace.d.ts +10 -0
- package/dist/src/executors/codex-cli-trace.d.ts.map +1 -0
- package/dist/src/executors/codex-cli-trace.js +123 -0
- package/dist/src/executors/codex-cli-trace.js.map +1 -0
- package/dist/src/executors/codex-cli.d.ts +18 -0
- package/dist/src/executors/codex-cli.d.ts.map +1 -0
- package/dist/src/executors/codex-cli.js +254 -0
- package/dist/src/executors/codex-cli.js.map +1 -0
- package/dist/src/executors/codex-sdk.d.ts +18 -0
- package/dist/src/executors/codex-sdk.d.ts.map +1 -0
- package/dist/src/executors/codex-sdk.js +214 -0
- package/dist/src/executors/codex-sdk.js.map +1 -0
- package/dist/src/executors/gemini.d.ts.map +1 -1
- package/dist/src/executors/gemini.js +28 -24
- package/dist/src/executors/gemini.js.map +1 -1
- package/dist/src/executors/index.d.ts.map +1 -1
- package/dist/src/executors/index.js +7 -2
- package/dist/src/executors/index.js.map +1 -1
- package/dist/src/executors/runtime-fingerprint.d.ts +7 -0
- package/dist/src/executors/runtime-fingerprint.d.ts.map +1 -0
- package/dist/src/executors/runtime-fingerprint.js +277 -0
- package/dist/src/executors/runtime-fingerprint.js.map +1 -0
- package/dist/src/executors/script.d.ts.map +1 -1
- package/dist/src/executors/script.js +48 -56
- package/dist/src/executors/script.js.map +1 -1
- package/dist/src/executors/shared.d.ts +78 -1
- package/dist/src/executors/shared.d.ts.map +1 -1
- package/dist/src/executors/shared.js +203 -1
- package/dist/src/executors/shared.js.map +1 -1
- package/dist/src/grading/assertions.d.ts.map +1 -1
- package/dist/src/grading/assertions.js +21 -6
- package/dist/src/grading/assertions.js.map +1 -1
- package/dist/src/grading/gold-dataset.d.ts +1 -1
- package/dist/src/grading/gold-dataset.js +1 -1
- package/dist/src/grading/index.d.ts.map +1 -1
- package/dist/src/grading/index.js +11 -0
- package/dist/src/grading/index.js.map +1 -1
- package/dist/src/grading/judge.d.ts.map +1 -1
- package/dist/src/grading/judge.js +75 -6
- package/dist/src/grading/judge.js.map +1 -1
- package/dist/src/inputs/eval-config.js +2 -2
- package/dist/src/inputs/eval-config.js.map +1 -1
- package/dist/src/inputs/load-samples.d.ts.map +1 -1
- package/dist/src/inputs/load-samples.js +30 -4
- package/dist/src/inputs/load-samples.js.map +1 -1
- package/dist/src/inputs/skill-loader.d.ts +2 -2
- package/dist/src/inputs/skill-loader.d.ts.map +1 -1
- package/dist/src/inputs/skill-loader.js +2 -2
- package/dist/src/inputs/skill-loader.js.map +1 -1
- package/dist/src/renderer/html-renderer.d.ts +5 -4
- package/dist/src/renderer/html-renderer.d.ts.map +1 -1
- package/dist/src/renderer/html-renderer.js +217 -93
- package/dist/src/renderer/html-renderer.js.map +1 -1
- package/dist/src/renderer/layout.d.ts +2 -1
- package/dist/src/renderer/layout.d.ts.map +1 -1
- package/dist/src/renderer/layout.js +30 -42
- package/dist/src/renderer/layout.js.map +1 -1
- package/dist/src/renderer/summary.d.ts +3 -3
- package/dist/src/renderer/summary.d.ts.map +1 -1
- package/dist/src/renderer/summary.js +232 -55
- package/dist/src/renderer/summary.js.map +1 -1
- package/dist/src/renderer/trends.d.ts.map +1 -1
- package/dist/src/renderer/trends.js +5 -3
- package/dist/src/renderer/trends.js.map +1 -1
- package/dist/src/server/report-server.js +4 -4
- package/dist/src/server/report-server.js.map +1 -1
- package/dist/src/server/report-store.d.ts +7 -5
- package/dist/src/server/report-store.d.ts.map +1 -1
- package/dist/src/server/report-store.js +39 -11
- package/dist/src/server/report-store.js.map +1 -1
- package/dist/src/types/eval.d.ts +27 -5
- package/dist/src/types/eval.d.ts.map +1 -1
- package/dist/src/types/executor.d.ts +5 -0
- package/dist/src/types/executor.d.ts.map +1 -1
- package/dist/src/types/judge.d.ts +10 -0
- package/dist/src/types/judge.d.ts.map +1 -1
- package/dist/src/types/report.d.ts +151 -29
- package/dist/src/types/report.d.ts.map +1 -1
- package/dist/src/types/storage.d.ts +7 -7
- package/dist/src/types/storage.d.ts.map +1 -1
- package/package.json +14 -5
- package/dist/src/cli.d.ts +0 -3
- package/dist/src/cli.d.ts.map +0 -1
- package/dist/src/cli.js.map +0 -1
- package/dist/src/eval-workflows/each-evaluation-workflow.d.ts +0 -153
- package/dist/src/eval-workflows/each-evaluation-workflow.d.ts.map +0 -1
- package/dist/src/eval-workflows/each-evaluation-workflow.js +0 -178
- package/dist/src/eval-workflows/each-evaluation-workflow.js.map +0 -1
- package/dist/src/executors/openai-cli.d.ts +0 -3
- package/dist/src/executors/openai-cli.d.ts.map +0 -1
- package/dist/src/executors/openai-cli.js +0 -60
- package/dist/src/executors/openai-cli.js.map +0 -1
|
@@ -1,41 +1,136 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* Auto-analysis: detect patterns and generate insights from evaluation results.
|
|
3
3
|
*/
|
|
4
|
+
import { normalizeCapability } from './sample-diagnostics.js';
|
|
4
5
|
/**
|
|
5
|
-
* Analyze an evaluation report and produce insights
|
|
6
|
+
* Analyze an evaluation report and produce structured insights.
|
|
6
7
|
*/
|
|
7
|
-
export function analyzeResults(report) {
|
|
8
|
+
export function analyzeResults(report, opts = {}) {
|
|
8
9
|
const insights = [];
|
|
9
|
-
const suggestions = [];
|
|
10
10
|
const variants = report.meta?.variants || [];
|
|
11
11
|
const results = report.results || [];
|
|
12
|
+
// sampleQuality aggregate is built from sample metadata only,
|
|
13
|
+
// independent of result data. Computed even when results.length === 0 or
|
|
14
|
+
// variants.length < 2 (e.g. dry-run / single-variant analysis).
|
|
15
|
+
const sampleQuality = opts.samples
|
|
16
|
+
? buildSampleQualityAggregate(opts.samples)
|
|
17
|
+
: undefined;
|
|
12
18
|
if (results.length === 0 || variants.length < 2) {
|
|
13
|
-
return { insights,
|
|
19
|
+
return { insights, ...(sampleQuality && { sampleQuality }) };
|
|
14
20
|
}
|
|
15
21
|
// 1. Low-discrimination assertions
|
|
16
|
-
detectLowDiscrimination(results, variants, insights
|
|
22
|
+
detectLowDiscrimination(results, variants, insights);
|
|
17
23
|
// 2. Uniform scores across variants
|
|
18
|
-
detectUniformScores(results, variants, insights
|
|
24
|
+
detectUniformScores(results, variants, insights);
|
|
19
25
|
// 3. All-pass / all-fail assertions
|
|
20
|
-
detectAllPassFail(results, variants, insights
|
|
26
|
+
detectAllPassFail(results, variants, insights);
|
|
21
27
|
// 4. High-cost samples
|
|
22
28
|
detectHighCost(results, variants, insights);
|
|
23
29
|
// 5. Efficiency gap (turns & cost)
|
|
24
|
-
detectEfficiencyGap(report, variants, insights
|
|
30
|
+
detectEfficiencyGap(report, variants, insights);
|
|
25
31
|
// 6. Agent tool usage patterns
|
|
26
|
-
detectToolPatterns(report, variants, insights
|
|
32
|
+
detectToolPatterns(report, variants, insights);
|
|
27
33
|
// 7. Tooling / permission issues
|
|
28
|
-
detectToolPermissionIssues(results, variants, insights
|
|
34
|
+
detectToolPermissionIssues(results, variants, insights);
|
|
29
35
|
// 8. Trace integrity
|
|
30
|
-
detectTraceIntegrity(report, variants, insights
|
|
36
|
+
detectTraceIntegrity(report, variants, insights);
|
|
31
37
|
// 9. Agent assertion discrimination
|
|
32
|
-
detectAgentAssertionDiscrimination(results, variants, insights
|
|
38
|
+
detectAgentAssertionDiscrimination(results, variants, insights);
|
|
33
39
|
// 10. Suggest --repeat when score variance is high and no repeat data
|
|
34
|
-
detectNeedRepeat(report, results, variants, insights
|
|
35
|
-
|
|
36
|
-
|
|
40
|
+
detectNeedRepeat(report, results, variants, insights);
|
|
41
|
+
return {
|
|
42
|
+
insights,
|
|
43
|
+
...(sampleQuality && { sampleQuality }),
|
|
44
|
+
};
|
|
37
45
|
}
|
|
38
|
-
|
|
46
|
+
/**
|
|
47
|
+
* Build sample design science aggregate from sample metadata.
|
|
48
|
+
*
|
|
49
|
+
* Pure function — no result/score data needed. Reads:
|
|
50
|
+
* - `Sample.capability` (string[], normalized case-insensitive + dash/camel/underscore stripped)
|
|
51
|
+
* - `Sample.difficulty` ('easy' | 'medium' | 'hard')
|
|
52
|
+
* - `Sample.construct` (free-form string)
|
|
53
|
+
* - `Sample.provenance` ('human' | 'llm-generated' | 'production-trace')
|
|
54
|
+
* - `Sample.rubric` (for avgRubricLength)
|
|
55
|
+
*
|
|
56
|
+
* Missing fields are bucketed under the `unspecified` key in the relevant
|
|
57
|
+
* distribution map, so users see "I have N samples without difficulty declared".
|
|
58
|
+
*
|
|
59
|
+
* Used by `bench diagnose` CLI to surface coverage gaps. Does NOT participate
|
|
60
|
+
* in grading / judge / verdict. See docs/sample-design-spec.md.
|
|
61
|
+
*/
|
|
62
|
+
export function buildSampleQualityAggregate(samples) {
|
|
63
|
+
const capabilityCoverage = {};
|
|
64
|
+
const difficultyDistribution = {
|
|
65
|
+
easy: 0, medium: 0, hard: 0, unspecified: 0,
|
|
66
|
+
};
|
|
67
|
+
const constructDistribution = {};
|
|
68
|
+
const provenanceBreakdown = {};
|
|
69
|
+
let totalRubricLength = 0;
|
|
70
|
+
let rubricCount = 0;
|
|
71
|
+
let withCapability = 0;
|
|
72
|
+
let withDifficulty = 0;
|
|
73
|
+
let withConstruct = 0;
|
|
74
|
+
let withProvenance = 0;
|
|
75
|
+
for (const sample of samples) {
|
|
76
|
+
// capability — normalize case + dash/camel/underscore so 'api-selection' / 'apiSelection' / 'API_Selection' merge.
|
|
77
|
+
if (Array.isArray(sample.capability) && sample.capability.length > 0) {
|
|
78
|
+
withCapability++;
|
|
79
|
+
const seen = new Set();
|
|
80
|
+
for (const rawCap of sample.capability) {
|
|
81
|
+
if (typeof rawCap !== 'string')
|
|
82
|
+
continue;
|
|
83
|
+
const cap = normalizeCapability(rawCap);
|
|
84
|
+
if (seen.has(cap))
|
|
85
|
+
continue; // 同 sample 内同 capability 重复声明只计 1
|
|
86
|
+
seen.add(cap);
|
|
87
|
+
capabilityCoverage[cap] = (capabilityCoverage[cap] || 0) + 1;
|
|
88
|
+
}
|
|
89
|
+
}
|
|
90
|
+
// difficulty
|
|
91
|
+
if (sample.difficulty) {
|
|
92
|
+
withDifficulty++;
|
|
93
|
+
difficultyDistribution[sample.difficulty]++;
|
|
94
|
+
}
|
|
95
|
+
else {
|
|
96
|
+
difficultyDistribution.unspecified++;
|
|
97
|
+
}
|
|
98
|
+
// construct (free-form)
|
|
99
|
+
if (sample.construct) {
|
|
100
|
+
withConstruct++;
|
|
101
|
+
constructDistribution[sample.construct] = (constructDistribution[sample.construct] || 0) + 1;
|
|
102
|
+
}
|
|
103
|
+
else {
|
|
104
|
+
constructDistribution.unspecified = (constructDistribution.unspecified || 0) + 1;
|
|
105
|
+
}
|
|
106
|
+
// provenance
|
|
107
|
+
if (sample.provenance) {
|
|
108
|
+
withProvenance++;
|
|
109
|
+
provenanceBreakdown[sample.provenance] = (provenanceBreakdown[sample.provenance] || 0) + 1;
|
|
110
|
+
}
|
|
111
|
+
else {
|
|
112
|
+
provenanceBreakdown.unspecified = (provenanceBreakdown.unspecified || 0) + 1;
|
|
113
|
+
}
|
|
114
|
+
// rubric length(only counted if present, NaN-safe)
|
|
115
|
+
if (sample.rubric) {
|
|
116
|
+
totalRubricLength += sample.rubric.trim().length;
|
|
117
|
+
rubricCount++;
|
|
118
|
+
}
|
|
119
|
+
}
|
|
120
|
+
return {
|
|
121
|
+
capabilityCoverage,
|
|
122
|
+
difficultyDistribution,
|
|
123
|
+
constructDistribution,
|
|
124
|
+
provenanceBreakdown,
|
|
125
|
+
avgRubricLength: rubricCount > 0 ? Math.round(totalRubricLength / rubricCount) : 0,
|
|
126
|
+
sampleCountWithCapability: withCapability,
|
|
127
|
+
sampleCountWithDifficulty: withDifficulty,
|
|
128
|
+
sampleCountWithConstruct: withConstruct,
|
|
129
|
+
sampleCountWithProvenance: withProvenance,
|
|
130
|
+
};
|
|
131
|
+
}
|
|
132
|
+
export function generateAnalysisSummary(report, lang = 'zh') {
|
|
133
|
+
const variants = report.meta?.variants || [];
|
|
39
134
|
if (variants.length < 2)
|
|
40
135
|
return undefined;
|
|
41
136
|
const stats = report.summary || {};
|
|
@@ -69,15 +164,25 @@ function generateSummary(report, variants) {
|
|
|
69
164
|
if (scoreDiff != null && tScore != null && cScore != null) {
|
|
70
165
|
const absDiff = Math.abs(scoreDiff);
|
|
71
166
|
if (absDiff < 0.1) {
|
|
72
|
-
lines.push(
|
|
167
|
+
lines.push(lang === 'zh'
|
|
168
|
+
? `【结论】${test} 与 ${control} 综合得分持平(${tScore.toFixed(2)} vs ${cScore.toFixed(2)}),质量无显著差异。`
|
|
169
|
+
: `【Conclusion】${test} and ${control} are effectively tied on composite score (${tScore.toFixed(2)} vs ${cScore.toFixed(2)}); no clear quality difference.`);
|
|
73
170
|
}
|
|
74
171
|
else if (scoreDiff > 0) {
|
|
75
|
-
const tag =
|
|
76
|
-
|
|
172
|
+
const tag = lang === 'zh'
|
|
173
|
+
? (absDiff > 0.3 ? '明显领先' : '略优')
|
|
174
|
+
: (absDiff > 0.3 ? 'clearly ahead' : 'slightly ahead');
|
|
175
|
+
lines.push(lang === 'zh'
|
|
176
|
+
? `【结论】${test} 综合得分 ${tag}(${tScore.toFixed(2)} vs ${cScore.toFixed(2)},+${scoreDiff.toFixed(2)})。`
|
|
177
|
+
: `【Conclusion】${test} is ${tag} on composite score (${tScore.toFixed(2)} vs ${cScore.toFixed(2)}, +${scoreDiff.toFixed(2)}).`);
|
|
77
178
|
}
|
|
78
179
|
else {
|
|
79
|
-
const tag =
|
|
80
|
-
|
|
180
|
+
const tag = lang === 'zh'
|
|
181
|
+
? (absDiff > 0.3 ? '明显落后' : '略低')
|
|
182
|
+
: (absDiff > 0.3 ? 'clearly behind' : 'slightly lower');
|
|
183
|
+
lines.push(lang === 'zh'
|
|
184
|
+
? `【结论】${test} 综合得分 ${tag}(${tScore.toFixed(2)} vs ${cScore.toFixed(2)},${scoreDiff.toFixed(2)})。`
|
|
185
|
+
: `【Conclusion】${test} is ${tag} on composite score (${tScore.toFixed(2)} vs ${cScore.toFixed(2)}, ${scoreDiff.toFixed(2)}).`);
|
|
81
186
|
}
|
|
82
187
|
}
|
|
83
188
|
// ── Section 2: Key differentiators with concrete numbers ──
|
|
@@ -90,23 +195,31 @@ function generateSummary(report, variants) {
|
|
|
90
195
|
// Both same — don't mention (e.g. both 5/5 is not interesting)
|
|
91
196
|
}
|
|
92
197
|
else if (tFact > cFact) {
|
|
93
|
-
diffs.push(
|
|
198
|
+
diffs.push(lang === 'zh'
|
|
199
|
+
? `事实性 ${tFact.toFixed(1)} vs ${cFact.toFixed(1)}(↑${(tFact - cFact).toFixed(1)})`
|
|
200
|
+
: `Fact score ${tFact.toFixed(1)} vs ${cFact.toFixed(1)} (up ${(tFact - cFact).toFixed(1)})`);
|
|
94
201
|
}
|
|
95
202
|
else {
|
|
96
|
-
diffs.push(
|
|
203
|
+
diffs.push(lang === 'zh'
|
|
204
|
+
? `事实性 ${tFact.toFixed(1)} vs ${cFact.toFixed(1)}(↓${(cFact - tFact).toFixed(1)})`
|
|
205
|
+
: `Fact score ${tFact.toFixed(1)} vs ${cFact.toFixed(1)} (down ${(cFact - tFact).toFixed(1)})`);
|
|
97
206
|
}
|
|
98
207
|
}
|
|
99
208
|
const tBehavior = ts.avgBehaviorScore;
|
|
100
209
|
const cBehavior = cs.avgBehaviorScore;
|
|
101
210
|
if (tBehavior != null && cBehavior != null && Math.abs(tBehavior - cBehavior) > 0.3) {
|
|
102
211
|
const dir = tBehavior > cBehavior ? '↑' : '↓';
|
|
103
|
-
diffs.push(
|
|
212
|
+
diffs.push(lang === 'zh'
|
|
213
|
+
? `行为合规 ${tBehavior.toFixed(1)} vs ${cBehavior.toFixed(1)}(${dir}${Math.abs(tBehavior - cBehavior).toFixed(1)})`
|
|
214
|
+
: `Behavior score ${tBehavior.toFixed(1)} vs ${cBehavior.toFixed(1)} (${dir === '↑' ? 'up' : 'down'} ${Math.abs(tBehavior - cBehavior).toFixed(1)})`);
|
|
104
215
|
}
|
|
105
216
|
const tJudge = ts.avgJudgeScore;
|
|
106
217
|
const cJudge = cs.avgJudgeScore;
|
|
107
218
|
if (tJudge != null && cJudge != null && Math.abs(tJudge - cJudge) >= 0.5) {
|
|
108
219
|
const dir = tJudge > cJudge ? '↑' : '↓';
|
|
109
|
-
diffs.push(
|
|
220
|
+
diffs.push(lang === 'zh'
|
|
221
|
+
? `LLM 评价 ${tJudge.toFixed(1)} vs ${cJudge.toFixed(1)}(${dir}${Math.abs(tJudge - cJudge).toFixed(1)})`
|
|
222
|
+
: `LLM judge ${tJudge.toFixed(1)} vs ${cJudge.toFixed(1)} (${dir === '↑' ? 'up' : 'down'} ${Math.abs(tJudge - cJudge).toFixed(1)})`);
|
|
110
223
|
}
|
|
111
224
|
// Efficiency — with percentages
|
|
112
225
|
const cTurns = cs.avgNumTurns;
|
|
@@ -114,10 +227,14 @@ function generateSummary(report, variants) {
|
|
|
114
227
|
if (cTurns > 0 && tTurns > 0 && cTurns !== tTurns) {
|
|
115
228
|
const pct = Math.abs(((tTurns - cTurns) / cTurns) * 100).toFixed(0);
|
|
116
229
|
if (tTurns < cTurns) {
|
|
117
|
-
diffs.push(
|
|
230
|
+
diffs.push(lang === 'zh'
|
|
231
|
+
? `轮次 ${tTurns.toFixed(1)} vs ${cTurns.toFixed(1)}(↓${pct}%,路径更高效)`
|
|
232
|
+
: `Turns ${tTurns.toFixed(1)} vs ${cTurns.toFixed(1)} (down ${pct}%, more efficient path)`);
|
|
118
233
|
}
|
|
119
234
|
else {
|
|
120
|
-
diffs.push(
|
|
235
|
+
diffs.push(lang === 'zh'
|
|
236
|
+
? `轮次 ${tTurns.toFixed(1)} vs ${cTurns.toFixed(1)}(↑${pct}%)`
|
|
237
|
+
: `Turns ${tTurns.toFixed(1)} vs ${cTurns.toFixed(1)} (up ${pct}%)`);
|
|
121
238
|
}
|
|
122
239
|
}
|
|
123
240
|
// Cost — with percentages
|
|
@@ -126,10 +243,14 @@ function generateSummary(report, variants) {
|
|
|
126
243
|
if (cCost > 0 && tCost > 0 && Math.abs(tCost - cCost) / cCost > 0.05) {
|
|
127
244
|
const pct = Math.abs(((tCost - cCost) / cCost) * 100).toFixed(0);
|
|
128
245
|
if (tCost < cCost) {
|
|
129
|
-
diffs.push(
|
|
246
|
+
diffs.push(lang === 'zh'
|
|
247
|
+
? `单用例成本 $${tCost.toFixed(4)} vs $${cCost.toFixed(4)}(↓${pct}%)`
|
|
248
|
+
: `Cost per sample $${tCost.toFixed(4)} vs $${cCost.toFixed(4)} (down ${pct}%)`);
|
|
130
249
|
}
|
|
131
250
|
else {
|
|
132
|
-
diffs.push(
|
|
251
|
+
diffs.push(lang === 'zh'
|
|
252
|
+
? `单用例成本 $${tCost.toFixed(4)} vs $${cCost.toFixed(4)}(↑${pct}%)`
|
|
253
|
+
: `Cost per sample $${tCost.toFixed(4)} vs $${cCost.toFixed(4)} (up ${pct}%)`);
|
|
133
254
|
}
|
|
134
255
|
}
|
|
135
256
|
// Duration — with percentages
|
|
@@ -140,20 +261,28 @@ function generateSummary(report, variants) {
|
|
|
140
261
|
const tSec = (tDur / 1000).toFixed(1);
|
|
141
262
|
const cSec = (cDur / 1000).toFixed(1);
|
|
142
263
|
if (tDur < cDur) {
|
|
143
|
-
diffs.push(
|
|
264
|
+
diffs.push(lang === 'zh'
|
|
265
|
+
? `耗时 ${tSec}s vs ${cSec}s(↓${pct}%)`
|
|
266
|
+
: `Duration ${tSec}s vs ${cSec}s (down ${pct}%)`);
|
|
144
267
|
}
|
|
145
268
|
else {
|
|
146
|
-
diffs.push(
|
|
269
|
+
diffs.push(lang === 'zh'
|
|
270
|
+
? `耗时 ${tSec}s vs ${cSec}s(↑${pct}%)`
|
|
271
|
+
: `Duration ${tSec}s vs ${cSec}s (up ${pct}%)`);
|
|
147
272
|
}
|
|
148
273
|
}
|
|
149
274
|
// Tool usage difference
|
|
150
275
|
const tTools = ts.avgToolCalls;
|
|
151
276
|
const cTools = cs.avgToolCalls;
|
|
152
277
|
if (tTools != null && cTools != null && Math.abs(tTools - cTools) > 0.5) {
|
|
153
|
-
diffs.push(
|
|
278
|
+
diffs.push(lang === 'zh'
|
|
279
|
+
? `工具调用 ${tTools.toFixed(1)} vs ${cTools.toFixed(1)} 次`
|
|
280
|
+
: `Tool calls ${tTools.toFixed(1)} vs ${cTools.toFixed(1)}`);
|
|
154
281
|
}
|
|
155
282
|
if (diffs.length > 0) {
|
|
156
|
-
lines.push(
|
|
283
|
+
lines.push(lang === 'zh'
|
|
284
|
+
? `【关键差异】${diffs.join(';')}。`
|
|
285
|
+
: `【Key differences】${diffs.join('; ')}.`);
|
|
157
286
|
}
|
|
158
287
|
// ── Section 3: Synthesis — connect the dots ──
|
|
159
288
|
const synthesis = [];
|
|
@@ -161,22 +290,32 @@ function generateSummary(report, variants) {
|
|
|
161
290
|
if (scoreDiff != null && cCost > 0 && tCost > 0) {
|
|
162
291
|
const costRatio = (tCost - cCost) / cCost;
|
|
163
292
|
if (Math.abs(scoreDiff) < 0.1 && costRatio < -0.15) {
|
|
164
|
-
synthesis.push(
|
|
293
|
+
synthesis.push(lang === 'zh'
|
|
294
|
+
? `质量相当但成本显著降低,${test} 是更经济的选择`
|
|
295
|
+
: `similar quality with materially lower cost; ${test} is the more economical choice`);
|
|
165
296
|
}
|
|
166
297
|
else if (scoreDiff > 0.1 && costRatio > 0.15) {
|
|
167
|
-
synthesis.push(
|
|
298
|
+
synthesis.push(lang === 'zh'
|
|
299
|
+
? '质量提升伴随成本上涨,需权衡投入产出比'
|
|
300
|
+
: 'quality improved, but cost also increased; weigh the return on investment');
|
|
168
301
|
}
|
|
169
302
|
else if (scoreDiff > 0.1 && costRatio <= 0) {
|
|
170
|
-
synthesis.push(
|
|
303
|
+
synthesis.push(lang === 'zh'
|
|
304
|
+
? `质量与成本双优,${test} 全面领先`
|
|
305
|
+
: `${test} leads on both quality and cost`);
|
|
171
306
|
}
|
|
172
307
|
else if (scoreDiff < -0.1 && costRatio < -0.15) {
|
|
173
|
-
synthesis.push(
|
|
308
|
+
synthesis.push(lang === 'zh'
|
|
309
|
+
? '成本虽降但质量下滑,需评估质量底线是否可接受'
|
|
310
|
+
: 'cost decreased, but quality dropped; check whether the quality floor is still acceptable');
|
|
174
311
|
}
|
|
175
312
|
}
|
|
176
313
|
// Tool success rate concern
|
|
177
314
|
const tToolSuccess = ts.toolSuccessRate;
|
|
178
315
|
if (tToolSuccess != null && tToolSuccess < 1 && tToolSuccess >= 0.5) {
|
|
179
|
-
synthesis.push(
|
|
316
|
+
synthesis.push(lang === 'zh'
|
|
317
|
+
? `${test} 存在工具调用失败(成功率 ${(tToolSuccess * 100).toFixed(0)}%),可能拉低了得分`
|
|
318
|
+
: `${test} had tool-call failures (${(tToolSuccess * 100).toFixed(0)}% success), which may have pulled the score down`);
|
|
180
319
|
}
|
|
181
320
|
// Variance / significance from --repeat
|
|
182
321
|
if (report.variance) {
|
|
@@ -189,23 +328,33 @@ function generateSummary(report, variants) {
|
|
|
189
328
|
const primaryVal = es.primary === 'g' ? es.hedgesG : es.cohensD;
|
|
190
329
|
const secondaryLabel = es.primary === 'g' ? 'd' : 'g';
|
|
191
330
|
const secondaryVal = es.primary === 'g' ? es.cohensD : es.hedgesG;
|
|
192
|
-
esText =
|
|
331
|
+
esText = lang === 'zh'
|
|
332
|
+
? `,效应量 ${es.primary}=${primaryVal.toFixed(2)}(${es.magnitude},${secondaryLabel}=${secondaryVal.toFixed(2)})`
|
|
333
|
+
: `, effect size ${es.primary}=${primaryVal.toFixed(2)} (${es.magnitude}, ${secondaryLabel}=${secondaryVal.toFixed(2)})`;
|
|
193
334
|
}
|
|
194
335
|
if (comp.significant) {
|
|
195
|
-
synthesis.push(
|
|
336
|
+
synthesis.push(lang === 'zh'
|
|
337
|
+
? `${v.runs} 轮重复评测显示差异具有统计显著性(t=${comp.tStatistic.toFixed(2)}, df=${comp.df.toFixed(1)}, p<0.05${esText})`
|
|
338
|
+
: `${v.runs} repeated runs show a statistically significant difference (t=${comp.tStatistic.toFixed(2)}, df=${comp.df.toFixed(1)}, p<0.05${esText})`);
|
|
196
339
|
}
|
|
197
340
|
else {
|
|
198
|
-
synthesis.push(
|
|
341
|
+
synthesis.push(lang === 'zh'
|
|
342
|
+
? `${v.runs} 轮重复评测未达到统计显著性(t=${comp.tStatistic.toFixed(2)}, df=${comp.df.toFixed(1)}${esText}),差异可能源于随机波动`
|
|
343
|
+
: `${v.runs} repeated runs did not reach statistical significance (t=${comp.tStatistic.toFixed(2)}, df=${comp.df.toFixed(1)}${esText}); the gap may be random variation`);
|
|
199
344
|
}
|
|
200
345
|
}
|
|
201
346
|
}
|
|
202
347
|
const testVd = v.perVariant[test];
|
|
203
348
|
if (testVd) {
|
|
204
|
-
synthesis.push(
|
|
349
|
+
synthesis.push(lang === 'zh'
|
|
350
|
+
? `${test} 跨轮 95% 置信区间 [${testVd.lower.toFixed(2)}, ${testVd.upper.toFixed(2)}]`
|
|
351
|
+
: `${test} cross-run 95% confidence interval [${testVd.lower.toFixed(2)}, ${testVd.upper.toFixed(2)}]`);
|
|
205
352
|
}
|
|
206
353
|
}
|
|
207
354
|
if (synthesis.length > 0) {
|
|
208
|
-
lines.push(
|
|
355
|
+
lines.push(lang === 'zh'
|
|
356
|
+
? `【综合洞察】${synthesis.join(';')}。`
|
|
357
|
+
: `【Synthesis】${synthesis.join('; ')}.`);
|
|
209
358
|
}
|
|
210
359
|
// Caveats and recommendations are handled by the issues table below,
|
|
211
360
|
// so the summary focuses only on verdict + differentiators + synthesis.
|
|
@@ -243,7 +392,7 @@ function collectAgentAssertionTypes(results, variants) {
|
|
|
243
392
|
}
|
|
244
393
|
return types;
|
|
245
394
|
}
|
|
246
|
-
function detectLowDiscrimination(results, variants, insights
|
|
395
|
+
function detectLowDiscrimination(results, variants, insights) {
|
|
247
396
|
// For each sample, check if all variants have the same assertion pass/fail pattern
|
|
248
397
|
const allPassedPatterns = [];
|
|
249
398
|
const allFailedPatterns = [];
|
|
@@ -282,22 +431,18 @@ function detectLowDiscrimination(results, variants, insights, suggestions) {
|
|
|
282
431
|
insights.push({
|
|
283
432
|
type: 'low_discrimination_all_passed',
|
|
284
433
|
severity: 'info',
|
|
285
|
-
message: `${allPassedPatterns.length} 个断言所有变体均通过,baseline 也能答对,区分度低`,
|
|
286
434
|
details: allPassedPatterns,
|
|
287
435
|
});
|
|
288
|
-
suggestions.push('对于所有变体均通过的断言,考虑替换为检测 skill 文档中独有细节的断言(如特定参数名、配置值)');
|
|
289
436
|
}
|
|
290
437
|
if (allFailedPatterns.length > 0) {
|
|
291
438
|
insights.push({
|
|
292
439
|
type: 'low_discrimination_all_failed',
|
|
293
440
|
severity: 'warning',
|
|
294
|
-
message: `${allFailedPatterns.length} 个断言所有变体均失败,断言可能过于严格或存在配置错误`,
|
|
295
441
|
details: allFailedPatterns,
|
|
296
442
|
});
|
|
297
|
-
suggestions.push('对于所有变体均失败的断言,检查断言条件是否正确,或降低匹配要求');
|
|
298
443
|
}
|
|
299
444
|
}
|
|
300
|
-
function detectUniformScores(results, variants, insights
|
|
445
|
+
function detectUniformScores(results, variants, insights) {
|
|
301
446
|
let uniformCount = 0;
|
|
302
447
|
const uniformSamples = [];
|
|
303
448
|
for (const r of results) {
|
|
@@ -317,15 +462,11 @@ function detectUniformScores(results, variants, insights, suggestions) {
|
|
|
317
462
|
insights.push({
|
|
318
463
|
type: 'uniform_scores',
|
|
319
464
|
severity: uniformCount === results.length ? 'warning' : 'info',
|
|
320
|
-
message: `${uniformCount}/${results.length} 个用例在各变体间分差 < 0.5,区分度较低`,
|
|
321
465
|
details: uniformSamples,
|
|
322
466
|
});
|
|
323
|
-
if (uniformCount === results.length) {
|
|
324
|
-
suggestions.push('所有用例分数差异都很小,建议增加更有挑战性的测试用例或更严格的评分标准');
|
|
325
|
-
}
|
|
326
467
|
}
|
|
327
468
|
}
|
|
328
|
-
function detectAllPassFail(results, variants, insights
|
|
469
|
+
function detectAllPassFail(results, variants, insights) {
|
|
329
470
|
let allPassCount = 0;
|
|
330
471
|
let allFailCount = 0;
|
|
331
472
|
for (const r of results) {
|
|
@@ -344,22 +485,18 @@ function detectAllPassFail(results, variants, insights, suggestions) {
|
|
|
344
485
|
insights.push({
|
|
345
486
|
type: 'all_pass',
|
|
346
487
|
severity: 'warning',
|
|
347
|
-
message: '所有断言在所有变体上全部通过,断言可能过于宽松',
|
|
348
488
|
details: { allPassCount },
|
|
349
489
|
});
|
|
350
|
-
suggestions.push('所有断言都通过了,考虑增加更严格的断言来更好地区分变体质量');
|
|
351
490
|
}
|
|
352
491
|
if (allFailCount === totalEntries && totalEntries > 0) {
|
|
353
492
|
insights.push({
|
|
354
493
|
type: 'all_fail',
|
|
355
494
|
severity: 'error',
|
|
356
|
-
message: '所有断言在所有变体上全部失败,请检查断言配置是否正确',
|
|
357
495
|
details: { allFailCount },
|
|
358
496
|
});
|
|
359
|
-
suggestions.push('所有断言都失败了,请检查评测配置是否有误');
|
|
360
497
|
}
|
|
361
498
|
}
|
|
362
|
-
function detectNeedRepeat(report, results, variants, insights
|
|
499
|
+
function detectNeedRepeat(report, results, variants, insights) {
|
|
363
500
|
// Skip if already has variance data (i.e. --repeat was used)
|
|
364
501
|
if (report.variance)
|
|
365
502
|
return;
|
|
@@ -373,15 +510,13 @@ function detectNeedRepeat(report, results, variants, insights, suggestions) {
|
|
|
373
510
|
insights.push({
|
|
374
511
|
type: 'suggest_repeat',
|
|
375
512
|
severity: 'info',
|
|
376
|
-
message: `${v} 的分数跨度较大(${s.minCompositeScore}~${s.maxCompositeScore}),建议使用 --repeat 3 多轮评测以获取方差分析和统计显著性检验`,
|
|
377
513
|
details: { variant: v, min: s.minCompositeScore, max: s.maxCompositeScore, spread },
|
|
378
514
|
});
|
|
379
|
-
suggestions.push(`运行 omk bench run --repeat 3 获取置信区间和 t 检验结果,量化变体间差异的统计显著性`);
|
|
380
515
|
return; // Only suggest once
|
|
381
516
|
}
|
|
382
517
|
}
|
|
383
518
|
}
|
|
384
|
-
function detectEfficiencyGap(report, variants, insights
|
|
519
|
+
function detectEfficiencyGap(report, variants, insights) {
|
|
385
520
|
if (variants.length < 2)
|
|
386
521
|
return;
|
|
387
522
|
const summary = report.summary || {};
|
|
@@ -422,14 +557,12 @@ function detectEfficiencyGap(report, variants, insights, suggestions) {
|
|
|
422
557
|
insights.push({
|
|
423
558
|
type: 'efficiency_gap',
|
|
424
559
|
severity: 'info',
|
|
425
|
-
message: details.join(';'),
|
|
426
560
|
details: { baseline: variants[0], variant: variants[i], baseTurns, otherTurns, baseCost, otherCost },
|
|
427
561
|
});
|
|
428
|
-
suggestions.push(`${variants[i]} 在效率维度与 ${variants[0]} 存在显著差异,这对导航型 Skill 是重要的价值体现`);
|
|
429
562
|
}
|
|
430
563
|
}
|
|
431
564
|
}
|
|
432
|
-
function detectToolPatterns(report, variants, insights
|
|
565
|
+
function detectToolPatterns(report, variants, insights) {
|
|
433
566
|
const summary = report.summary || {};
|
|
434
567
|
const hasTools = variants.some((v) => summary[v]?.avgToolCalls != null && summary[v].avgToolCalls > 0);
|
|
435
568
|
if (!hasTools)
|
|
@@ -444,10 +577,8 @@ function detectToolPatterns(report, variants, insights, suggestions) {
|
|
|
444
577
|
insights.push({
|
|
445
578
|
type: 'low_tool_success_rate',
|
|
446
579
|
severity: 'warning',
|
|
447
|
-
message: `${v} 的工具调用成功率仅 ${(s.toolSuccessRate * 100).toFixed(0)}%,可能存在工具选择或参数问题`,
|
|
448
580
|
details: { variant: v, toolSuccessRate: s.toolSuccessRate, avgToolCalls: s.avgToolCalls },
|
|
449
581
|
});
|
|
450
|
-
suggestions.push(`检查 ${v} 的工具调用失败模式,考虑在 skill 中增加工具使用指导`);
|
|
451
582
|
}
|
|
452
583
|
}
|
|
453
584
|
// Compare tool counts between variants
|
|
@@ -462,16 +593,13 @@ function detectToolPatterns(report, variants, insights, suggestions) {
|
|
|
462
593
|
insights.push({
|
|
463
594
|
type: 'tool_count_gap',
|
|
464
595
|
severity: 'info',
|
|
465
|
-
message: diff > 0
|
|
466
|
-
? `${variants[i]} 平均多调用 ${diff.toFixed(1)} 次工具(${other.avgToolCalls} vs ${base.avgToolCalls})`
|
|
467
|
-
: `${variants[i]} 平均少调用 ${Math.abs(diff).toFixed(1)} 次工具(${other.avgToolCalls} vs ${base.avgToolCalls})`,
|
|
468
596
|
details: { baseline: variants[0], variant: variants[i], baseTools: base.avgToolCalls, otherTools: other.avgToolCalls },
|
|
469
597
|
});
|
|
470
598
|
}
|
|
471
599
|
}
|
|
472
600
|
}
|
|
473
601
|
}
|
|
474
|
-
function detectToolPermissionIssues(results, variants, insights
|
|
602
|
+
function detectToolPermissionIssues(results, variants, insights) {
|
|
475
603
|
const permissionErrors = [];
|
|
476
604
|
for (const result of results) {
|
|
477
605
|
for (const variant of variants) {
|
|
@@ -496,12 +624,10 @@ function detectToolPermissionIssues(results, variants, insights, suggestions) {
|
|
|
496
624
|
insights.push({
|
|
497
625
|
type: 'tool_permission_error',
|
|
498
626
|
severity: 'warning',
|
|
499
|
-
message: `检测到 ${permissionErrors.length} 次工具权限错误,实验结论可能被环境问题污染`,
|
|
500
627
|
details: permissionErrors.slice(0, 10),
|
|
501
628
|
});
|
|
502
|
-
suggestions.push('先处理工具权限错误,再解读 agent 分数差异;若是 Glob/rg 权限问题,优先避免在控制实验中依赖该工具');
|
|
503
629
|
}
|
|
504
|
-
function detectTraceIntegrity(report, variants, insights
|
|
630
|
+
function detectTraceIntegrity(report, variants, insights) {
|
|
505
631
|
const summary = report.summary || {};
|
|
506
632
|
const agentAssertionTypes = collectAgentAssertionTypes(report.results || [], variants);
|
|
507
633
|
const needsTraceHeavyCoverage = [...agentAssertionTypes].some((type) => TRACE_HEAVY_AGENT_ASSERTION_TYPES.has(type));
|
|
@@ -524,13 +650,11 @@ function detectTraceIntegrity(report, variants, insights, suggestions) {
|
|
|
524
650
|
insights.push({
|
|
525
651
|
type: 'trace_integrity_gap',
|
|
526
652
|
severity: 'warning',
|
|
527
|
-
message: `${weakCoverage.length} 个 variant 的 trace 覆盖率低于 75%,报告可能不足以解释 agent 行为差异`,
|
|
528
653
|
details: weakCoverage,
|
|
529
654
|
});
|
|
530
|
-
suggestions.push('优先补齐 turns、toolCalls、timing、full output 的采集与落盘,确保报告能解释工具路径和错误恢复过程');
|
|
531
655
|
}
|
|
532
656
|
}
|
|
533
|
-
function detectAgentAssertionDiscrimination(results, variants, insights
|
|
657
|
+
function detectAgentAssertionDiscrimination(results, variants, insights) {
|
|
534
658
|
const assertionTypes = collectAgentAssertionTypes(results, variants);
|
|
535
659
|
const hasTraceHeavyAssertions = [...assertionTypes].some((type) => TRACE_HEAVY_AGENT_ASSERTION_TYPES.has(type));
|
|
536
660
|
if (!hasTraceHeavyAssertions)
|
|
@@ -579,7 +703,6 @@ function detectAgentAssertionDiscrimination(results, variants, insights, suggest
|
|
|
579
703
|
insights.push({
|
|
580
704
|
type: 'agent_assertion_discrimination_low',
|
|
581
705
|
severity: 'warning',
|
|
582
|
-
message: `agent 断言区分度偏低,只有 ${(discriminationRate * 100).toFixed(0)}% 的断言真正拉开了变体差异`,
|
|
583
706
|
details: {
|
|
584
707
|
total: evaluated.length,
|
|
585
708
|
discriminative,
|
|
@@ -588,13 +711,11 @@ function detectAgentAssertionDiscrimination(results, variants, insights, suggest
|
|
|
588
711
|
examples: evaluated.slice(0, 10),
|
|
589
712
|
},
|
|
590
713
|
});
|
|
591
|
-
suggestions.push('重写 agent 断言时,优先约束工具路径、关键文件读取和 turns 上限,避免大量“全过”或“全挂”的弱断言');
|
|
592
714
|
}
|
|
593
715
|
else {
|
|
594
716
|
insights.push({
|
|
595
717
|
type: 'agent_assertion_discrimination_ok',
|
|
596
718
|
severity: 'info',
|
|
597
|
-
message: `agent 断言区分度达标,${(discriminationRate * 100).toFixed(0)}% 的断言能区分变体差异`,
|
|
598
719
|
details: {
|
|
599
720
|
total: evaluated.length,
|
|
600
721
|
discriminative,
|
|
@@ -623,7 +744,6 @@ function detectHighCost(results, variants, insights) {
|
|
|
623
744
|
insights.push({
|
|
624
745
|
type: 'high_cost_sample',
|
|
625
746
|
severity: 'info',
|
|
626
|
-
message: `${expensive.length} 个用例成本显著高于平均值 (>${(avg * 2).toFixed(4)} USD)`,
|
|
627
747
|
details: expensive,
|
|
628
748
|
});
|
|
629
749
|
}
|