oh-my-knowledge 0.22.0 → 0.24.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (184) hide show
  1. package/README.md +63 -63
  2. package/README.zh.md +64 -62
  3. package/dist/src/analysis/report-diagnostics.d.ts +28 -3
  4. package/dist/src/analysis/report-diagnostics.d.ts.map +1 -1
  5. package/dist/src/analysis/report-diagnostics.js +201 -81
  6. package/dist/src/analysis/report-diagnostics.js.map +1 -1
  7. package/dist/src/analysis/sample-diagnostics.d.ts +9 -2
  8. package/dist/src/analysis/sample-diagnostics.d.ts.map +1 -1
  9. package/dist/src/analysis/sample-diagnostics.js +224 -25
  10. package/dist/src/analysis/sample-diagnostics.js.map +1 -1
  11. package/dist/src/authoring/evolver.d.ts +7 -2
  12. package/dist/src/authoring/evolver.d.ts.map +1 -1
  13. package/dist/src/authoring/evolver.js +43 -9
  14. package/dist/src/authoring/evolver.js.map +1 -1
  15. package/dist/src/authoring/generator.d.ts +24 -0
  16. package/dist/src/authoring/generator.d.ts.map +1 -1
  17. package/dist/src/authoring/generator.js +67 -5
  18. package/dist/src/authoring/generator.js.map +1 -1
  19. package/dist/src/cli/coverage-renderer.d.ts +15 -0
  20. package/dist/src/cli/coverage-renderer.d.ts.map +1 -0
  21. package/dist/src/cli/coverage-renderer.js +74 -0
  22. package/dist/src/cli/coverage-renderer.js.map +1 -0
  23. package/dist/src/cli/i18n-dict.d.ts +1 -1
  24. package/dist/src/cli/i18n-dict.d.ts.map +1 -1
  25. package/dist/src/cli/i18n-dict.js +62 -40
  26. package/dist/src/cli/i18n-dict.js.map +1 -1
  27. package/dist/src/cli/index.d.ts +3 -0
  28. package/dist/src/cli/index.d.ts.map +1 -0
  29. package/dist/src/{cli.js → cli/index.js} +123 -384
  30. package/dist/src/cli/index.js.map +1 -0
  31. package/dist/src/cli/parse-run-config.d.ts +56 -0
  32. package/dist/src/cli/parse-run-config.d.ts.map +1 -0
  33. package/dist/src/cli/parse-run-config.js +195 -0
  34. package/dist/src/cli/parse-run-config.js.map +1 -0
  35. package/dist/src/cli/progress.d.ts +25 -0
  36. package/dist/src/cli/progress.d.ts.map +1 -0
  37. package/dist/src/cli/progress.js +62 -0
  38. package/dist/src/cli/progress.js.map +1 -0
  39. package/dist/src/cli/update-check.d.ts +3 -0
  40. package/dist/src/cli/update-check.d.ts.map +1 -0
  41. package/dist/src/cli/update-check.js +37 -0
  42. package/dist/src/cli/update-check.js.map +1 -0
  43. package/dist/src/eval-core/cache.d.ts +7 -5
  44. package/dist/src/eval-core/cache.d.ts.map +1 -1
  45. package/dist/src/eval-core/cache.js +11 -7
  46. package/dist/src/eval-core/cache.js.map +1 -1
  47. package/dist/src/eval-core/comparability.d.ts +11 -0
  48. package/dist/src/eval-core/comparability.d.ts.map +1 -0
  49. package/dist/src/eval-core/comparability.js +271 -0
  50. package/dist/src/eval-core/comparability.js.map +1 -0
  51. package/dist/src/eval-core/dependency-checker.d.ts +1 -1
  52. package/dist/src/eval-core/dependency-checker.js +1 -1
  53. package/dist/src/eval-core/evaluation-execution.d.ts +5 -2
  54. package/dist/src/eval-core/evaluation-execution.d.ts.map +1 -1
  55. package/dist/src/eval-core/evaluation-execution.js +29 -21
  56. package/dist/src/eval-core/evaluation-execution.js.map +1 -1
  57. package/dist/src/eval-core/evaluation-job.d.ts +2 -2
  58. package/dist/src/eval-core/evaluation-job.d.ts.map +1 -1
  59. package/dist/src/eval-core/evaluation-job.js +2 -2
  60. package/dist/src/eval-core/evaluation-job.js.map +1 -1
  61. package/dist/src/eval-core/evaluation-reporting.d.ts +3 -1
  62. package/dist/src/eval-core/evaluation-reporting.d.ts.map +1 -1
  63. package/dist/src/eval-core/evaluation-reporting.js +65 -5
  64. package/dist/src/eval-core/evaluation-reporting.js.map +1 -1
  65. package/dist/src/eval-core/execution-strategy.js +6 -6
  66. package/dist/src/eval-core/execution-strategy.js.map +1 -1
  67. package/dist/src/eval-core/schema.d.ts.map +1 -1
  68. package/dist/src/eval-core/schema.js +14 -1
  69. package/dist/src/eval-core/schema.js.map +1 -1
  70. package/dist/src/eval-core/verdict.d.ts +3 -3
  71. package/dist/src/eval-core/verdict.js +3 -3
  72. package/dist/src/eval-workflows/batch-evaluation-workflow.d.ts +111 -0
  73. package/dist/src/eval-workflows/batch-evaluation-workflow.d.ts.map +1 -0
  74. package/dist/src/eval-workflows/batch-evaluation-workflow.js +215 -0
  75. package/dist/src/eval-workflows/batch-evaluation-workflow.js.map +1 -0
  76. package/dist/src/eval-workflows/evaluation-pipeline.d.ts +7 -5
  77. package/dist/src/eval-workflows/evaluation-pipeline.d.ts.map +1 -1
  78. package/dist/src/eval-workflows/evaluation-pipeline.js +13 -8
  79. package/dist/src/eval-workflows/evaluation-pipeline.js.map +1 -1
  80. package/dist/src/eval-workflows/evaluation-preparation.d.ts +10 -8
  81. package/dist/src/eval-workflows/evaluation-preparation.d.ts.map +1 -1
  82. package/dist/src/eval-workflows/evaluation-preparation.js +8 -4
  83. package/dist/src/eval-workflows/evaluation-preparation.js.map +1 -1
  84. package/dist/src/eval-workflows/run-evaluation.d.ts +23 -20
  85. package/dist/src/eval-workflows/run-evaluation.d.ts.map +1 -1
  86. package/dist/src/eval-workflows/run-evaluation.js +34 -25
  87. package/dist/src/eval-workflows/run-evaluation.js.map +1 -1
  88. package/dist/src/executors/claude-cli.d.ts.map +1 -1
  89. package/dist/src/executors/claude-cli.js +12 -7
  90. package/dist/src/executors/claude-cli.js.map +1 -1
  91. package/dist/src/executors/claude-sdk.d.ts +1 -1
  92. package/dist/src/executors/claude-sdk.js +1 -1
  93. package/dist/src/executors/codex-cli-trace.d.ts +10 -0
  94. package/dist/src/executors/codex-cli-trace.d.ts.map +1 -0
  95. package/dist/src/executors/codex-cli-trace.js +123 -0
  96. package/dist/src/executors/codex-cli-trace.js.map +1 -0
  97. package/dist/src/executors/codex-cli.d.ts +18 -0
  98. package/dist/src/executors/codex-cli.d.ts.map +1 -0
  99. package/dist/src/executors/codex-cli.js +254 -0
  100. package/dist/src/executors/codex-cli.js.map +1 -0
  101. package/dist/src/executors/codex-sdk.d.ts +18 -0
  102. package/dist/src/executors/codex-sdk.d.ts.map +1 -0
  103. package/dist/src/executors/codex-sdk.js +214 -0
  104. package/dist/src/executors/codex-sdk.js.map +1 -0
  105. package/dist/src/executors/gemini.d.ts.map +1 -1
  106. package/dist/src/executors/gemini.js +28 -24
  107. package/dist/src/executors/gemini.js.map +1 -1
  108. package/dist/src/executors/index.d.ts.map +1 -1
  109. package/dist/src/executors/index.js +7 -2
  110. package/dist/src/executors/index.js.map +1 -1
  111. package/dist/src/executors/runtime-fingerprint.d.ts +7 -0
  112. package/dist/src/executors/runtime-fingerprint.d.ts.map +1 -0
  113. package/dist/src/executors/runtime-fingerprint.js +277 -0
  114. package/dist/src/executors/runtime-fingerprint.js.map +1 -0
  115. package/dist/src/executors/script.d.ts.map +1 -1
  116. package/dist/src/executors/script.js +48 -56
  117. package/dist/src/executors/script.js.map +1 -1
  118. package/dist/src/executors/shared.d.ts +78 -1
  119. package/dist/src/executors/shared.d.ts.map +1 -1
  120. package/dist/src/executors/shared.js +203 -1
  121. package/dist/src/executors/shared.js.map +1 -1
  122. package/dist/src/grading/assertions.d.ts.map +1 -1
  123. package/dist/src/grading/assertions.js +21 -6
  124. package/dist/src/grading/assertions.js.map +1 -1
  125. package/dist/src/grading/gold-dataset.d.ts +1 -1
  126. package/dist/src/grading/gold-dataset.js +1 -1
  127. package/dist/src/grading/index.d.ts.map +1 -1
  128. package/dist/src/grading/index.js +11 -0
  129. package/dist/src/grading/index.js.map +1 -1
  130. package/dist/src/grading/judge.d.ts.map +1 -1
  131. package/dist/src/grading/judge.js +75 -6
  132. package/dist/src/grading/judge.js.map +1 -1
  133. package/dist/src/inputs/eval-config.js +2 -2
  134. package/dist/src/inputs/eval-config.js.map +1 -1
  135. package/dist/src/inputs/load-samples.d.ts.map +1 -1
  136. package/dist/src/inputs/load-samples.js +30 -4
  137. package/dist/src/inputs/load-samples.js.map +1 -1
  138. package/dist/src/inputs/skill-loader.d.ts +2 -2
  139. package/dist/src/inputs/skill-loader.d.ts.map +1 -1
  140. package/dist/src/inputs/skill-loader.js +2 -2
  141. package/dist/src/inputs/skill-loader.js.map +1 -1
  142. package/dist/src/renderer/html-renderer.d.ts +5 -4
  143. package/dist/src/renderer/html-renderer.d.ts.map +1 -1
  144. package/dist/src/renderer/html-renderer.js +217 -93
  145. package/dist/src/renderer/html-renderer.js.map +1 -1
  146. package/dist/src/renderer/layout.d.ts +2 -1
  147. package/dist/src/renderer/layout.d.ts.map +1 -1
  148. package/dist/src/renderer/layout.js +30 -42
  149. package/dist/src/renderer/layout.js.map +1 -1
  150. package/dist/src/renderer/summary.d.ts +3 -3
  151. package/dist/src/renderer/summary.d.ts.map +1 -1
  152. package/dist/src/renderer/summary.js +232 -55
  153. package/dist/src/renderer/summary.js.map +1 -1
  154. package/dist/src/renderer/trends.d.ts.map +1 -1
  155. package/dist/src/renderer/trends.js +5 -3
  156. package/dist/src/renderer/trends.js.map +1 -1
  157. package/dist/src/server/report-server.js +4 -4
  158. package/dist/src/server/report-server.js.map +1 -1
  159. package/dist/src/server/report-store.d.ts +7 -5
  160. package/dist/src/server/report-store.d.ts.map +1 -1
  161. package/dist/src/server/report-store.js +39 -11
  162. package/dist/src/server/report-store.js.map +1 -1
  163. package/dist/src/types/eval.d.ts +27 -5
  164. package/dist/src/types/eval.d.ts.map +1 -1
  165. package/dist/src/types/executor.d.ts +5 -0
  166. package/dist/src/types/executor.d.ts.map +1 -1
  167. package/dist/src/types/judge.d.ts +10 -0
  168. package/dist/src/types/judge.d.ts.map +1 -1
  169. package/dist/src/types/report.d.ts +151 -29
  170. package/dist/src/types/report.d.ts.map +1 -1
  171. package/dist/src/types/storage.d.ts +7 -7
  172. package/dist/src/types/storage.d.ts.map +1 -1
  173. package/package.json +14 -5
  174. package/dist/src/cli.d.ts +0 -3
  175. package/dist/src/cli.d.ts.map +0 -1
  176. package/dist/src/cli.js.map +0 -1
  177. package/dist/src/eval-workflows/each-evaluation-workflow.d.ts +0 -153
  178. package/dist/src/eval-workflows/each-evaluation-workflow.d.ts.map +0 -1
  179. package/dist/src/eval-workflows/each-evaluation-workflow.js +0 -178
  180. package/dist/src/eval-workflows/each-evaluation-workflow.js.map +0 -1
  181. package/dist/src/executors/openai-cli.d.ts +0 -3
  182. package/dist/src/executors/openai-cli.d.ts.map +0 -1
  183. package/dist/src/executors/openai-cli.js +0 -60
  184. package/dist/src/executors/openai-cli.js.map +0 -1
@@ -1,41 +1,136 @@
1
1
  /**
2
2
  * Auto-analysis: detect patterns and generate insights from evaluation results.
3
3
  */
4
+ import { normalizeCapability } from './sample-diagnostics.js';
4
5
  /**
5
- * Analyze an evaluation report and produce insights + suggestions.
6
+ * Analyze an evaluation report and produce structured insights.
6
7
  */
7
- export function analyzeResults(report) {
8
+ export function analyzeResults(report, opts = {}) {
8
9
  const insights = [];
9
- const suggestions = [];
10
10
  const variants = report.meta?.variants || [];
11
11
  const results = report.results || [];
12
+ // sampleQuality aggregate is built from sample metadata only,
13
+ // independent of result data. Computed even when results.length === 0 or
14
+ // variants.length < 2 (e.g. dry-run / single-variant analysis).
15
+ const sampleQuality = opts.samples
16
+ ? buildSampleQualityAggregate(opts.samples)
17
+ : undefined;
12
18
  if (results.length === 0 || variants.length < 2) {
13
- return { insights, suggestions };
19
+ return { insights, ...(sampleQuality && { sampleQuality }) };
14
20
  }
15
21
  // 1. Low-discrimination assertions
16
- detectLowDiscrimination(results, variants, insights, suggestions);
22
+ detectLowDiscrimination(results, variants, insights);
17
23
  // 2. Uniform scores across variants
18
- detectUniformScores(results, variants, insights, suggestions);
24
+ detectUniformScores(results, variants, insights);
19
25
  // 3. All-pass / all-fail assertions
20
- detectAllPassFail(results, variants, insights, suggestions);
26
+ detectAllPassFail(results, variants, insights);
21
27
  // 4. High-cost samples
22
28
  detectHighCost(results, variants, insights);
23
29
  // 5. Efficiency gap (turns & cost)
24
- detectEfficiencyGap(report, variants, insights, suggestions);
30
+ detectEfficiencyGap(report, variants, insights);
25
31
  // 6. Agent tool usage patterns
26
- detectToolPatterns(report, variants, insights, suggestions);
32
+ detectToolPatterns(report, variants, insights);
27
33
  // 7. Tooling / permission issues
28
- detectToolPermissionIssues(results, variants, insights, suggestions);
34
+ detectToolPermissionIssues(results, variants, insights);
29
35
  // 8. Trace integrity
30
- detectTraceIntegrity(report, variants, insights, suggestions);
36
+ detectTraceIntegrity(report, variants, insights);
31
37
  // 9. Agent assertion discrimination
32
- detectAgentAssertionDiscrimination(results, variants, insights, suggestions);
38
+ detectAgentAssertionDiscrimination(results, variants, insights);
33
39
  // 10. Suggest --repeat when score variance is high and no repeat data
34
- detectNeedRepeat(report, results, variants, insights, suggestions);
35
- const summary = generateSummary(report, variants);
36
- return { summary, insights, suggestions };
40
+ detectNeedRepeat(report, results, variants, insights);
41
+ return {
42
+ insights,
43
+ ...(sampleQuality && { sampleQuality }),
44
+ };
37
45
  }
38
- function generateSummary(report, variants) {
46
+ /**
47
+ * Build sample design science aggregate from sample metadata.
48
+ *
49
+ * Pure function — no result/score data needed. Reads:
50
+ * - `Sample.capability` (string[], normalized case-insensitive + dash/camel/underscore stripped)
51
+ * - `Sample.difficulty` ('easy' | 'medium' | 'hard')
52
+ * - `Sample.construct` (free-form string)
53
+ * - `Sample.provenance` ('human' | 'llm-generated' | 'production-trace')
54
+ * - `Sample.rubric` (for avgRubricLength)
55
+ *
56
+ * Missing fields are bucketed under the `unspecified` key in the relevant
57
+ * distribution map, so users see "I have N samples without difficulty declared".
58
+ *
59
+ * Used by `bench diagnose` CLI to surface coverage gaps. Does NOT participate
60
+ * in grading / judge / verdict. See docs/sample-design-spec.md.
61
+ */
62
+ export function buildSampleQualityAggregate(samples) {
63
+ const capabilityCoverage = {};
64
+ const difficultyDistribution = {
65
+ easy: 0, medium: 0, hard: 0, unspecified: 0,
66
+ };
67
+ const constructDistribution = {};
68
+ const provenanceBreakdown = {};
69
+ let totalRubricLength = 0;
70
+ let rubricCount = 0;
71
+ let withCapability = 0;
72
+ let withDifficulty = 0;
73
+ let withConstruct = 0;
74
+ let withProvenance = 0;
75
+ for (const sample of samples) {
76
+ // capability — normalize case + dash/camel/underscore so 'api-selection' / 'apiSelection' / 'API_Selection' merge.
77
+ if (Array.isArray(sample.capability) && sample.capability.length > 0) {
78
+ withCapability++;
79
+ const seen = new Set();
80
+ for (const rawCap of sample.capability) {
81
+ if (typeof rawCap !== 'string')
82
+ continue;
83
+ const cap = normalizeCapability(rawCap);
84
+ if (seen.has(cap))
85
+ continue; // 同 sample 内同 capability 重复声明只计 1
86
+ seen.add(cap);
87
+ capabilityCoverage[cap] = (capabilityCoverage[cap] || 0) + 1;
88
+ }
89
+ }
90
+ // difficulty
91
+ if (sample.difficulty) {
92
+ withDifficulty++;
93
+ difficultyDistribution[sample.difficulty]++;
94
+ }
95
+ else {
96
+ difficultyDistribution.unspecified++;
97
+ }
98
+ // construct (free-form)
99
+ if (sample.construct) {
100
+ withConstruct++;
101
+ constructDistribution[sample.construct] = (constructDistribution[sample.construct] || 0) + 1;
102
+ }
103
+ else {
104
+ constructDistribution.unspecified = (constructDistribution.unspecified || 0) + 1;
105
+ }
106
+ // provenance
107
+ if (sample.provenance) {
108
+ withProvenance++;
109
+ provenanceBreakdown[sample.provenance] = (provenanceBreakdown[sample.provenance] || 0) + 1;
110
+ }
111
+ else {
112
+ provenanceBreakdown.unspecified = (provenanceBreakdown.unspecified || 0) + 1;
113
+ }
114
+ // rubric length(only counted if present, NaN-safe)
115
+ if (sample.rubric) {
116
+ totalRubricLength += sample.rubric.trim().length;
117
+ rubricCount++;
118
+ }
119
+ }
120
+ return {
121
+ capabilityCoverage,
122
+ difficultyDistribution,
123
+ constructDistribution,
124
+ provenanceBreakdown,
125
+ avgRubricLength: rubricCount > 0 ? Math.round(totalRubricLength / rubricCount) : 0,
126
+ sampleCountWithCapability: withCapability,
127
+ sampleCountWithDifficulty: withDifficulty,
128
+ sampleCountWithConstruct: withConstruct,
129
+ sampleCountWithProvenance: withProvenance,
130
+ };
131
+ }
132
+ export function generateAnalysisSummary(report, lang = 'zh') {
133
+ const variants = report.meta?.variants || [];
39
134
  if (variants.length < 2)
40
135
  return undefined;
41
136
  const stats = report.summary || {};
@@ -69,15 +164,25 @@ function generateSummary(report, variants) {
69
164
  if (scoreDiff != null && tScore != null && cScore != null) {
70
165
  const absDiff = Math.abs(scoreDiff);
71
166
  if (absDiff < 0.1) {
72
- lines.push(`【结论】${test} 与 ${control} 综合得分持平(${tScore.toFixed(2)} vs ${cScore.toFixed(2)}),质量无显著差异。`);
167
+ lines.push(lang === 'zh'
168
+ ? `【结论】${test} 与 ${control} 综合得分持平(${tScore.toFixed(2)} vs ${cScore.toFixed(2)}),质量无显著差异。`
169
+ : `【Conclusion】${test} and ${control} are effectively tied on composite score (${tScore.toFixed(2)} vs ${cScore.toFixed(2)}); no clear quality difference.`);
73
170
  }
74
171
  else if (scoreDiff > 0) {
75
- const tag = absDiff > 0.3 ? '明显领先' : '略优';
76
- lines.push(`【结论】${test} 综合得分 ${tag}(${tScore.toFixed(2)} vs ${cScore.toFixed(2)},+${scoreDiff.toFixed(2)})。`);
172
+ const tag = lang === 'zh'
173
+ ? (absDiff > 0.3 ? '明显领先' : '略优')
174
+ : (absDiff > 0.3 ? 'clearly ahead' : 'slightly ahead');
175
+ lines.push(lang === 'zh'
176
+ ? `【结论】${test} 综合得分 ${tag}(${tScore.toFixed(2)} vs ${cScore.toFixed(2)},+${scoreDiff.toFixed(2)})。`
177
+ : `【Conclusion】${test} is ${tag} on composite score (${tScore.toFixed(2)} vs ${cScore.toFixed(2)}, +${scoreDiff.toFixed(2)}).`);
77
178
  }
78
179
  else {
79
- const tag = absDiff > 0.3 ? '明显落后' : '略低';
80
- lines.push(`【结论】${test} 综合得分 ${tag}(${tScore.toFixed(2)} vs ${cScore.toFixed(2)},${scoreDiff.toFixed(2)})。`);
180
+ const tag = lang === 'zh'
181
+ ? (absDiff > 0.3 ? '明显落后' : '略低')
182
+ : (absDiff > 0.3 ? 'clearly behind' : 'slightly lower');
183
+ lines.push(lang === 'zh'
184
+ ? `【结论】${test} 综合得分 ${tag}(${tScore.toFixed(2)} vs ${cScore.toFixed(2)},${scoreDiff.toFixed(2)})。`
185
+ : `【Conclusion】${test} is ${tag} on composite score (${tScore.toFixed(2)} vs ${cScore.toFixed(2)}, ${scoreDiff.toFixed(2)}).`);
81
186
  }
82
187
  }
83
188
  // ── Section 2: Key differentiators with concrete numbers ──
@@ -90,23 +195,31 @@ function generateSummary(report, variants) {
90
195
  // Both same — don't mention (e.g. both 5/5 is not interesting)
91
196
  }
92
197
  else if (tFact > cFact) {
93
- diffs.push(`事实性 ${tFact.toFixed(1)} vs ${cFact.toFixed(1)}(↑${(tFact - cFact).toFixed(1)})`);
198
+ diffs.push(lang === 'zh'
199
+ ? `事实性 ${tFact.toFixed(1)} vs ${cFact.toFixed(1)}(↑${(tFact - cFact).toFixed(1)})`
200
+ : `Fact score ${tFact.toFixed(1)} vs ${cFact.toFixed(1)} (up ${(tFact - cFact).toFixed(1)})`);
94
201
  }
95
202
  else {
96
- diffs.push(`事实性 ${tFact.toFixed(1)} vs ${cFact.toFixed(1)}(↓${(cFact - tFact).toFixed(1)})`);
203
+ diffs.push(lang === 'zh'
204
+ ? `事实性 ${tFact.toFixed(1)} vs ${cFact.toFixed(1)}(↓${(cFact - tFact).toFixed(1)})`
205
+ : `Fact score ${tFact.toFixed(1)} vs ${cFact.toFixed(1)} (down ${(cFact - tFact).toFixed(1)})`);
97
206
  }
98
207
  }
99
208
  const tBehavior = ts.avgBehaviorScore;
100
209
  const cBehavior = cs.avgBehaviorScore;
101
210
  if (tBehavior != null && cBehavior != null && Math.abs(tBehavior - cBehavior) > 0.3) {
102
211
  const dir = tBehavior > cBehavior ? '↑' : '↓';
103
- diffs.push(`行为合规 ${tBehavior.toFixed(1)} vs ${cBehavior.toFixed(1)}(${dir}${Math.abs(tBehavior - cBehavior).toFixed(1)})`);
212
+ diffs.push(lang === 'zh'
213
+ ? `行为合规 ${tBehavior.toFixed(1)} vs ${cBehavior.toFixed(1)}(${dir}${Math.abs(tBehavior - cBehavior).toFixed(1)})`
214
+ : `Behavior score ${tBehavior.toFixed(1)} vs ${cBehavior.toFixed(1)} (${dir === '↑' ? 'up' : 'down'} ${Math.abs(tBehavior - cBehavior).toFixed(1)})`);
104
215
  }
105
216
  const tJudge = ts.avgJudgeScore;
106
217
  const cJudge = cs.avgJudgeScore;
107
218
  if (tJudge != null && cJudge != null && Math.abs(tJudge - cJudge) >= 0.5) {
108
219
  const dir = tJudge > cJudge ? '↑' : '↓';
109
- diffs.push(`LLM 评价 ${tJudge.toFixed(1)} vs ${cJudge.toFixed(1)}(${dir}${Math.abs(tJudge - cJudge).toFixed(1)})`);
220
+ diffs.push(lang === 'zh'
221
+ ? `LLM 评价 ${tJudge.toFixed(1)} vs ${cJudge.toFixed(1)}(${dir}${Math.abs(tJudge - cJudge).toFixed(1)})`
222
+ : `LLM judge ${tJudge.toFixed(1)} vs ${cJudge.toFixed(1)} (${dir === '↑' ? 'up' : 'down'} ${Math.abs(tJudge - cJudge).toFixed(1)})`);
110
223
  }
111
224
  // Efficiency — with percentages
112
225
  const cTurns = cs.avgNumTurns;
@@ -114,10 +227,14 @@ function generateSummary(report, variants) {
114
227
  if (cTurns > 0 && tTurns > 0 && cTurns !== tTurns) {
115
228
  const pct = Math.abs(((tTurns - cTurns) / cTurns) * 100).toFixed(0);
116
229
  if (tTurns < cTurns) {
117
- diffs.push(`轮次 ${tTurns.toFixed(1)} vs ${cTurns.toFixed(1)}(↓${pct}%,路径更高效)`);
230
+ diffs.push(lang === 'zh'
231
+ ? `轮次 ${tTurns.toFixed(1)} vs ${cTurns.toFixed(1)}(↓${pct}%,路径更高效)`
232
+ : `Turns ${tTurns.toFixed(1)} vs ${cTurns.toFixed(1)} (down ${pct}%, more efficient path)`);
118
233
  }
119
234
  else {
120
- diffs.push(`轮次 ${tTurns.toFixed(1)} vs ${cTurns.toFixed(1)}(↑${pct}%)`);
235
+ diffs.push(lang === 'zh'
236
+ ? `轮次 ${tTurns.toFixed(1)} vs ${cTurns.toFixed(1)}(↑${pct}%)`
237
+ : `Turns ${tTurns.toFixed(1)} vs ${cTurns.toFixed(1)} (up ${pct}%)`);
121
238
  }
122
239
  }
123
240
  // Cost — with percentages
@@ -126,10 +243,14 @@ function generateSummary(report, variants) {
126
243
  if (cCost > 0 && tCost > 0 && Math.abs(tCost - cCost) / cCost > 0.05) {
127
244
  const pct = Math.abs(((tCost - cCost) / cCost) * 100).toFixed(0);
128
245
  if (tCost < cCost) {
129
- diffs.push(`单用例成本 $${tCost.toFixed(4)} vs $${cCost.toFixed(4)}(↓${pct}%)`);
246
+ diffs.push(lang === 'zh'
247
+ ? `单用例成本 $${tCost.toFixed(4)} vs $${cCost.toFixed(4)}(↓${pct}%)`
248
+ : `Cost per sample $${tCost.toFixed(4)} vs $${cCost.toFixed(4)} (down ${pct}%)`);
130
249
  }
131
250
  else {
132
- diffs.push(`单用例成本 $${tCost.toFixed(4)} vs $${cCost.toFixed(4)}(↑${pct}%)`);
251
+ diffs.push(lang === 'zh'
252
+ ? `单用例成本 $${tCost.toFixed(4)} vs $${cCost.toFixed(4)}(↑${pct}%)`
253
+ : `Cost per sample $${tCost.toFixed(4)} vs $${cCost.toFixed(4)} (up ${pct}%)`);
133
254
  }
134
255
  }
135
256
  // Duration — with percentages
@@ -140,20 +261,28 @@ function generateSummary(report, variants) {
140
261
  const tSec = (tDur / 1000).toFixed(1);
141
262
  const cSec = (cDur / 1000).toFixed(1);
142
263
  if (tDur < cDur) {
143
- diffs.push(`耗时 ${tSec}s vs ${cSec}s(↓${pct}%)`);
264
+ diffs.push(lang === 'zh'
265
+ ? `耗时 ${tSec}s vs ${cSec}s(↓${pct}%)`
266
+ : `Duration ${tSec}s vs ${cSec}s (down ${pct}%)`);
144
267
  }
145
268
  else {
146
- diffs.push(`耗时 ${tSec}s vs ${cSec}s(↑${pct}%)`);
269
+ diffs.push(lang === 'zh'
270
+ ? `耗时 ${tSec}s vs ${cSec}s(↑${pct}%)`
271
+ : `Duration ${tSec}s vs ${cSec}s (up ${pct}%)`);
147
272
  }
148
273
  }
149
274
  // Tool usage difference
150
275
  const tTools = ts.avgToolCalls;
151
276
  const cTools = cs.avgToolCalls;
152
277
  if (tTools != null && cTools != null && Math.abs(tTools - cTools) > 0.5) {
153
- diffs.push(`工具调用 ${tTools.toFixed(1)} vs ${cTools.toFixed(1)} 次`);
278
+ diffs.push(lang === 'zh'
279
+ ? `工具调用 ${tTools.toFixed(1)} vs ${cTools.toFixed(1)} 次`
280
+ : `Tool calls ${tTools.toFixed(1)} vs ${cTools.toFixed(1)}`);
154
281
  }
155
282
  if (diffs.length > 0) {
156
- lines.push(`【关键差异】${diffs.join(';')}。`);
283
+ lines.push(lang === 'zh'
284
+ ? `【关键差异】${diffs.join(';')}。`
285
+ : `【Key differences】${diffs.join('; ')}.`);
157
286
  }
158
287
  // ── Section 3: Synthesis — connect the dots ──
159
288
  const synthesis = [];
@@ -161,22 +290,32 @@ function generateSummary(report, variants) {
161
290
  if (scoreDiff != null && cCost > 0 && tCost > 0) {
162
291
  const costRatio = (tCost - cCost) / cCost;
163
292
  if (Math.abs(scoreDiff) < 0.1 && costRatio < -0.15) {
164
- synthesis.push(`质量相当但成本显著降低,${test} 是更经济的选择`);
293
+ synthesis.push(lang === 'zh'
294
+ ? `质量相当但成本显著降低,${test} 是更经济的选择`
295
+ : `similar quality with materially lower cost; ${test} is the more economical choice`);
165
296
  }
166
297
  else if (scoreDiff > 0.1 && costRatio > 0.15) {
167
- synthesis.push(`质量提升伴随成本上涨,需权衡投入产出比`);
298
+ synthesis.push(lang === 'zh'
299
+ ? '质量提升伴随成本上涨,需权衡投入产出比'
300
+ : 'quality improved, but cost also increased; weigh the return on investment');
168
301
  }
169
302
  else if (scoreDiff > 0.1 && costRatio <= 0) {
170
- synthesis.push(`质量与成本双优,${test} 全面领先`);
303
+ synthesis.push(lang === 'zh'
304
+ ? `质量与成本双优,${test} 全面领先`
305
+ : `${test} leads on both quality and cost`);
171
306
  }
172
307
  else if (scoreDiff < -0.1 && costRatio < -0.15) {
173
- synthesis.push(`成本虽降但质量下滑,需评估质量底线是否可接受`);
308
+ synthesis.push(lang === 'zh'
309
+ ? '成本虽降但质量下滑,需评估质量底线是否可接受'
310
+ : 'cost decreased, but quality dropped; check whether the quality floor is still acceptable');
174
311
  }
175
312
  }
176
313
  // Tool success rate concern
177
314
  const tToolSuccess = ts.toolSuccessRate;
178
315
  if (tToolSuccess != null && tToolSuccess < 1 && tToolSuccess >= 0.5) {
179
- synthesis.push(`${test} 存在工具调用失败(成功率 ${(tToolSuccess * 100).toFixed(0)}%),可能拉低了得分`);
316
+ synthesis.push(lang === 'zh'
317
+ ? `${test} 存在工具调用失败(成功率 ${(tToolSuccess * 100).toFixed(0)}%),可能拉低了得分`
318
+ : `${test} had tool-call failures (${(tToolSuccess * 100).toFixed(0)}% success), which may have pulled the score down`);
180
319
  }
181
320
  // Variance / significance from --repeat
182
321
  if (report.variance) {
@@ -189,23 +328,33 @@ function generateSummary(report, variants) {
189
328
  const primaryVal = es.primary === 'g' ? es.hedgesG : es.cohensD;
190
329
  const secondaryLabel = es.primary === 'g' ? 'd' : 'g';
191
330
  const secondaryVal = es.primary === 'g' ? es.cohensD : es.hedgesG;
192
- esText = `,效应量 ${es.primary}=${primaryVal.toFixed(2)}(${es.magnitude},${secondaryLabel}=${secondaryVal.toFixed(2)})`;
331
+ esText = lang === 'zh'
332
+ ? `,效应量 ${es.primary}=${primaryVal.toFixed(2)}(${es.magnitude},${secondaryLabel}=${secondaryVal.toFixed(2)})`
333
+ : `, effect size ${es.primary}=${primaryVal.toFixed(2)} (${es.magnitude}, ${secondaryLabel}=${secondaryVal.toFixed(2)})`;
193
334
  }
194
335
  if (comp.significant) {
195
- synthesis.push(`${v.runs} 轮重复评测显示差异具有统计显著性(t=${comp.tStatistic.toFixed(2)}, df=${comp.df.toFixed(1)}, p<0.05${esText})`);
336
+ synthesis.push(lang === 'zh'
337
+ ? `${v.runs} 轮重复评测显示差异具有统计显著性(t=${comp.tStatistic.toFixed(2)}, df=${comp.df.toFixed(1)}, p<0.05${esText})`
338
+ : `${v.runs} repeated runs show a statistically significant difference (t=${comp.tStatistic.toFixed(2)}, df=${comp.df.toFixed(1)}, p<0.05${esText})`);
196
339
  }
197
340
  else {
198
- synthesis.push(`${v.runs} 轮重复评测未达到统计显著性(t=${comp.tStatistic.toFixed(2)}, df=${comp.df.toFixed(1)}${esText}),差异可能源于随机波动`);
341
+ synthesis.push(lang === 'zh'
342
+ ? `${v.runs} 轮重复评测未达到统计显著性(t=${comp.tStatistic.toFixed(2)}, df=${comp.df.toFixed(1)}${esText}),差异可能源于随机波动`
343
+ : `${v.runs} repeated runs did not reach statistical significance (t=${comp.tStatistic.toFixed(2)}, df=${comp.df.toFixed(1)}${esText}); the gap may be random variation`);
199
344
  }
200
345
  }
201
346
  }
202
347
  const testVd = v.perVariant[test];
203
348
  if (testVd) {
204
- synthesis.push(`${test} 跨轮 95% 置信区间 [${testVd.lower.toFixed(2)}, ${testVd.upper.toFixed(2)}]`);
349
+ synthesis.push(lang === 'zh'
350
+ ? `${test} 跨轮 95% 置信区间 [${testVd.lower.toFixed(2)}, ${testVd.upper.toFixed(2)}]`
351
+ : `${test} cross-run 95% confidence interval [${testVd.lower.toFixed(2)}, ${testVd.upper.toFixed(2)}]`);
205
352
  }
206
353
  }
207
354
  if (synthesis.length > 0) {
208
- lines.push(`【综合洞察】${synthesis.join(';')}。`);
355
+ lines.push(lang === 'zh'
356
+ ? `【综合洞察】${synthesis.join(';')}。`
357
+ : `【Synthesis】${synthesis.join('; ')}.`);
209
358
  }
210
359
  // Caveats and recommendations are handled by the issues table below,
211
360
  // so the summary focuses only on verdict + differentiators + synthesis.
@@ -243,7 +392,7 @@ function collectAgentAssertionTypes(results, variants) {
243
392
  }
244
393
  return types;
245
394
  }
246
- function detectLowDiscrimination(results, variants, insights, suggestions) {
395
+ function detectLowDiscrimination(results, variants, insights) {
247
396
  // For each sample, check if all variants have the same assertion pass/fail pattern
248
397
  const allPassedPatterns = [];
249
398
  const allFailedPatterns = [];
@@ -282,22 +431,18 @@ function detectLowDiscrimination(results, variants, insights, suggestions) {
282
431
  insights.push({
283
432
  type: 'low_discrimination_all_passed',
284
433
  severity: 'info',
285
- message: `${allPassedPatterns.length} 个断言所有变体均通过,baseline 也能答对,区分度低`,
286
434
  details: allPassedPatterns,
287
435
  });
288
- suggestions.push('对于所有变体均通过的断言,考虑替换为检测 skill 文档中独有细节的断言(如特定参数名、配置值)');
289
436
  }
290
437
  if (allFailedPatterns.length > 0) {
291
438
  insights.push({
292
439
  type: 'low_discrimination_all_failed',
293
440
  severity: 'warning',
294
- message: `${allFailedPatterns.length} 个断言所有变体均失败,断言可能过于严格或存在配置错误`,
295
441
  details: allFailedPatterns,
296
442
  });
297
- suggestions.push('对于所有变体均失败的断言,检查断言条件是否正确,或降低匹配要求');
298
443
  }
299
444
  }
300
- function detectUniformScores(results, variants, insights, suggestions) {
445
+ function detectUniformScores(results, variants, insights) {
301
446
  let uniformCount = 0;
302
447
  const uniformSamples = [];
303
448
  for (const r of results) {
@@ -317,15 +462,11 @@ function detectUniformScores(results, variants, insights, suggestions) {
317
462
  insights.push({
318
463
  type: 'uniform_scores',
319
464
  severity: uniformCount === results.length ? 'warning' : 'info',
320
- message: `${uniformCount}/${results.length} 个用例在各变体间分差 < 0.5,区分度较低`,
321
465
  details: uniformSamples,
322
466
  });
323
- if (uniformCount === results.length) {
324
- suggestions.push('所有用例分数差异都很小,建议增加更有挑战性的测试用例或更严格的评分标准');
325
- }
326
467
  }
327
468
  }
328
- function detectAllPassFail(results, variants, insights, suggestions) {
469
+ function detectAllPassFail(results, variants, insights) {
329
470
  let allPassCount = 0;
330
471
  let allFailCount = 0;
331
472
  for (const r of results) {
@@ -344,22 +485,18 @@ function detectAllPassFail(results, variants, insights, suggestions) {
344
485
  insights.push({
345
486
  type: 'all_pass',
346
487
  severity: 'warning',
347
- message: '所有断言在所有变体上全部通过,断言可能过于宽松',
348
488
  details: { allPassCount },
349
489
  });
350
- suggestions.push('所有断言都通过了,考虑增加更严格的断言来更好地区分变体质量');
351
490
  }
352
491
  if (allFailCount === totalEntries && totalEntries > 0) {
353
492
  insights.push({
354
493
  type: 'all_fail',
355
494
  severity: 'error',
356
- message: '所有断言在所有变体上全部失败,请检查断言配置是否正确',
357
495
  details: { allFailCount },
358
496
  });
359
- suggestions.push('所有断言都失败了,请检查评测配置是否有误');
360
497
  }
361
498
  }
362
- function detectNeedRepeat(report, results, variants, insights, suggestions) {
499
+ function detectNeedRepeat(report, results, variants, insights) {
363
500
  // Skip if already has variance data (i.e. --repeat was used)
364
501
  if (report.variance)
365
502
  return;
@@ -373,15 +510,13 @@ function detectNeedRepeat(report, results, variants, insights, suggestions) {
373
510
  insights.push({
374
511
  type: 'suggest_repeat',
375
512
  severity: 'info',
376
- message: `${v} 的分数跨度较大(${s.minCompositeScore}~${s.maxCompositeScore}),建议使用 --repeat 3 多轮评测以获取方差分析和统计显著性检验`,
377
513
  details: { variant: v, min: s.minCompositeScore, max: s.maxCompositeScore, spread },
378
514
  });
379
- suggestions.push(`运行 omk bench run --repeat 3 获取置信区间和 t 检验结果,量化变体间差异的统计显著性`);
380
515
  return; // Only suggest once
381
516
  }
382
517
  }
383
518
  }
384
- function detectEfficiencyGap(report, variants, insights, suggestions) {
519
+ function detectEfficiencyGap(report, variants, insights) {
385
520
  if (variants.length < 2)
386
521
  return;
387
522
  const summary = report.summary || {};
@@ -422,14 +557,12 @@ function detectEfficiencyGap(report, variants, insights, suggestions) {
422
557
  insights.push({
423
558
  type: 'efficiency_gap',
424
559
  severity: 'info',
425
- message: details.join(';'),
426
560
  details: { baseline: variants[0], variant: variants[i], baseTurns, otherTurns, baseCost, otherCost },
427
561
  });
428
- suggestions.push(`${variants[i]} 在效率维度与 ${variants[0]} 存在显著差异,这对导航型 Skill 是重要的价值体现`);
429
562
  }
430
563
  }
431
564
  }
432
- function detectToolPatterns(report, variants, insights, suggestions) {
565
+ function detectToolPatterns(report, variants, insights) {
433
566
  const summary = report.summary || {};
434
567
  const hasTools = variants.some((v) => summary[v]?.avgToolCalls != null && summary[v].avgToolCalls > 0);
435
568
  if (!hasTools)
@@ -444,10 +577,8 @@ function detectToolPatterns(report, variants, insights, suggestions) {
444
577
  insights.push({
445
578
  type: 'low_tool_success_rate',
446
579
  severity: 'warning',
447
- message: `${v} 的工具调用成功率仅 ${(s.toolSuccessRate * 100).toFixed(0)}%,可能存在工具选择或参数问题`,
448
580
  details: { variant: v, toolSuccessRate: s.toolSuccessRate, avgToolCalls: s.avgToolCalls },
449
581
  });
450
- suggestions.push(`检查 ${v} 的工具调用失败模式,考虑在 skill 中增加工具使用指导`);
451
582
  }
452
583
  }
453
584
  // Compare tool counts between variants
@@ -462,16 +593,13 @@ function detectToolPatterns(report, variants, insights, suggestions) {
462
593
  insights.push({
463
594
  type: 'tool_count_gap',
464
595
  severity: 'info',
465
- message: diff > 0
466
- ? `${variants[i]} 平均多调用 ${diff.toFixed(1)} 次工具(${other.avgToolCalls} vs ${base.avgToolCalls})`
467
- : `${variants[i]} 平均少调用 ${Math.abs(diff).toFixed(1)} 次工具(${other.avgToolCalls} vs ${base.avgToolCalls})`,
468
596
  details: { baseline: variants[0], variant: variants[i], baseTools: base.avgToolCalls, otherTools: other.avgToolCalls },
469
597
  });
470
598
  }
471
599
  }
472
600
  }
473
601
  }
474
- function detectToolPermissionIssues(results, variants, insights, suggestions) {
602
+ function detectToolPermissionIssues(results, variants, insights) {
475
603
  const permissionErrors = [];
476
604
  for (const result of results) {
477
605
  for (const variant of variants) {
@@ -496,12 +624,10 @@ function detectToolPermissionIssues(results, variants, insights, suggestions) {
496
624
  insights.push({
497
625
  type: 'tool_permission_error',
498
626
  severity: 'warning',
499
- message: `检测到 ${permissionErrors.length} 次工具权限错误,实验结论可能被环境问题污染`,
500
627
  details: permissionErrors.slice(0, 10),
501
628
  });
502
- suggestions.push('先处理工具权限错误,再解读 agent 分数差异;若是 Glob/rg 权限问题,优先避免在控制实验中依赖该工具');
503
629
  }
504
- function detectTraceIntegrity(report, variants, insights, suggestions) {
630
+ function detectTraceIntegrity(report, variants, insights) {
505
631
  const summary = report.summary || {};
506
632
  const agentAssertionTypes = collectAgentAssertionTypes(report.results || [], variants);
507
633
  const needsTraceHeavyCoverage = [...agentAssertionTypes].some((type) => TRACE_HEAVY_AGENT_ASSERTION_TYPES.has(type));
@@ -524,13 +650,11 @@ function detectTraceIntegrity(report, variants, insights, suggestions) {
524
650
  insights.push({
525
651
  type: 'trace_integrity_gap',
526
652
  severity: 'warning',
527
- message: `${weakCoverage.length} 个 variant 的 trace 覆盖率低于 75%,报告可能不足以解释 agent 行为差异`,
528
653
  details: weakCoverage,
529
654
  });
530
- suggestions.push('优先补齐 turns、toolCalls、timing、full output 的采集与落盘,确保报告能解释工具路径和错误恢复过程');
531
655
  }
532
656
  }
533
- function detectAgentAssertionDiscrimination(results, variants, insights, suggestions) {
657
+ function detectAgentAssertionDiscrimination(results, variants, insights) {
534
658
  const assertionTypes = collectAgentAssertionTypes(results, variants);
535
659
  const hasTraceHeavyAssertions = [...assertionTypes].some((type) => TRACE_HEAVY_AGENT_ASSERTION_TYPES.has(type));
536
660
  if (!hasTraceHeavyAssertions)
@@ -579,7 +703,6 @@ function detectAgentAssertionDiscrimination(results, variants, insights, suggest
579
703
  insights.push({
580
704
  type: 'agent_assertion_discrimination_low',
581
705
  severity: 'warning',
582
- message: `agent 断言区分度偏低,只有 ${(discriminationRate * 100).toFixed(0)}% 的断言真正拉开了变体差异`,
583
706
  details: {
584
707
  total: evaluated.length,
585
708
  discriminative,
@@ -588,13 +711,11 @@ function detectAgentAssertionDiscrimination(results, variants, insights, suggest
588
711
  examples: evaluated.slice(0, 10),
589
712
  },
590
713
  });
591
- suggestions.push('重写 agent 断言时,优先约束工具路径、关键文件读取和 turns 上限,避免大量“全过”或“全挂”的弱断言');
592
714
  }
593
715
  else {
594
716
  insights.push({
595
717
  type: 'agent_assertion_discrimination_ok',
596
718
  severity: 'info',
597
- message: `agent 断言区分度达标,${(discriminationRate * 100).toFixed(0)}% 的断言能区分变体差异`,
598
719
  details: {
599
720
  total: evaluated.length,
600
721
  discriminative,
@@ -623,7 +744,6 @@ function detectHighCost(results, variants, insights) {
623
744
  insights.push({
624
745
  type: 'high_cost_sample',
625
746
  severity: 'info',
626
- message: `${expensive.length} 个用例成本显著高于平均值 (>${(avg * 2).toFixed(4)} USD)`,
627
747
  details: expensive,
628
748
  });
629
749
  }