oh-my-knowledge 0.48.0 → 0.49.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (243) hide show
  1. package/README.md +50 -18
  2. package/README.zh.md +55 -23
  3. package/dist/analysis/coverage-analyzer.d.ts +1 -0
  4. package/dist/analysis/coverage-analyzer.js +125 -62
  5. package/dist/analysis/failure-clusterer.js +2 -1
  6. package/dist/analysis/gap-analyzer.d.ts +2 -2
  7. package/dist/analysis/gap-analyzer.js +13 -3
  8. package/dist/analysis/hedging-classifier.d.ts +2 -2
  9. package/dist/analysis/hedging-classifier.js +3 -4
  10. package/dist/analysis/report-diagnostics.js +9 -7
  11. package/dist/analysis/sample-diagnostics.js +6 -6
  12. package/dist/artifact-graph/doctor.js +15 -7
  13. package/dist/assets/agent-skills/omk/SKILL.md +27 -7
  14. package/dist/assets/agent-skills/omk/references/commands.md +18 -17
  15. package/dist/authoring/evolver.d.ts +10 -6
  16. package/dist/authoring/evolver.js +496 -83
  17. package/dist/authoring/generator.d.ts +3 -3
  18. package/dist/authoring/generator.js +5 -10
  19. package/dist/authoring/sample-fixer.d.ts +8 -6
  20. package/dist/authoring/sample-fixer.js +76 -5
  21. package/dist/cli/commands/doctor.js +31 -14
  22. package/dist/cli/commands/eval/index.d.ts +3 -0
  23. package/dist/cli/commands/eval/index.js +163 -17
  24. package/dist/cli/commands/evolve.d.ts +4 -4
  25. package/dist/cli/commands/evolve.js +27 -13
  26. package/dist/cli/commands/init.js +16 -3
  27. package/dist/cli/commands/observe/inbox.js +28 -21
  28. package/dist/cli/commands/observe/index.js +20 -11
  29. package/dist/cli/commands/observe/ingest.d.ts +3 -0
  30. package/dist/cli/commands/observe/ingest.js +30 -2
  31. package/dist/cli/commands/sample.d.ts +6 -3
  32. package/dist/cli/commands/sample.js +72 -68
  33. package/dist/cli/lib/codex-model-hint.d.ts +9 -0
  34. package/dist/cli/lib/codex-model-hint.js +45 -0
  35. package/dist/cli/lib/generation-failure-hint.d.ts +2 -0
  36. package/dist/cli/lib/generation-failure-hint.js +61 -0
  37. package/dist/cli/lib/i18n-dict/common.d.ts +1 -1
  38. package/dist/cli/lib/i18n-dict/common.js +4 -0
  39. package/dist/cli/lib/i18n-dict/gen.d.ts +1 -1
  40. package/dist/cli/lib/i18n-dict/gen.js +38 -6
  41. package/dist/cli/lib/i18n-dict/help.js +6 -6
  42. package/dist/cli/lib/i18n-dict/init.d.ts +1 -1
  43. package/dist/cli/lib/i18n-dict/init.js +13 -9
  44. package/dist/cli/lib/i18n-dict/run.d.ts +1 -1
  45. package/dist/cli/lib/i18n-dict/run.js +34 -2
  46. package/dist/cli/lib/llm-failure-classifier.d.ts +2 -0
  47. package/dist/cli/lib/llm-failure-classifier.js +8 -0
  48. package/dist/cli/lib/parse-run-config.d.ts +6 -5
  49. package/dist/cli/lib/parse-run-config.js +16 -9
  50. package/dist/cli/lib/runtime-defaults.d.ts +21 -0
  51. package/dist/cli/lib/runtime-defaults.js +79 -0
  52. package/dist/diagnosis/observe-mapper.js +14 -15
  53. package/dist/diagnosis/observe-producer.js +3 -1
  54. package/dist/diagnosis/studio-projection.js +14 -7
  55. package/dist/diagnosis/types.d.ts +2 -0
  56. package/dist/diagnosis/types.js +12 -0
  57. package/dist/doctor/endpoint-rule.js +2 -1
  58. package/dist/eval-core/artifact-file-names.js +18 -1
  59. package/dist/eval-core/artifact-index.d.ts +7 -11
  60. package/dist/eval-core/artifact-index.js +139 -80
  61. package/dist/eval-core/cache.d.ts +12 -3
  62. package/dist/eval-core/cache.js +89 -29
  63. package/dist/eval-core/comparability.js +10 -6
  64. package/dist/eval-core/evaluation-execution.d.ts +2 -1
  65. package/dist/eval-core/evaluation-execution.js +122 -37
  66. package/dist/eval-core/evaluation-job.d.ts +4 -1
  67. package/dist/eval-core/evaluation-job.js +4 -1
  68. package/dist/eval-core/evaluation-reporting.d.ts +15 -13
  69. package/dist/eval-core/evaluation-reporting.js +54 -52
  70. package/dist/eval-core/execution-strategy.d.ts +2 -0
  71. package/dist/eval-core/execution-strategy.js +11 -9
  72. package/dist/eval-core/fact-checker.js +15 -7
  73. package/dist/eval-core/holdout.js +3 -2
  74. package/dist/eval-core/judge-independence.d.ts +2 -2
  75. package/dist/eval-core/mock-hook.cjs +23 -6
  76. package/dist/eval-core/mocks-runtime.js +30 -8
  77. package/dist/eval-core/report-document.d.ts +12 -0
  78. package/dist/eval-core/report-document.js +1151 -0
  79. package/dist/eval-core/report-extensions.d.ts +4 -0
  80. package/dist/eval-core/report-extensions.js +500 -0
  81. package/dist/eval-core/report-file-migration.js +7 -2
  82. package/dist/eval-core/resume-compatibility.d.ts +31 -0
  83. package/dist/eval-core/resume-compatibility.js +141 -0
  84. package/dist/eval-core/sample-fingerprint.d.ts +12 -0
  85. package/dist/eval-core/sample-fingerprint.js +193 -0
  86. package/dist/eval-core/schema.js +86 -31
  87. package/dist/eval-core/verdict.d.ts +8 -4
  88. package/dist/eval-core/verdict.js +24 -10
  89. package/dist/eval-workflows/batch-evaluation-workflow.d.ts +2 -1
  90. package/dist/eval-workflows/batch-evaluation-workflow.js +25 -12
  91. package/dist/eval-workflows/evaluation-pipeline/preflight-warnings.d.ts +10 -5
  92. package/dist/eval-workflows/evaluation-pipeline/preflight-warnings.js +58 -21
  93. package/dist/eval-workflows/evaluation-pipeline/report-finalize.js +3 -1
  94. package/dist/eval-workflows/evaluation-pipeline/run-state.d.ts +4 -1
  95. package/dist/eval-workflows/evaluation-pipeline/run-state.js +4 -1
  96. package/dist/eval-workflows/evaluation-pipeline/test-set-hash.d.ts +6 -5
  97. package/dist/eval-workflows/evaluation-pipeline/test-set-hash.js +17 -10
  98. package/dist/eval-workflows/evaluation-pipeline.js +12 -7
  99. package/dist/eval-workflows/run-evaluation.d.ts +9 -7
  100. package/dist/eval-workflows/run-evaluation.js +79 -51
  101. package/dist/executors/anthropic-api.js +65 -9
  102. package/dist/executors/claude-cli.js +16 -79
  103. package/dist/executors/claude-protocol.d.ts +28 -0
  104. package/dist/executors/claude-protocol.js +180 -0
  105. package/dist/executors/claude-sdk-trace.js +56 -28
  106. package/dist/executors/claude-sdk.d.ts +1 -0
  107. package/dist/executors/claude-sdk.js +39 -93
  108. package/dist/executors/codex-cli-trace.js +166 -31
  109. package/dist/executors/codex-cli.d.ts +6 -8
  110. package/dist/executors/codex-cli.js +49 -151
  111. package/dist/executors/codex-protocol.d.ts +24 -0
  112. package/dist/executors/codex-protocol.js +234 -0
  113. package/dist/executors/codex-sdk.js +68 -120
  114. package/dist/executors/gemini.js +88 -13
  115. package/dist/executors/index.d.ts +2 -3
  116. package/dist/executors/index.js +5 -3
  117. package/dist/executors/openai-api.js +70 -9
  118. package/dist/executors/runtime-fingerprint.js +88 -11
  119. package/dist/executors/script-command.d.ts +8 -0
  120. package/dist/executors/script-command.js +87 -0
  121. package/dist/executors/script.js +202 -29
  122. package/dist/executors/shared.d.ts +35 -3
  123. package/dist/executors/shared.js +113 -15
  124. package/dist/grading/assertions.d.ts +1 -1
  125. package/dist/grading/assertions.js +19 -9
  126. package/dist/grading/diagnostic.d.ts +9 -2
  127. package/dist/grading/diagnostic.js +25 -2
  128. package/dist/grading/index.js +10 -4
  129. package/dist/grading/judge.js +19 -6
  130. package/dist/grading/layered-scores.d.ts +2 -3
  131. package/dist/grading/layered-scores.js +2 -3
  132. package/dist/inputs/load-samples.d.ts +1 -2
  133. package/dist/inputs/load-samples.js +23 -1
  134. package/dist/inputs/mcp-resolver.js +6 -3
  135. package/dist/inputs/sample-document.d.ts +11 -0
  136. package/dist/inputs/sample-document.js +96 -0
  137. package/dist/managed/evidence.d.ts +1 -0
  138. package/dist/managed/evidence.js +1 -1
  139. package/dist/managed/store.js +200 -91
  140. package/dist/observability/codex-trace-adapter.d.ts +5 -0
  141. package/dist/observability/codex-trace-adapter.js +850 -0
  142. package/dist/observability/experience.d.ts +32 -6
  143. package/dist/observability/experience.js +2695 -459
  144. package/dist/observability/feedback-matchers.js +16 -1
  145. package/dist/observability/inbox-view-model.d.ts +1 -1
  146. package/dist/observability/inbox-view-model.js +19 -14
  147. package/dist/observability/inbox.d.ts +7 -1
  148. package/dist/observability/inbox.js +632 -124
  149. package/dist/observability/problem-patterns.js +2 -0
  150. package/dist/observability/review-state.d.ts +6 -0
  151. package/dist/observability/review-state.js +235 -63
  152. package/dist/observability/skill-chain-advisories.js +1 -1
  153. package/dist/observability/skill-chain.js +17 -4
  154. package/dist/observability/skill-health-analyzer.d.ts +32 -7
  155. package/dist/observability/skill-health-analyzer.js +194 -121
  156. package/dist/observability/skill-health-report.d.ts +10 -0
  157. package/dist/observability/skill-health-report.js +620 -0
  158. package/dist/observability/soft-standards/constants.d.ts +0 -1
  159. package/dist/observability/soft-standards/constants.js +0 -1
  160. package/dist/observability/soft-standards/index.d.ts +1 -1
  161. package/dist/observability/soft-standards/index.js +1 -1
  162. package/dist/observability/soft-standards/llm-extractor.js +8 -10
  163. package/dist/observability/soft-standards/skill-standards-store.d.ts +2 -1
  164. package/dist/observability/soft-standards/skill-standards-store.js +59 -18
  165. package/dist/observability/soft-standards/types.d.ts +2 -2
  166. package/dist/observability/trace-adapter.d.ts +12 -7
  167. package/dist/observability/trace-adapter.js +11 -9
  168. package/dist/observability/trace-attribution.d.ts +13 -5
  169. package/dist/observability/trace-attribution.js +315 -21
  170. package/dist/observability/trace-ingestion.d.ts +9 -0
  171. package/dist/observability/trace-ingestion.js +80 -0
  172. package/dist/observability/trace-ir.d.ts +113 -0
  173. package/dist/observability/trace-ir.js +87 -0
  174. package/dist/observability/trace-segmenter.d.ts +19 -6
  175. package/dist/observability/trace-segmenter.js +377 -196
  176. package/dist/observability/trace-session-index.d.ts +19 -0
  177. package/dist/observability/trace-session-index.js +68 -0
  178. package/dist/observability/trace-source.d.ts +12 -4
  179. package/dist/observability/trace-source.js +939 -215
  180. package/dist/renderer/html-renderer.js +37 -6
  181. package/dist/renderer/icons.js +3 -0
  182. package/dist/renderer/observation-inbox-renderer.js +208 -90
  183. package/dist/renderer/skill-detail-renderer.js +452 -109
  184. package/dist/renderer/skill-health-renderer.js +69 -12
  185. package/dist/renderer/summary.js +28 -7
  186. package/dist/renderer/table.js +21 -4
  187. package/dist/renderer/test-view.d.ts +1 -0
  188. package/dist/renderer/test-view.js +44 -9
  189. package/dist/server/indexed-report-store.js +14 -18
  190. package/dist/server/job-store.js +64 -26
  191. package/dist/server/report-server.js +190 -78
  192. package/dist/server/report-store.js +57 -80
  193. package/dist/server/skill-index.js +143 -49
  194. package/dist/server/skill-insights.js +44 -5
  195. package/dist/shared/artifact-graph.d.ts +3 -0
  196. package/dist/shared/artifact-graph.js +224 -0
  197. package/dist/shared/assertion-types.d.ts +8 -0
  198. package/dist/shared/assertion-types.js +46 -0
  199. package/dist/shared/atomic-json.d.ts +8 -0
  200. package/dist/shared/atomic-json.js +33 -0
  201. package/dist/shared/diagnosis-schema.d.ts +9 -0
  202. package/dist/shared/diagnosis-schema.js +181 -0
  203. package/dist/shared/doctor-report.d.ts +3 -0
  204. package/dist/shared/doctor-report.js +103 -0
  205. package/dist/shared/evaluation-job.d.ts +6 -0
  206. package/dist/shared/evaluation-job.js +217 -0
  207. package/dist/shared/executor-result.d.ts +17 -0
  208. package/dist/shared/executor-result.js +221 -0
  209. package/dist/shared/file-lock.d.ts +12 -0
  210. package/dist/shared/file-lock.js +129 -0
  211. package/dist/shared/json-value.d.ts +5 -0
  212. package/dist/shared/json-value.js +36 -0
  213. package/dist/shared/keyed-mutex.d.ts +7 -0
  214. package/dist/shared/keyed-mutex.js +24 -0
  215. package/dist/shared/record-count.d.ts +8 -0
  216. package/dist/shared/record-count.js +43 -0
  217. package/dist/shared/sample-contract.d.ts +3 -0
  218. package/dist/shared/sample-contract.js +332 -0
  219. package/dist/shared/timestamp.d.ts +6 -0
  220. package/dist/shared/timestamp.js +64 -0
  221. package/dist/shared/token-usage.d.ts +19 -0
  222. package/dist/shared/token-usage.js +50 -0
  223. package/dist/shared/tool-call-status.d.ts +8 -0
  224. package/dist/shared/tool-call-status.js +28 -0
  225. package/dist/shared/tool-identity.d.ts +21 -0
  226. package/dist/shared/tool-identity.js +84 -0
  227. package/dist/shared/tool-search.js +73 -16
  228. package/dist/shared/trace-projection.d.ts +5 -0
  229. package/dist/shared/trace-projection.js +20 -0
  230. package/dist/shared/trace-source-kind.d.ts +3 -0
  231. package/dist/shared/trace-source-kind.js +12 -0
  232. package/dist/types/diagnosis.d.ts +2 -0
  233. package/dist/types/eval.d.ts +4 -0
  234. package/dist/types/executor.d.ts +32 -5
  235. package/dist/types/index.d.ts +1 -0
  236. package/dist/types/index.js +1 -0
  237. package/dist/types/judge.d.ts +2 -0
  238. package/dist/types/observability.d.ts +116 -9
  239. package/dist/types/report.d.ts +58 -6
  240. package/dist/types/skill-index.d.ts +7 -0
  241. package/dist/types/trace.d.ts +2 -0
  242. package/dist/types/trace.js +1 -0
  243. package/package.json +9 -5
@@ -1,16 +1,24 @@
1
- import { readFileSync, writeFileSync, mkdirSync, existsSync } from 'node:fs';
1
+ import { copyFileSync, existsSync, mkdirSync, mkdtempSync, readFileSync, rmSync, statSync, writeFileSync, } from 'node:fs';
2
+ import { tmpdir } from 'node:os';
2
3
  import { resolve, join, dirname, basename } from 'node:path';
3
4
  import { runEvaluation } from '../eval-workflows/run-evaluation.js';
4
- import { createExecutor, DEFAULT_MODEL, JUDGE_MODEL } from '../executors/index.js';
5
- import { persistReport, DEFAULT_OUTPUT_DIR, runIdSuffix, hashString } from '../eval-core/evaluation-reporting.js';
5
+ import { createExecutor } from '../executors/index.js';
6
+ import { persistReport, DEFAULT_OUTPUT_DIR, EVALUATION_REPORT_SCHEMA_VERSION, getCliVersion, hashSample, runIdSuffix, hashString, } from '../eval-core/evaluation-reporting.js';
6
7
  import { createOverlayReportStore } from '../server/report-store.js';
7
8
  import { projectReportsDir, globalReportsDir } from '../eval-core/measurement-dirs.js';
8
9
  import { analyzeResults } from '../analysis/report-diagnostics.js';
9
10
  import { loadSamples } from '../inputs/load-samples.js';
11
+ import { writeFixedSamplesToSources } from '../inputs/sample-document.js';
10
12
  import { hashArtifactSource } from '../inputs/content-hash.js';
11
13
  import { MIN_HOLDOUT_SUBSET, pickByStride, splitHoldout, subsetCompositeScore } from '../eval-core/holdout.js';
12
14
  import { bootstrapDiffCI, DEFAULT_BOOTSTRAP_ALPHA, DEFAULT_BOOTSTRAP_SAMPLES } from '../eval-core/bootstrap.js';
13
- import { fixSamples } from './sample-fixer.js';
15
+ import { fixSamples, sampleFixWithinScope, stampFixMetadata } from './sample-fixer.js';
16
+ import { toolCallStatus } from '../shared/tool-call-status.js';
17
+ import { ownRecordValue, setOwnRecordValue } from '../shared/record-count.js';
18
+ import { getJudgePromptHash } from '../grading/judge.js';
19
+ import { getDiagnosticPromptHash, resolveDiagnosticTarget, } from '../grading/diagnostic.js';
20
+ import { getExecutorRuntimeFingerprint } from '../executors/runtime-fingerprint.js';
21
+ import { parseReportDocument } from '../eval-core/report-document.js';
14
22
  const IMPROVE_SYSTEM_PROMPT = `你是一个 AI 提示词改进专家。你的任务是分析评测结果中的薄弱环节,针对性地改进 skill(系统提示词),使其在评测中获得更高的分数。
15
23
 
16
24
  改进原则:
@@ -46,58 +54,146 @@ function canonicalStringify(value) {
46
54
  const entries = Object.keys(value).sort();
47
55
  return '{' + entries.map((k) => JSON.stringify(k) + ':' + canonicalStringify(value[k])).join(',') + '}';
48
56
  }
49
- function hashSampleForReuse(sample) {
50
- return hashString(canonicalStringify({
51
- prompt: sample.prompt,
52
- rubric: sample.rubric ?? null,
53
- dimensions: sample.dimensions ?? null,
54
- assertions: sample.assertions ?? null,
55
- schema: sample.schema ?? null,
56
- }));
57
- }
58
57
  function sameJudgeModels(report, judges) {
59
58
  const metaJudges = report.meta.judgeModels ?? [];
60
59
  if (metaJudges.length !== judges.length)
61
60
  return false;
62
61
  return metaJudges.every((j, i) => j.executor === judges[i].executor && j.model === judges[i].model);
63
62
  }
64
- function sampleHashesMatch(report, samples) {
63
+ function sampleHashesMatch(report, samples, samplesBaseDir) {
65
64
  const hashes = report.meta.sampleHashes;
66
- if (!hashes)
65
+ if (!hashes
66
+ || report.meta.sampleCount !== samples.length
67
+ || Object.keys(hashes).length !== samples.length)
67
68
  return false;
68
- return samples.every((s) => hashes[s.sample_id] === hashSampleForReuse(s));
69
+ return samples.every((s) => ownRecordValue(hashes, s.sample_id) === hashSample(s, samplesBaseDir));
70
+ }
71
+ function omitKeys(value, keys) {
72
+ const copy = { ...value };
73
+ for (const key of keys)
74
+ Reflect.deleteProperty(copy, key);
75
+ return copy;
69
76
  }
70
- function singleVariantReport(report, variantKey) {
71
- const summary = report.summary[variantKey];
77
+ export function singleVariantReport(report, variantKey) {
78
+ const summary = ownRecordValue(report.summary, variantKey);
79
+ if (!summary) {
80
+ throw new Error(`report is missing summary for variant "${variantKey}"`);
81
+ }
82
+ const artifactHash = report.meta.artifactHashes
83
+ ? ownRecordValue(report.meta.artifactHashes, variantKey)
84
+ : undefined;
85
+ const isolation = report.meta.skillIsolation
86
+ ? ownRecordValue(report.meta.skillIsolation, variantKey)
87
+ : undefined;
88
+ const results = report.results.map((entry) => {
89
+ const variant = ownRecordValue(entry.variants, variantKey);
90
+ if (!variant) {
91
+ throw new Error(`report is missing variant "${variantKey}" for sample "${entry.sample_id}"`);
92
+ }
93
+ return {
94
+ sample_id: entry.sample_id,
95
+ variants: { [variantKey]: variant },
96
+ };
97
+ });
98
+ const totalCostUSD = Number(results.reduce((total, entry) => total + entry.variants[variantKey].costUSD, 0).toFixed(6));
99
+ const executorRuntime = report.meta.executorRuntimes
100
+ ? ownRecordValue(report.meta.executorRuntimes, variantKey)
101
+ : report.meta.executorRuntime;
102
+ const request = report.meta.request
103
+ ? {
104
+ ...report.meta.request,
105
+ artifacts: report.meta.request.artifacts.filter((artifact) => artifact.name === variantKey),
106
+ }
107
+ : undefined;
108
+ if (report.meta.request && request?.artifacts.length !== 1) {
109
+ throw new Error(`report request is missing artifact "${variantKey}"`);
110
+ }
111
+ const sourceHumanAgreement = report.meta.humanAgreement;
112
+ const sourceJob = report.meta.job;
113
+ const sharedMeta = omitKeys(report.meta, [
114
+ 'variants',
115
+ 'taskCount',
116
+ 'totalCostUSD',
117
+ 'totalCostReported',
118
+ 'artifactHashes',
119
+ 'executorRuntime',
120
+ 'executorRuntimes',
121
+ 'pairComparisons',
122
+ 'humanAgreement',
123
+ 'variantConfigs',
124
+ 'skillIsolation',
125
+ 'request',
126
+ 'job',
127
+ 'evolve',
128
+ ]);
129
+ const sharedReport = omitKeys(report, ['analysis', 'variance']);
72
130
  return {
73
- ...report,
131
+ ...sharedReport,
74
132
  meta: {
75
- ...report.meta,
133
+ ...sharedMeta,
76
134
  variants: [variantKey],
77
- artifactHashes: report.meta.artifactHashes?.[variantKey]
78
- ? { [variantKey]: report.meta.artifactHashes[variantKey] }
135
+ taskCount: report.meta.sampleCount,
136
+ totalCostUSD,
137
+ ...(summary.execCostReported === false || summary.judgeCostReported === false
138
+ ? { totalCostReported: false }
139
+ : {}),
140
+ artifactHashes: artifactHash
141
+ ? { [variantKey]: artifactHash }
79
142
  : {},
143
+ ...(executorRuntime
144
+ ? {
145
+ executorRuntime,
146
+ executorRuntimes: { [variantKey]: executorRuntime },
147
+ }
148
+ : {}),
80
149
  ...(report.meta.variantConfigs ? { variantConfigs: report.meta.variantConfigs.filter((cfg) => cfg.variant === variantKey) } : {}),
81
- ...(report.meta.skillIsolation ? { skillIsolation: { [variantKey]: report.meta.skillIsolation[variantKey] ?? null } } : {}),
150
+ ...(report.meta.skillIsolation ? { skillIsolation: { [variantKey]: isolation ?? null } } : {}),
151
+ ...(sourceHumanAgreement?.variant === variantKey
152
+ ? { humanAgreement: sourceHumanAgreement }
153
+ : {}),
154
+ ...(request ? { request } : {}),
155
+ ...(sourceJob && request
156
+ ? {
157
+ job: {
158
+ ...sourceJob,
159
+ request: structuredClone(request),
160
+ },
161
+ }
162
+ : {}),
82
163
  },
83
164
  summary: { [variantKey]: summary },
84
- results: report.results.map((entry) => ({
85
- sample_id: entry.sample_id,
86
- variants: entry.variants[variantKey] ? { [variantKey]: entry.variants[variantKey] } : {},
87
- })),
165
+ results,
88
166
  };
89
167
  }
90
168
  async function findReusableBaselineReport(opts) {
91
169
  // baseline 复用读 overlay(项目 .omk/reports ∪ 全局):eval 写默认翻项目后,复用既能命中 eval 新写的项目
92
170
  // baseline,又继续覆盖全局(含 evolve 自身写到全局的合并报告),复用命中率与报告数字不降。
93
171
  const store = createOverlayReportStore(projectReportsDir(), globalReportsDir());
94
- const { samples } = loadSamples(opts.samplesPath);
172
+ const { samples, baseDir: samplesBaseDir } = loadSamples(opts.samplesPath);
95
173
  const artifactHash = opts.artifactHash;
96
174
  const reports = await store.findByArtifactHash(artifactHash);
175
+ const expectedExecutorRuntime = getExecutorRuntimeFingerprint(opts.executorName, opts.model);
176
+ const expectedJudgeRuntimes = opts.judgeModels.map((judge) => getExecutorRuntimeFingerprint(judge.executor, judge.model, {
177
+ skillDir: opts.skillDir,
178
+ }));
179
+ const expectedJudgePromptHash = getJudgePromptHash(true);
180
+ const expectedDiagnosticTarget = resolveDiagnosticTarget(opts.judgeModels, opts.executorName, opts.model);
181
+ const expectedDiagnostic = opts.noDiagnostic
182
+ ? { enabled: false }
183
+ : {
184
+ enabled: true,
185
+ executor: expectedDiagnosticTarget.executor,
186
+ model: expectedDiagnosticTarget.model,
187
+ runtime: getExecutorRuntimeFingerprint(expectedDiagnosticTarget.executor, expectedDiagnosticTarget.model).fingerprint,
188
+ promptHash: getDiagnosticPromptHash(),
189
+ };
97
190
  for (const report of reports) {
98
- // schemaVersion < 2 的报告 artifactHashes 是旧文本哈,与当前树哈不同空间:即便值偶合也不该复用
99
- // (口径不同会让 lineage 串错身份)。直接跳过,让旧 baseline 重跑出树哈报告。
100
- if ((report.meta.schemaVersion ?? 0) < 2)
191
+ // Reuse is a measurement-contract decision, not just a content-hash lookup.
192
+ // Old schema / CLI / prompt/runtime evidence can produce numerically similar
193
+ // scores under a different construct, so fail closed and rerun.
194
+ if (report.meta.schemaVersion !== EVALUATION_REPORT_SCHEMA_VERSION)
195
+ continue;
196
+ if (report.meta.cliVersion !== getCliVersion())
101
197
  continue;
102
198
  if (report.meta.model !== opts.model || report.meta.executor !== opts.executorName)
103
199
  continue;
@@ -107,13 +203,67 @@ async function findReusableBaselineReport(opts) {
107
203
  continue;
108
204
  if (report.meta.budgetExhausted)
109
205
  continue;
206
+ if ((report.meta.request?.noDiagnostic === true)
207
+ !== (opts.noDiagnostic === true))
208
+ continue;
209
+ const actualDiagnostic = report.meta.diagnostic
210
+ ? {
211
+ enabled: report.meta.diagnostic.enabled,
212
+ ...(report.meta.diagnostic.executor
213
+ ? { executor: report.meta.diagnostic.executor }
214
+ : {}),
215
+ ...(report.meta.diagnostic.model
216
+ ? { model: report.meta.diagnostic.model }
217
+ : {}),
218
+ ...(report.meta.diagnostic.runtime
219
+ ? { runtime: report.meta.diagnostic.runtime.fingerprint }
220
+ : {}),
221
+ ...(report.meta.diagnostic.promptHash
222
+ ? { promptHash: report.meta.diagnostic.promptHash }
223
+ : {}),
224
+ }
225
+ : undefined;
226
+ if (canonicalStringify(actualDiagnostic) !== canonicalStringify(expectedDiagnostic))
227
+ continue;
110
228
  if (!sameJudgeModels(report, opts.judgeModels))
111
229
  continue;
112
- if (!sampleHashesMatch(report, samples))
230
+ if (report.meta.judgePromptHash !== expectedJudgePromptHash)
231
+ continue;
232
+ if (report.meta.judgeModels.some((judge, index) => !judge.runtime
233
+ || judge.runtime.fingerprint !== expectedJudgeRuntimes[index]?.fingerprint))
234
+ continue;
235
+ if (!sampleHashesMatch(report, samples, samplesBaseDir))
236
+ continue;
237
+ const variantKey = report.meta.variants.find((name) => report.meta.artifactHashes
238
+ && ownRecordValue(report.meta.artifactHashes, name) === artifactHash);
239
+ if (!variantKey || !ownRecordValue(report.summary, variantKey))
240
+ continue;
241
+ const executorRuntime = report.meta.executorRuntimes
242
+ ? ownRecordValue(report.meta.executorRuntimes, variantKey)
243
+ : report.meta.executorRuntime;
244
+ if (executorRuntime?.fingerprint !== expectedExecutorRuntime.fingerprint)
113
245
  continue;
114
- const variantKey = report.meta.variants.find((name) => report.meta.artifactHashes?.[name] === artifactHash);
115
- if (!variantKey || !report.summary[variantKey])
246
+ const config = report.meta.variantConfigs?.find((candidate) => candidate.variant === variantKey);
247
+ if (!config
248
+ || config.artifactKind !== 'skill'
249
+ || config.executionStrategy !== 'system-prompt'
250
+ || config.experimentType !== 'artifact-injection'
251
+ || config.hasArtifactContent !== true
252
+ || config.allowedSkills !== undefined)
116
253
  continue;
254
+ if (!report.meta.skillIsolation
255
+ || !Object.hasOwn(report.meta.skillIsolation, variantKey)
256
+ || ownRecordValue(report.meta.skillIsolation, variantKey) !== null)
257
+ continue;
258
+ if (opts.noDiagnostic !== true) {
259
+ const missingDiagnostic = report.results.some((entry) => {
260
+ const variant = ownRecordValue(entry.variants, variantKey);
261
+ return variant?.assertions?.details.some((detail) => !detail.passed) === true
262
+ && variant.diagnostic === undefined;
263
+ });
264
+ if (missingDiagnostic)
265
+ continue;
266
+ }
117
267
  return singleVariantReport(report, variantKey);
118
268
  }
119
269
  return null;
@@ -280,11 +430,20 @@ function createSampleFixExecutor(executorName) {
280
430
  lean: opts.lean,
281
431
  ...(opts.cwd && { cwd: opts.cwd }),
282
432
  });
283
- return { ok: result.ok, text: result.output ?? '', costUSD: result.costUSD };
433
+ return {
434
+ ok: result.ok,
435
+ text: result.output ?? '',
436
+ costUSD: result.costUSD,
437
+ costReported: result.costReportedByExecutor !== false,
438
+ };
284
439
  };
285
440
  }
441
+ export function agentSampleEditWithinScope(previous, next) {
442
+ return sampleFixWithinScope(previous, next);
443
+ }
286
444
  async function autoFixSamplesAgent(opts) {
287
- const { samples } = loadSamples(opts.samplesPath);
445
+ const loaded = loadSamples(opts.samplesPath);
446
+ const { samples } = loaded;
288
447
  const sampleMap = new Map(samples.map((s) => [s.sample_id, s]));
289
448
  const fixContexts = [];
290
449
  for (const entry of opts.report.results) {
@@ -310,7 +469,7 @@ async function autoFixSamplesAgent(opts) {
310
469
  const toolCalls = (variantObj.toolCalls ?? []);
311
470
  const failedList = assertionDetails.filter((a) => !a.passed).map((a) => `${a.type}: ${a.value}`).join('\n');
312
471
  const toolSummary = toolCalls.length > 0
313
- ? toolCalls.map((tc, i) => `[${i}] ${tc.tool} success=${tc.success}`).join('\n')
472
+ ? toolCalls.map((tc, i) => `[${i}] ${tc.tool} status=${toolCallStatus(tc)}`).join('\n')
314
473
  : '(无工具调用)';
315
474
  fixContexts.push({
316
475
  sampleId: sid,
@@ -318,13 +477,30 @@ async function autoFixSamplesAgent(opts) {
318
477
  failedAssertions: `失败断言:\n${failedList}\n\n实际工具调用:\n${toolSummary}`,
319
478
  });
320
479
  }
321
- if (fixContexts.length === 0)
322
- return { fixedCount: 0, costUSD: 0 };
480
+ if (fixContexts.length === 0) {
481
+ return { fixedCount: 0, costUSD: 0, costReported: true };
482
+ }
323
483
  const skillPreview = opts.skillContent.length > 4000
324
484
  ? opts.skillContent.slice(0, 4000) + '\n\n... (truncated)'
325
485
  : opts.skillContent;
326
486
  const sampleSections = fixContexts.map((ctx) => `### ${ctx.sampleId}\n\n${ctx.diag}\n\n${ctx.failedAssertions}`).join('\n\n---\n\n');
327
- const prompt = `以下有 ${fixContexts.length} 条失败的评测用例需要分析修复。
487
+ const tempRoot = mkdtempSync(join(tmpdir(), 'omk-sample-fix-'));
488
+ const sourceIsDirectory = statSync(opts.samplesPath).isDirectory();
489
+ const tempSamplesPath = sourceIsDirectory
490
+ ? join(tempRoot, 'samples')
491
+ : join(tempRoot, basename(opts.samplesPath));
492
+ if (sourceIsDirectory)
493
+ mkdirSync(tempSamplesPath);
494
+ for (const sourceFile of loaded.sourceFiles) {
495
+ const targetFile = sourceIsDirectory
496
+ ? join(tempSamplesPath, basename(sourceFile))
497
+ : tempSamplesPath;
498
+ copyFileSync(sourceFile, targetFile);
499
+ }
500
+ try {
501
+ const tempLoaded = loadSamples(tempSamplesPath);
502
+ const editableFiles = tempLoaded.sourceFiles.map((file) => `- ${file}`).join('\n');
503
+ const prompt = `以下有 ${fixContexts.length} 条失败的评测用例需要分析修复。
328
504
 
329
505
  ## Skill 原文(参考,不可修改)
330
506
 
@@ -334,23 +510,67 @@ ${skillPreview}
334
510
 
335
511
  ${sampleSections}
336
512
 
337
- 请使用 Edit 工具修改文件 ${opts.samplesPath},只改有问题的 sample 的 assertions / mocks / environment 字段。如果判断是 LLM 行为问题(低分合理),不要改该 sample。`;
338
- const executor = createExecutor(opts.executorName);
339
- const beforeContent = readFileSync(opts.samplesPath, 'utf-8');
340
- const result = await executor({
341
- model: opts.model,
342
- system: FIX_AGENT_SYSTEM_PROMPT,
343
- prompt,
344
- cwd: dirname(opts.samplesPath),
345
- timeoutMs: 300_000,
346
- });
347
- const afterContent = readFileSync(opts.samplesPath, 'utf-8');
348
- const changed = afterContent !== beforeContent;
349
- const fixedCount = changed ? fixContexts.length : 0;
350
- return { fixedCount, costUSD: result.costUSD };
513
+ 请只使用 Edit 工具修改以下临时副本:
514
+ ${editableFiles}
515
+
516
+ 只改有问题 sample 的 assertions / mocks / mocksStrict / environment 字段。如果判断是 LLM 行为问题(低分合理),不要改该 sample。`;
517
+ const executor = createExecutor(opts.executorName);
518
+ const result = await executor({
519
+ model: opts.model,
520
+ system: FIX_AGENT_SYSTEM_PROMPT,
521
+ prompt,
522
+ cwd: tempRoot,
523
+ timeoutMs: 300_000,
524
+ });
525
+ const costReported = result.costReportedByExecutor !== false;
526
+ if (!result.ok) {
527
+ return { fixedCount: 0, costUSD: result.costUSD, costReported };
528
+ }
529
+ let edited;
530
+ try {
531
+ edited = loadSamples(tempSamplesPath).samples;
532
+ }
533
+ catch (error) {
534
+ process.stderr.write(`[omk] auto-fix-samples 已拒绝非法评测用例修改:`
535
+ + `${error instanceof Error ? error.message : String(error)}\n`);
536
+ return { fixedCount: 0, costUSD: result.costUSD, costReported };
537
+ }
538
+ const originalById = new Map(samples.map((sample) => [sample.sample_id, sample]));
539
+ const editedById = new Map(edited.map((sample) => [sample.sample_id, sample]));
540
+ if (editedById.size !== originalById.size
541
+ || [...originalById.keys()].some((sampleId) => !editedById.has(sampleId))) {
542
+ process.stderr.write('[omk] auto-fix-samples 已拒绝新增、删除或改名评测用例。\n');
543
+ return { fixedCount: 0, costUSD: result.costUSD, costReported };
544
+ }
545
+ const fixableIds = new Set(fixContexts.map((context) => context.sampleId));
546
+ const changedIds = new Set();
547
+ for (const [sampleId, nextSample] of editedById.entries()) {
548
+ const previousSample = originalById.get(sampleId);
549
+ if (canonicalStringify(nextSample) === canonicalStringify(previousSample))
550
+ continue;
551
+ if (!fixableIds.has(sampleId)
552
+ || !agentSampleEditWithinScope(previousSample, nextSample)) {
553
+ process.stderr.write(`[omk] auto-fix-samples 已拒绝越界修改 sample「${sampleId}」;`
554
+ + '只允许修改待修复用例的 assertions / mocks / mocksStrict / environment。\n');
555
+ return { fixedCount: 0, costUSD: result.costUSD, costReported };
556
+ }
557
+ stampFixMetadata(nextSample, opts.report.id);
558
+ changedIds.add(sampleId);
559
+ }
560
+ writeFixedSamplesToSources(loaded, edited, changedIds);
561
+ return {
562
+ fixedCount: changedIds.size,
563
+ costUSD: result.costUSD,
564
+ costReported,
565
+ };
566
+ }
567
+ finally {
568
+ rmSync(tempRoot, { recursive: true, force: true });
569
+ }
351
570
  }
352
571
  async function autoFixSamplesAfterSkillRound(opts) {
353
- const { samples } = loadSamples(opts.samplesPath);
572
+ const loaded = loadSamples(opts.samplesPath);
573
+ const { samples } = loaded;
354
574
  if (opts.improveMode === 'agent') {
355
575
  return autoFixSamplesAgent(opts);
356
576
  }
@@ -364,9 +584,14 @@ async function autoFixSamplesAfterSkillRound(opts) {
364
584
  maxAttemptsPerSample: opts.maxAttemptsPerSample,
365
585
  });
366
586
  if (result.fixedCount > 0) {
367
- writeFileSync(opts.samplesPath, JSON.stringify(result.samples, null, 2));
587
+ const changedIds = new Set(result.fixes.filter((fix) => fix.changed).map((fix) => fix.sampleId));
588
+ writeFixedSamplesToSources(loaded, result.samples, changedIds);
368
589
  }
369
- return { fixedCount: result.fixedCount, costUSD: result.costUSD };
590
+ return {
591
+ fixedCount: result.fixedCount,
592
+ costUSD: result.costUSD,
593
+ costReported: result.costReported,
594
+ };
370
595
  }
371
596
  /** Number of skill lines below which the edit budget never trips — so a tiny skill
372
597
  * isn't frozen by a percentage threshold that a few lines already blow past. */
@@ -458,16 +683,130 @@ function parseImprovedSkill(output) {
458
683
  }
459
684
  return content;
460
685
  }
461
- export function mergeEvolveReports(roundReports, skillName, totalCostUSD, samples, skillPath) {
686
+ function sourceVariantForRound({ round, report }) {
687
+ if (report.meta.variants.length !== 1) {
688
+ throw new Error(`evolve 第 ${round} 轮报告必须且只能包含一个 variant。`);
689
+ }
690
+ const variant = report.meta.variants[0];
691
+ if (!ownRecordValue(report.summary, variant)
692
+ || report.results.some((entry) => !ownRecordValue(entry.variants, variant))) {
693
+ throw new Error(`evolve 第 ${round} 轮报告缺少 variant「${variant}」的完整结果。`);
694
+ }
695
+ return variant;
696
+ }
697
+ function assertComparableEvolveRounds(roundReports, sourceVariants) {
698
+ const first = roundReports[0].report;
699
+ const firstSampleIds = first.results.map((entry) => entry.sample_id);
700
+ const firstSampleSet = new Set(firstSampleIds);
701
+ if (firstSampleSet.size !== firstSampleIds.length) {
702
+ throw new Error('evolve 首轮报告包含重复 sample_id。');
703
+ }
704
+ const comparableMeta = (report, variant) => {
705
+ const config = report.meta.variantConfigs?.find((candidate) => candidate.variant === variant);
706
+ const runtime = report.meta.executorRuntimes
707
+ ? ownRecordValue(report.meta.executorRuntimes, variant)
708
+ : report.meta.executorRuntime;
709
+ const hasIsolation = Boolean(report.meta.skillIsolation
710
+ && Object.hasOwn(report.meta.skillIsolation, variant));
711
+ return {
712
+ model: report.meta.model,
713
+ executor: report.meta.executor,
714
+ effort: report.meta.effort,
715
+ schemaVersion: report.meta.schemaVersion,
716
+ cliVersion: report.meta.cliVersion,
717
+ nodeVersion: report.meta.nodeVersion,
718
+ judgeRepeat: report.meta.judgeRepeat,
719
+ noJudge: report.meta.noJudge,
720
+ judgePromptHash: report.meta.judgePromptHash,
721
+ diagnostic: report.meta.diagnostic
722
+ ? {
723
+ enabled: report.meta.diagnostic.enabled,
724
+ executor: report.meta.diagnostic.executor,
725
+ model: report.meta.diagnostic.model,
726
+ runtime: report.meta.diagnostic.runtime?.fingerprint,
727
+ promptHash: report.meta.diagnostic.promptHash,
728
+ }
729
+ : undefined,
730
+ evaluationFramework: report.meta.evaluationFramework,
731
+ debiasMode: report.meta.debiasMode,
732
+ judgeModels: report.meta.judgeModels,
733
+ sampleHashes: report.meta.sampleHashes,
734
+ requestProtocol: report.meta.request
735
+ ? {
736
+ timeoutMs: report.meta.request.timeoutMs,
737
+ retry: report.meta.request.retry ?? 0,
738
+ budget: report.meta.request.budget,
739
+ noDiagnostic: report.meta.request.noDiagnostic === true,
740
+ }
741
+ : undefined,
742
+ executorRuntime: runtime
743
+ ? {
744
+ fingerprint: runtime.fingerprint,
745
+ capabilities: runtime.capabilities,
746
+ }
747
+ : null,
748
+ execution: config
749
+ ? {
750
+ artifactKind: config.artifactKind,
751
+ executionStrategy: config.executionStrategy,
752
+ experimentType: config.experimentType,
753
+ hasArtifactContent: config.hasArtifactContent,
754
+ cwd: config.cwd,
755
+ allowedSkills: config.allowedSkills,
756
+ }
757
+ : null,
758
+ skillIsolation: {
759
+ present: hasIsolation,
760
+ value: hasIsolation
761
+ ? ownRecordValue(report.meta.skillIsolation, variant)
762
+ : undefined,
763
+ },
764
+ };
765
+ };
766
+ const expectedMeta = canonicalStringify(comparableMeta(first, sourceVariants[0]));
767
+ for (let index = 0; index < roundReports.length; index++) {
768
+ const { round, report } = roundReports[index];
769
+ if (!Number.isSafeInteger(round) || round < 0) {
770
+ throw new Error(`evolve 轮次必须是非负安全整数,收到「${String(round)}」。`);
771
+ }
772
+ if (index > 0 && round <= roundReports[index - 1].round) {
773
+ throw new Error('evolve 轮次必须严格递增且不能重复。');
774
+ }
775
+ if (!parseReportDocument(report, report.id, report.id)) {
776
+ throw new Error(`evolve 第 ${round} 轮报告不符合持久化契约。`);
777
+ }
778
+ if (report.meta.budgetExhausted === true) {
779
+ throw new Error(`evolve 第 ${round} 轮报告因预算耗尽而不完整。`);
780
+ }
781
+ if (report.meta.sampleCount !== first.meta.sampleCount
782
+ || report.results.length !== first.results.length
783
+ || canonicalStringify(comparableMeta(report, sourceVariants[index])) !== expectedMeta) {
784
+ throw new Error(`evolve 第 ${round} 轮与首轮的测量配置或用例集合不可比。`);
785
+ }
786
+ const sampleIds = new Set(report.results.map((entry) => entry.sample_id));
787
+ if (sampleIds.size !== firstSampleSet.size
788
+ || firstSampleIds.some((sampleId) => !sampleIds.has(sampleId))) {
789
+ throw new Error(`evolve 第 ${round} 轮的 sample_id 集合与首轮不一致。`);
790
+ }
791
+ if (!ownRecordValue(report.meta.artifactHashes, sourceVariants[index])) {
792
+ throw new Error(`evolve 第 ${round} 轮缺少知识载体内容指纹。`);
793
+ }
794
+ }
795
+ }
796
+ export function mergeEvolveReports(roundReports, skillName, processCostUSD, samples, skillPath, processCostReported = true) {
797
+ if (roundReports.length === 0) {
798
+ throw new Error('evolve 合并报告至少需要一轮评测结果。');
799
+ }
462
800
  const firstReport = roundReports[0].report;
801
+ const sourceVariants = roundReports.map(sourceVariantForRound);
802
+ assertComparableEvolveRounds(roundReports, sourceVariants);
463
803
  // Build variant labels: "round-0", "round-1", "round-2", ...
464
804
  const variantLabels = roundReports.map(({ round }) => `round-${round}`);
465
805
  // Build summary: map each variant label to its round's summary
466
806
  const summary = {};
467
807
  for (let i = 0; i < roundReports.length; i++) {
468
808
  const { report } = roundReports[i];
469
- const originalKey = Object.keys(report.summary)[0];
470
- summary[variantLabels[i]] = report.summary[originalKey];
809
+ setOwnRecordValue(summary, variantLabels[i], ownRecordValue(report.summary, sourceVariants[i]));
471
810
  }
472
811
  // Build results: merge per-sample variant data across rounds
473
812
  const sampleIds = firstReport.results.map((r) => r.sample_id);
@@ -475,41 +814,108 @@ export function mergeEvolveReports(roundReports, skillName, totalCostUSD, sample
475
814
  const variants = {};
476
815
  for (let i = 0; i < roundReports.length; i++) {
477
816
  const entry = roundReports[i].report.results.find((r) => r.sample_id === sampleId);
478
- if (entry) {
479
- const originalKey = Object.keys(entry.variants)[0];
480
- variants[variantLabels[i]] = entry.variants[originalKey];
481
- }
817
+ setOwnRecordValue(variants, variantLabels[i], ownRecordValue(entry.variants, sourceVariants[i]));
482
818
  }
483
819
  return { sample_id: sampleId, variants };
484
820
  });
485
- // Build variantConfigs: collect from each round, relabel variant name; mark round-0 as baseline
486
- const variantConfigs = roundReports.flatMap(({ round, report }, i) => (report.meta.variantConfigs || []).map((cfg) => ({
487
- ...cfg,
488
- variant: variantLabels[i],
489
- ...(round === 0 ? { experimentType: 'baseline', artifactKind: 'baseline', executionStrategy: 'baseline' } : {}),
490
- })));
821
+ // round-0 是当前知识载体版本的 control,不是「无知识载体」的 baseline kind。
822
+ // 保留每轮真实 artifact / execution 语义,只重标实验角色。
823
+ const variantConfigs = roundReports.map(({ report }, i) => {
824
+ const cfg = report.meta.variantConfigs?.find((candidate) => candidate.variant === sourceVariants[i]);
825
+ return cfg ? {
826
+ ...cfg,
827
+ variant: variantLabels[i],
828
+ experimentRole: i === 0 ? 'control' : 'treatment',
829
+ } : undefined;
830
+ });
491
831
  // Build artifactHashes: collect from each round, relabel key
492
832
  const artifactHashes = {};
493
833
  for (let i = 0; i < roundReports.length; i++) {
494
- const hashes = roundReports[i].report.meta.artifactHashes || {};
495
- const originalKey = Object.keys(hashes)[0];
496
- if (originalKey)
497
- artifactHashes[variantLabels[i]] = hashes[originalKey];
834
+ setOwnRecordValue(artifactHashes, variantLabels[i], ownRecordValue(roundReports[i].report.meta.artifactHashes, sourceVariants[i]));
498
835
  }
836
+ const executorRuntimes = {};
837
+ let hasCompleteExecutorRuntimes = true;
838
+ const sourceExecutorRuntimes = roundReports.map(({ report }, index) => {
839
+ const runtime = report.meta.executorRuntimes
840
+ ? ownRecordValue(report.meta.executorRuntimes, sourceVariants[index])
841
+ : report.meta.executorRuntime;
842
+ if (!runtime) {
843
+ hasCompleteExecutorRuntimes = false;
844
+ return undefined;
845
+ }
846
+ setOwnRecordValue(executorRuntimes, variantLabels[index], runtime);
847
+ return runtime;
848
+ });
849
+ const commonExecutorRuntime = sourceExecutorRuntimes[0]
850
+ && sourceExecutorRuntimes.every((runtime) => runtime !== undefined
851
+ && canonicalStringify(runtime) === canonicalStringify(sourceExecutorRuntimes[0]))
852
+ ? sourceExecutorRuntimes[0]
853
+ : undefined;
854
+ const skillIsolation = {};
855
+ let hasCompleteSkillIsolation = true;
856
+ for (let index = 0; index < roundReports.length; index++) {
857
+ const source = roundReports[index].report.meta.skillIsolation;
858
+ if (!source || !Object.hasOwn(source, sourceVariants[index])) {
859
+ hasCompleteSkillIsolation = false;
860
+ break;
861
+ }
862
+ setOwnRecordValue(skillIsolation, variantLabels[index], ownRecordValue(source, sourceVariants[index]) ?? null);
863
+ }
864
+ // request / run / job 描述的是某一次源评测,不能伪装成 aggregate 的生命周期。
865
+ // pairComparisons / humanAgreement / budget 也绑定源报告的 variant 或单轮预算,
866
+ // 合并后必须重新计算才有意义,因此不从首轮继承。
867
+ const sharedMeta = omitKeys(firstReport.meta, [
868
+ 'variants',
869
+ 'taskCount',
870
+ 'totalCostUSD',
871
+ 'totalCostReported',
872
+ 'timestamp',
873
+ 'artifactHashes',
874
+ 'executorRuntime',
875
+ 'executorRuntimes',
876
+ 'pairComparisons',
877
+ 'humanAgreement',
878
+ 'budget',
879
+ 'budgetExhausted',
880
+ 'variantConfigs',
881
+ 'skillIsolation',
882
+ 'request',
883
+ 'run',
884
+ 'job',
885
+ 'evolve',
886
+ ]);
499
887
  const runId = `evolve-${skillName}-${runIdSuffix()}`;
888
+ const measurementCostUSD = Number(results.reduce((total, entry) => total + variantLabels.reduce((sampleTotal, variant) => sampleTotal + (ownRecordValue(entry.variants, variant)?.costUSD ?? 0), 0), 0).toFixed(6));
889
+ const measurementCostReported = roundReports.every(({ report: source }) => Object.values(source.summary).every((variant) => variant.execCostReported !== false
890
+ && variant.judgeCostReported !== false));
500
891
  const report = {
501
892
  kind: 'evaluation',
502
893
  id: runId,
503
894
  meta: {
504
- ...firstReport.meta,
895
+ ...sharedMeta,
505
896
  variants: variantLabels,
506
- variantConfigs,
897
+ taskCount: firstReport.meta.sampleCount * variantLabels.length,
898
+ ...(variantConfigs.every((config) => config !== undefined)
899
+ ? { variantConfigs: variantConfigs }
900
+ : {}),
507
901
  artifactHashes,
508
- totalCostUSD: Number(totalCostUSD.toFixed(6)),
902
+ ...(hasCompleteExecutorRuntimes ? { executorRuntimes } : {}),
903
+ ...(commonExecutorRuntime ? { executorRuntime: commonExecutorRuntime } : {}),
904
+ ...(hasCompleteSkillIsolation ? { skillIsolation } : {}),
905
+ totalCostUSD: measurementCostUSD,
906
+ ...(measurementCostReported ? {} : { totalCostReported: false }),
509
907
  timestamp: new Date().toISOString(),
510
908
  evolve: {
511
909
  skillName,
512
910
  ...(skillPath ? { skillPath } : {}),
911
+ processCostUSD: Number(processCostUSD.toFixed(6)),
912
+ ...(processCostReported ? {} : { processCostReported: false }),
913
+ sourceReports: roundReports.map(({ round, accepted, report: source }, index) => ({
914
+ round,
915
+ accepted,
916
+ reportId: source.id,
917
+ variant: sourceVariants[index],
918
+ })),
513
919
  },
514
920
  },
515
921
  summary,
@@ -522,15 +928,15 @@ export function mergeEvolveReports(roundReports, skillName, totalCostUSD, sample
522
928
  report.analysis = analyzeResults(report, { samples });
523
929
  return report;
524
930
  }
525
- export async function evolveSkill({ skillPath, samplesPath, rounds = 5, target = null, stopOnAssertionsPass = false, autoFixSamples = false, sampleFixMaxAttempts = 2, reuseLatestEval = false, model = DEFAULT_MODEL, judgeModels, improveModel = DEFAULT_MODEL, improveMode = 'agent', executorName = 'claude', concurrency = 1, timeoutMs, skipConnectivity = false, effort, noDiagnostic, skipDoctor, holdoutRatio = 0, significanceGate = true, significanceAlpha = DEFAULT_BOOTSTRAP_ALPHA, testRatio = 0, editBudget = 0.2, rejectMemory = true, writeBackToSource = true, onProgress = null, onRoundProgress = null, }) {
931
+ export async function evolveSkill({ skillPath, samplesPath, rounds = 5, target = null, stopOnAssertionsPass = false, autoFixSamples = false, sampleFixMaxAttempts = 2, reuseLatestEval = false, model, judgeModels, improveModel = model, improveMode = 'agent', executorName, concurrency = 1, timeoutMs, skipConnectivity = false, effort, noDiagnostic, skipDoctor, holdoutRatio = 0, significanceGate = true, significanceAlpha = DEFAULT_BOOTSTRAP_ALPHA, testRatio = 0, editBudget = 0.2, rejectMemory = true, writeBackToSource = true, onProgress = null, onRoundProgress = null, }) {
526
932
  if (judgeModels && judgeModels.length > 1) {
527
933
  throw new Error('evolveSkill does not support multi-judge ensemble (received '
528
934
  + `${judgeModels.length} judges). Pass a single-judge array, e.g. `
529
- + `[{ executor: 'claude', model: 'haiku' }]`);
935
+ + `[{ executor: executorName, model }]`);
530
936
  }
531
937
  const effectiveJudgeModels = judgeModels && judgeModels.length > 0
532
938
  ? judgeModels
533
- : [{ executor: executorName, model: JUDGE_MODEL }];
939
+ : [{ executor: executorName, model }];
534
940
  const absSkillPath = resolve(skillPath);
535
941
  const absSamplesPath = resolve(samplesPath);
536
942
  const skillDir = dirname(absSkillPath);
@@ -635,6 +1041,7 @@ export async function evolveSkill({ skillPath, samplesPath, rounds = 5, target =
635
1041
  ? await findReusableBaselineReport({
636
1042
  artifactHash: baselineArtifactHash,
637
1043
  samplesPath: absSamplesPath,
1044
+ skillDir,
638
1045
  model,
639
1046
  executorName,
640
1047
  judgeModels: effectiveJudgeModels,
@@ -667,7 +1074,7 @@ export async function evolveSkill({ skillPath, samplesPath, rounds = 5, target =
667
1074
  if (writeBackToSource)
668
1075
  writeFileSync(absSkillPath, currentBest);
669
1076
  const { samples } = loadSamples(absSamplesPath);
670
- const mergedReport = mergeEvolveReports(roundReports, skillName, totalCostUSD, samples, absSkillPath);
1077
+ const mergedReport = mergeEvolveReports(roundReports, skillName, totalCostUSD, samples, absSkillPath, totalCostReported);
671
1078
  persistReport(mergedReport, DEFAULT_OUTPUT_DIR);
672
1079
  return {
673
1080
  startScore: bestScore,
@@ -784,6 +1191,7 @@ export async function evolveSkill({ skillPath, samplesPath, rounds = 5, target =
784
1191
  continue;
785
1192
  }
786
1193
  let preEvalSampleFixCost = 0;
1194
+ let preEvalSampleFixCostReported = true;
787
1195
  if (autoFixSamples) {
788
1196
  const sampleFix = await autoFixSamplesAfterSkillRound({
789
1197
  samplesPath: absSamplesPath,
@@ -801,7 +1209,10 @@ export async function evolveSkill({ skillPath, samplesPath, rounds = 5, target =
801
1209
  sampleFixes.push({ round, fixedCount: sampleFix.fixedCount, costUSD: sampleFix.costUSD });
802
1210
  }
803
1211
  preEvalSampleFixCost = sampleFix.costUSD;
1212
+ preEvalSampleFixCostReported = sampleFix.costReported;
804
1213
  totalCostUSD += sampleFix.costUSD;
1214
+ if (!sampleFix.costReported)
1215
+ totalCostReported = false;
805
1216
  }
806
1217
  // Evaluate candidate with any sample fixes already applied.
807
1218
  const candidateReport = await evaluate(candidatePath, {
@@ -810,7 +1221,9 @@ export async function evolveSkill({ skillPath, samplesPath, rounds = 5, target =
810
1221
  const candidateVariantKey = Object.keys(candidateReport.summary)[0];
811
1222
  const candidateScore = decisionScore(candidateReport, candidateVariantKey);
812
1223
  const roundCost = improveCostUSD + preEvalSampleFixCost + candidateReport.meta.totalCostUSD;
813
- const roundCostReported = improveCostReported && !reportHasUnreportedCost(candidateReport);
1224
+ const roundCostReported = improveCostReported
1225
+ && preEvalSampleFixCostReported
1226
+ && !reportHasUnreportedCost(candidateReport);
814
1227
  if (!roundCostReported)
815
1228
  totalCostReported = false;
816
1229
  totalCostUSD += improveCostUSD + candidateReport.meta.totalCostUSD;
@@ -875,7 +1288,7 @@ export async function evolveSkill({ skillPath, samplesPath, rounds = 5, target =
875
1288
  if (roundReports.length > 0) {
876
1289
  // load samples once to enable analysis.sampleQuality on the merged report.
877
1290
  const { samples } = loadSamples(absSamplesPath);
878
- const mergedReport = mergeEvolveReports(roundReports, skillName, totalCostUSD, samples, absSkillPath);
1291
+ const mergedReport = mergeEvolveReports(roundReports, skillName, totalCostUSD, samples, absSkillPath, totalCostReported);
879
1292
  persistReport(mergedReport, DEFAULT_OUTPUT_DIR);
880
1293
  reportId = mergedReport.id;
881
1294
  }