oh-my-knowledge 0.48.0 → 0.49.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (243) hide show
  1. package/README.md +50 -18
  2. package/README.zh.md +55 -23
  3. package/dist/analysis/coverage-analyzer.d.ts +1 -0
  4. package/dist/analysis/coverage-analyzer.js +125 -62
  5. package/dist/analysis/failure-clusterer.js +2 -1
  6. package/dist/analysis/gap-analyzer.d.ts +2 -2
  7. package/dist/analysis/gap-analyzer.js +13 -3
  8. package/dist/analysis/hedging-classifier.d.ts +2 -2
  9. package/dist/analysis/hedging-classifier.js +3 -4
  10. package/dist/analysis/report-diagnostics.js +9 -7
  11. package/dist/analysis/sample-diagnostics.js +6 -6
  12. package/dist/artifact-graph/doctor.js +15 -7
  13. package/dist/assets/agent-skills/omk/SKILL.md +27 -7
  14. package/dist/assets/agent-skills/omk/references/commands.md +18 -17
  15. package/dist/authoring/evolver.d.ts +10 -6
  16. package/dist/authoring/evolver.js +496 -83
  17. package/dist/authoring/generator.d.ts +3 -3
  18. package/dist/authoring/generator.js +5 -10
  19. package/dist/authoring/sample-fixer.d.ts +8 -6
  20. package/dist/authoring/sample-fixer.js +76 -5
  21. package/dist/cli/commands/doctor.js +31 -14
  22. package/dist/cli/commands/eval/index.d.ts +3 -0
  23. package/dist/cli/commands/eval/index.js +163 -17
  24. package/dist/cli/commands/evolve.d.ts +4 -4
  25. package/dist/cli/commands/evolve.js +27 -13
  26. package/dist/cli/commands/init.js +16 -3
  27. package/dist/cli/commands/observe/inbox.js +28 -21
  28. package/dist/cli/commands/observe/index.js +20 -11
  29. package/dist/cli/commands/observe/ingest.d.ts +3 -0
  30. package/dist/cli/commands/observe/ingest.js +30 -2
  31. package/dist/cli/commands/sample.d.ts +6 -3
  32. package/dist/cli/commands/sample.js +72 -68
  33. package/dist/cli/lib/codex-model-hint.d.ts +9 -0
  34. package/dist/cli/lib/codex-model-hint.js +45 -0
  35. package/dist/cli/lib/generation-failure-hint.d.ts +2 -0
  36. package/dist/cli/lib/generation-failure-hint.js +61 -0
  37. package/dist/cli/lib/i18n-dict/common.d.ts +1 -1
  38. package/dist/cli/lib/i18n-dict/common.js +4 -0
  39. package/dist/cli/lib/i18n-dict/gen.d.ts +1 -1
  40. package/dist/cli/lib/i18n-dict/gen.js +38 -6
  41. package/dist/cli/lib/i18n-dict/help.js +6 -6
  42. package/dist/cli/lib/i18n-dict/init.d.ts +1 -1
  43. package/dist/cli/lib/i18n-dict/init.js +13 -9
  44. package/dist/cli/lib/i18n-dict/run.d.ts +1 -1
  45. package/dist/cli/lib/i18n-dict/run.js +34 -2
  46. package/dist/cli/lib/llm-failure-classifier.d.ts +2 -0
  47. package/dist/cli/lib/llm-failure-classifier.js +8 -0
  48. package/dist/cli/lib/parse-run-config.d.ts +6 -5
  49. package/dist/cli/lib/parse-run-config.js +16 -9
  50. package/dist/cli/lib/runtime-defaults.d.ts +21 -0
  51. package/dist/cli/lib/runtime-defaults.js +79 -0
  52. package/dist/diagnosis/observe-mapper.js +14 -15
  53. package/dist/diagnosis/observe-producer.js +3 -1
  54. package/dist/diagnosis/studio-projection.js +14 -7
  55. package/dist/diagnosis/types.d.ts +2 -0
  56. package/dist/diagnosis/types.js +12 -0
  57. package/dist/doctor/endpoint-rule.js +2 -1
  58. package/dist/eval-core/artifact-file-names.js +18 -1
  59. package/dist/eval-core/artifact-index.d.ts +7 -11
  60. package/dist/eval-core/artifact-index.js +139 -80
  61. package/dist/eval-core/cache.d.ts +12 -3
  62. package/dist/eval-core/cache.js +89 -29
  63. package/dist/eval-core/comparability.js +10 -6
  64. package/dist/eval-core/evaluation-execution.d.ts +2 -1
  65. package/dist/eval-core/evaluation-execution.js +122 -37
  66. package/dist/eval-core/evaluation-job.d.ts +4 -1
  67. package/dist/eval-core/evaluation-job.js +4 -1
  68. package/dist/eval-core/evaluation-reporting.d.ts +15 -13
  69. package/dist/eval-core/evaluation-reporting.js +54 -52
  70. package/dist/eval-core/execution-strategy.d.ts +2 -0
  71. package/dist/eval-core/execution-strategy.js +11 -9
  72. package/dist/eval-core/fact-checker.js +15 -7
  73. package/dist/eval-core/holdout.js +3 -2
  74. package/dist/eval-core/judge-independence.d.ts +2 -2
  75. package/dist/eval-core/mock-hook.cjs +23 -6
  76. package/dist/eval-core/mocks-runtime.js +30 -8
  77. package/dist/eval-core/report-document.d.ts +12 -0
  78. package/dist/eval-core/report-document.js +1151 -0
  79. package/dist/eval-core/report-extensions.d.ts +4 -0
  80. package/dist/eval-core/report-extensions.js +500 -0
  81. package/dist/eval-core/report-file-migration.js +7 -2
  82. package/dist/eval-core/resume-compatibility.d.ts +31 -0
  83. package/dist/eval-core/resume-compatibility.js +141 -0
  84. package/dist/eval-core/sample-fingerprint.d.ts +12 -0
  85. package/dist/eval-core/sample-fingerprint.js +193 -0
  86. package/dist/eval-core/schema.js +86 -31
  87. package/dist/eval-core/verdict.d.ts +8 -4
  88. package/dist/eval-core/verdict.js +24 -10
  89. package/dist/eval-workflows/batch-evaluation-workflow.d.ts +2 -1
  90. package/dist/eval-workflows/batch-evaluation-workflow.js +25 -12
  91. package/dist/eval-workflows/evaluation-pipeline/preflight-warnings.d.ts +10 -5
  92. package/dist/eval-workflows/evaluation-pipeline/preflight-warnings.js +58 -21
  93. package/dist/eval-workflows/evaluation-pipeline/report-finalize.js +3 -1
  94. package/dist/eval-workflows/evaluation-pipeline/run-state.d.ts +4 -1
  95. package/dist/eval-workflows/evaluation-pipeline/run-state.js +4 -1
  96. package/dist/eval-workflows/evaluation-pipeline/test-set-hash.d.ts +6 -5
  97. package/dist/eval-workflows/evaluation-pipeline/test-set-hash.js +17 -10
  98. package/dist/eval-workflows/evaluation-pipeline.js +12 -7
  99. package/dist/eval-workflows/run-evaluation.d.ts +9 -7
  100. package/dist/eval-workflows/run-evaluation.js +79 -51
  101. package/dist/executors/anthropic-api.js +65 -9
  102. package/dist/executors/claude-cli.js +16 -79
  103. package/dist/executors/claude-protocol.d.ts +28 -0
  104. package/dist/executors/claude-protocol.js +180 -0
  105. package/dist/executors/claude-sdk-trace.js +56 -28
  106. package/dist/executors/claude-sdk.d.ts +1 -0
  107. package/dist/executors/claude-sdk.js +39 -93
  108. package/dist/executors/codex-cli-trace.js +166 -31
  109. package/dist/executors/codex-cli.d.ts +6 -8
  110. package/dist/executors/codex-cli.js +49 -151
  111. package/dist/executors/codex-protocol.d.ts +24 -0
  112. package/dist/executors/codex-protocol.js +234 -0
  113. package/dist/executors/codex-sdk.js +68 -120
  114. package/dist/executors/gemini.js +88 -13
  115. package/dist/executors/index.d.ts +2 -3
  116. package/dist/executors/index.js +5 -3
  117. package/dist/executors/openai-api.js +70 -9
  118. package/dist/executors/runtime-fingerprint.js +88 -11
  119. package/dist/executors/script-command.d.ts +8 -0
  120. package/dist/executors/script-command.js +87 -0
  121. package/dist/executors/script.js +202 -29
  122. package/dist/executors/shared.d.ts +35 -3
  123. package/dist/executors/shared.js +113 -15
  124. package/dist/grading/assertions.d.ts +1 -1
  125. package/dist/grading/assertions.js +19 -9
  126. package/dist/grading/diagnostic.d.ts +9 -2
  127. package/dist/grading/diagnostic.js +25 -2
  128. package/dist/grading/index.js +10 -4
  129. package/dist/grading/judge.js +19 -6
  130. package/dist/grading/layered-scores.d.ts +2 -3
  131. package/dist/grading/layered-scores.js +2 -3
  132. package/dist/inputs/load-samples.d.ts +1 -2
  133. package/dist/inputs/load-samples.js +23 -1
  134. package/dist/inputs/mcp-resolver.js +6 -3
  135. package/dist/inputs/sample-document.d.ts +11 -0
  136. package/dist/inputs/sample-document.js +96 -0
  137. package/dist/managed/evidence.d.ts +1 -0
  138. package/dist/managed/evidence.js +1 -1
  139. package/dist/managed/store.js +200 -91
  140. package/dist/observability/codex-trace-adapter.d.ts +5 -0
  141. package/dist/observability/codex-trace-adapter.js +850 -0
  142. package/dist/observability/experience.d.ts +32 -6
  143. package/dist/observability/experience.js +2695 -459
  144. package/dist/observability/feedback-matchers.js +16 -1
  145. package/dist/observability/inbox-view-model.d.ts +1 -1
  146. package/dist/observability/inbox-view-model.js +19 -14
  147. package/dist/observability/inbox.d.ts +7 -1
  148. package/dist/observability/inbox.js +632 -124
  149. package/dist/observability/problem-patterns.js +2 -0
  150. package/dist/observability/review-state.d.ts +6 -0
  151. package/dist/observability/review-state.js +235 -63
  152. package/dist/observability/skill-chain-advisories.js +1 -1
  153. package/dist/observability/skill-chain.js +17 -4
  154. package/dist/observability/skill-health-analyzer.d.ts +32 -7
  155. package/dist/observability/skill-health-analyzer.js +194 -121
  156. package/dist/observability/skill-health-report.d.ts +10 -0
  157. package/dist/observability/skill-health-report.js +620 -0
  158. package/dist/observability/soft-standards/constants.d.ts +0 -1
  159. package/dist/observability/soft-standards/constants.js +0 -1
  160. package/dist/observability/soft-standards/index.d.ts +1 -1
  161. package/dist/observability/soft-standards/index.js +1 -1
  162. package/dist/observability/soft-standards/llm-extractor.js +8 -10
  163. package/dist/observability/soft-standards/skill-standards-store.d.ts +2 -1
  164. package/dist/observability/soft-standards/skill-standards-store.js +59 -18
  165. package/dist/observability/soft-standards/types.d.ts +2 -2
  166. package/dist/observability/trace-adapter.d.ts +12 -7
  167. package/dist/observability/trace-adapter.js +11 -9
  168. package/dist/observability/trace-attribution.d.ts +13 -5
  169. package/dist/observability/trace-attribution.js +315 -21
  170. package/dist/observability/trace-ingestion.d.ts +9 -0
  171. package/dist/observability/trace-ingestion.js +80 -0
  172. package/dist/observability/trace-ir.d.ts +113 -0
  173. package/dist/observability/trace-ir.js +87 -0
  174. package/dist/observability/trace-segmenter.d.ts +19 -6
  175. package/dist/observability/trace-segmenter.js +377 -196
  176. package/dist/observability/trace-session-index.d.ts +19 -0
  177. package/dist/observability/trace-session-index.js +68 -0
  178. package/dist/observability/trace-source.d.ts +12 -4
  179. package/dist/observability/trace-source.js +939 -215
  180. package/dist/renderer/html-renderer.js +37 -6
  181. package/dist/renderer/icons.js +3 -0
  182. package/dist/renderer/observation-inbox-renderer.js +208 -90
  183. package/dist/renderer/skill-detail-renderer.js +452 -109
  184. package/dist/renderer/skill-health-renderer.js +69 -12
  185. package/dist/renderer/summary.js +28 -7
  186. package/dist/renderer/table.js +21 -4
  187. package/dist/renderer/test-view.d.ts +1 -0
  188. package/dist/renderer/test-view.js +44 -9
  189. package/dist/server/indexed-report-store.js +14 -18
  190. package/dist/server/job-store.js +64 -26
  191. package/dist/server/report-server.js +190 -78
  192. package/dist/server/report-store.js +57 -80
  193. package/dist/server/skill-index.js +143 -49
  194. package/dist/server/skill-insights.js +44 -5
  195. package/dist/shared/artifact-graph.d.ts +3 -0
  196. package/dist/shared/artifact-graph.js +224 -0
  197. package/dist/shared/assertion-types.d.ts +8 -0
  198. package/dist/shared/assertion-types.js +46 -0
  199. package/dist/shared/atomic-json.d.ts +8 -0
  200. package/dist/shared/atomic-json.js +33 -0
  201. package/dist/shared/diagnosis-schema.d.ts +9 -0
  202. package/dist/shared/diagnosis-schema.js +181 -0
  203. package/dist/shared/doctor-report.d.ts +3 -0
  204. package/dist/shared/doctor-report.js +103 -0
  205. package/dist/shared/evaluation-job.d.ts +6 -0
  206. package/dist/shared/evaluation-job.js +217 -0
  207. package/dist/shared/executor-result.d.ts +17 -0
  208. package/dist/shared/executor-result.js +221 -0
  209. package/dist/shared/file-lock.d.ts +12 -0
  210. package/dist/shared/file-lock.js +129 -0
  211. package/dist/shared/json-value.d.ts +5 -0
  212. package/dist/shared/json-value.js +36 -0
  213. package/dist/shared/keyed-mutex.d.ts +7 -0
  214. package/dist/shared/keyed-mutex.js +24 -0
  215. package/dist/shared/record-count.d.ts +8 -0
  216. package/dist/shared/record-count.js +43 -0
  217. package/dist/shared/sample-contract.d.ts +3 -0
  218. package/dist/shared/sample-contract.js +332 -0
  219. package/dist/shared/timestamp.d.ts +6 -0
  220. package/dist/shared/timestamp.js +64 -0
  221. package/dist/shared/token-usage.d.ts +19 -0
  222. package/dist/shared/token-usage.js +50 -0
  223. package/dist/shared/tool-call-status.d.ts +8 -0
  224. package/dist/shared/tool-call-status.js +28 -0
  225. package/dist/shared/tool-identity.d.ts +21 -0
  226. package/dist/shared/tool-identity.js +84 -0
  227. package/dist/shared/tool-search.js +73 -16
  228. package/dist/shared/trace-projection.d.ts +5 -0
  229. package/dist/shared/trace-projection.js +20 -0
  230. package/dist/shared/trace-source-kind.d.ts +3 -0
  231. package/dist/shared/trace-source-kind.js +12 -0
  232. package/dist/types/diagnosis.d.ts +2 -0
  233. package/dist/types/eval.d.ts +4 -0
  234. package/dist/types/executor.d.ts +32 -5
  235. package/dist/types/index.d.ts +1 -0
  236. package/dist/types/index.js +1 -0
  237. package/dist/types/judge.d.ts +2 -0
  238. package/dist/types/observability.d.ts +116 -9
  239. package/dist/types/report.d.ts +58 -6
  240. package/dist/types/skill-index.d.ts +7 -0
  241. package/dist/types/trace.d.ts +2 -0
  242. package/dist/types/trace.js +1 -0
  243. package/package.json +9 -5
@@ -1,19 +1,21 @@
1
1
  import { resolve } from 'node:path';
2
2
  import { DEFAULT_OUTPUT_DIR, persistReport } from '../eval-core/evaluation-reporting.js';
3
- import { createExecutor, DEFAULT_MODEL, JUDGE_MODEL } from '../executors/index.js';
3
+ import { createExecutor } from '../executors/index.js';
4
4
  import { discoverBatchSkills } from '../inputs/skill-loader.js';
5
5
  import { confidenceInterval, tTest, effectSize } from '../eval-core/statistics.js';
6
6
  import { executeBatchEvaluationRuns, buildBatchVariantSpecs } from './batch-evaluation-workflow.js';
7
7
  import { buildDryRunTaskReport, prepareEvaluationRun, } from './evaluation-preparation.js';
8
8
  import { executeEvaluationPipeline } from './evaluation-pipeline.js';
9
+ import { checkResumeCompatibility } from '../eval-core/resume-compatibility.js';
9
10
  import { findSaturationPoint } from '../analysis/saturation.js';
10
11
  import { bootstrapMeanCI, DEFAULT_BOOTSTRAP_ALPHA, DEFAULT_BOOTSTRAP_SAMPLES } from '../eval-core/bootstrap.js';
11
- export async function runEvaluation({ samplesPath, skillDir, variantSpecs = [], model = DEFAULT_MODEL, outputDir = DEFAULT_OUTPUT_DIR, project, owner, tags, noJudge = false, dryRun = false, concurrency = 1, timeoutMs, noCache = false, executorName = 'claude', jobStore = null, persistJob = true, onProgress = null, skipConnectivity = false, skipDoctor = false, lang = 'zh', mcpConfig, verbose = false, retry = 0, resume, runId, layeredStats = false, repeat, holdoutRatio, batch, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget, strictBaseline, effort, noDiagnostic, }) {
12
+ import { ownRecordValue, setOwnRecordValue } from '../shared/record-count.js';
13
+ export async function runEvaluation({ samplesPath, skillDir, variantSpecs = [], model, outputDir = DEFAULT_OUTPUT_DIR, project, owner, tags, noJudge = false, dryRun = false, concurrency = 1, timeoutMs, noCache = false, executorName, jobStore = null, persistJob = true, onProgress = null, skipConnectivity = false, skipDoctor = false, lang = 'zh', mcpConfig, verbose = false, retry = 0, resume, runId, layeredStats = false, repeat, holdoutRatio, batch, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget, strictBaseline, effort, noDiagnostic, }) {
12
14
  // Unified judgeModels → derive single-judge fields for downstream pipeline / grading
13
15
  // (which still operate on string `judgeModel` + `judgeExecutorName` fields per call).
14
16
  const effectiveJudgeModels = judgeModels && judgeModels.length > 0
15
17
  ? judgeModels
16
- : [{ executor: executorName, model: JUDGE_MODEL }];
18
+ : [{ executor: executorName, model }];
17
19
  const judgeModel = effectiveJudgeModels[0].model;
18
20
  const judgeExecutorName = effectiveJudgeModels[0].executor;
19
21
  const { samples, artifacts: resolvedArtifacts, tasks, variantNames, requires, samplesBaseDir, samplesSourceFiles } = await prepareEvaluationRun({
@@ -66,8 +68,6 @@ export async function runEvaluation({ samplesPath, skillDir, variantSpecs = [],
66
68
  }
67
69
  }
68
70
  }
69
- // --resume 时自动跳过 LLM 连通性检测(原 run 已经验过, 重跑是浪费 LLM 调用)
70
- const effectiveSkipConnectivity = resume ? true : skipConnectivity;
71
71
  if (dryRun) {
72
72
  // Emit power warnings during dry-run too — this is exactly when users
73
73
  // preview the run, the right moment to flag "you might be wasting it".
@@ -75,7 +75,10 @@ export async function runEvaluation({ samplesPath, skillDir, variantSpecs = [],
75
75
  for (const w of buildPowerWarnings(samples.length, repeat ?? 1, lang)) {
76
76
  process.stderr.write(`${w}\n`);
77
77
  }
78
- for (const w of buildIsolationWarnings(resolvedArtifacts, strictBaseline)) {
78
+ for (const w of buildIsolationWarnings(resolvedArtifacts, strictBaseline, {
79
+ executorName,
80
+ lang,
81
+ })) {
79
82
  process.stderr.write(`${w}\n`);
80
83
  }
81
84
  return {
@@ -101,27 +104,47 @@ export async function runEvaluation({ samplesPath, skillDir, variantSpecs = [],
101
104
  const store = createOverlayReportStore(resolve(outputDir || DEFAULT_OUTPUT_DIR), globalReportsDir());
102
105
  const existing = await store.get(resume);
103
106
  if (existing?.kind === 'evaluation') {
104
- existingResults = {};
105
- for (const entry of existing.results || []) {
106
- existingResults[entry.sample_id] = entry.variants;
107
- }
108
- if (onProgress) {
109
- const count = Object.values(existingResults).reduce((sum, v) => sum + Object.values(v).filter((r) => r.ok).length, 0);
110
- process.stderr.write(`\n📂 resumed ${count} completed results from report ${resume}\n`);
111
- }
112
- // Skill isolation 一致性校验:resumed report 的 meta.skillIsolation
113
- // 跟当前 strict-baseline 状态不一致时 stderr warn 不阻塞,让用户判断
114
- // 是否要 --no-cache 强制重跑(避免混合污染 / 干净 entries)。
115
- const reportIsolation = existing.meta?.skillIsolation;
116
- const expectedIsolated = strictBaseline !== false; // default true
117
- if (!reportIsolation && expectedIsolated) {
118
- process.stderr.write(`\n⚠️ resumed report ${resume} 无 meta.skillIsolation 字段,baseline 可能被 ~/.claude/skills/ 污染。\n`
119
- + ` 当前 --strict-baseline 默认开启,新 entries 会被隔离 → 与现有 entries 不可比。\n`
120
- + ` 建议:加 --no-cache 强制重跑,或显式 --no-strict-baseline 接受混合数据(慎用)。\n`);
107
+ const compatibility = checkResumeCompatibility(existing, {
108
+ variants: variantNames,
109
+ model,
110
+ executorName,
111
+ effort,
112
+ noJudge,
113
+ judgeModels: effectiveJudgeModels,
114
+ judgeRepeat,
115
+ lengthDebias,
116
+ budget,
117
+ timeoutMs,
118
+ retry,
119
+ noDiagnostic,
120
+ skillDir,
121
+ samples,
122
+ samplesBaseDir,
123
+ tasks,
124
+ artifacts: resolvedArtifacts,
125
+ });
126
+ if (compatibility.compatible) {
127
+ const sourceBySample = new Map(existing.results.map((entry) => [entry.sample_id, entry.variants]));
128
+ const resumedResults = Object.create(null);
129
+ for (const task of tasks) {
130
+ const result = sourceBySample.get(task.sample_id)?.[task.variant];
131
+ if (!result?.ok)
132
+ continue;
133
+ const variants = ownRecordValue(resumedResults, task.sample_id)
134
+ ?? setOwnRecordValue(resumedResults, task.sample_id, {});
135
+ setOwnRecordValue(variants, task.variant, result);
136
+ }
137
+ existingResults = resumedResults;
138
+ const count = Object.values(resumedResults).reduce((sum, variants) => sum + Object.keys(variants).length, 0);
139
+ process.stderr.write(lang === 'zh'
140
+ ? `\n已从报告 ${resume} 恢复 ${count} 条兼容的成功结果。\n`
141
+ : `\nResumed ${count} compatible successful result(s) from report ${resume}.\n`);
121
142
  }
122
- else if (reportIsolation && !expectedIsolated) {
123
- process.stderr.write(`\n⚠️ resumed report ${resume} 已 strict-isolated(meta.skillIsolation 存在),但本次 --no-strict-baseline。\n`
124
- + ` 新 entries 不隔离 → 与现有 entries 不可比。建议恢复默认 strict-baseline。\n`);
143
+ else {
144
+ const fields = compatibility.mismatches.join(', ');
145
+ process.stderr.write(lang === 'zh'
146
+ ? `\n报告 ${resume} 与当前评测契约不兼容,将从头运行。差异字段:${fields}。\n`
147
+ : `\nReport ${resume} is incompatible with the current evaluation contract; starting from scratch. Mismatched fields: ${fields}.\n`);
125
148
  }
126
149
  }
127
150
  else if (existing?.kind === 'batch-evaluation') {
@@ -131,6 +154,10 @@ export async function runEvaluation({ samplesPath, skillDir, variantSpecs = [],
131
154
  process.stderr.write(`\n⚠️ report ${resume} not found, starting from scratch\n`);
132
155
  }
133
156
  }
157
+ // 只有通过完整契约校验、真正接受恢复结果时,才能沿用原 run 的连通性结论。
158
+ const effectiveSkipConnectivity = existingResults !== undefined
159
+ ? true
160
+ : skipConnectivity;
134
161
  const executor = createExecutor(executorName);
135
162
  const judgeExecutor = createExecutor(judgeExecutorName || executorName);
136
163
  return executeEvaluationPipeline({
@@ -201,7 +228,7 @@ const LAYER_EXTRACTORS = {
201
228
  };
202
229
  function buildMetricStats(runs, variant, extractor) {
203
230
  const scores = runs
204
- .map((run) => extractor(run.summary?.[variant]))
231
+ .map((run) => extractor(ownRecordValue(run.summary, variant)))
205
232
  .filter((x) => typeof x === 'number');
206
233
  if (scores.length === 0)
207
234
  return null;
@@ -236,8 +263,8 @@ function buildSaturationData(runs, bootstrapSamples = DEFAULT_BOOTSTRAP_SAMPLES,
236
263
  if (variants.length === 0)
237
264
  return undefined;
238
265
  // Per-variant: cumulative composite scores after each repeat.
239
- const cumulativeByVariant = {};
240
- const tracesByVariant = {};
266
+ const cumulativeByVariant = Object.create(null);
267
+ const tracesByVariant = Object.create(null);
241
268
  const checkpointSampleCounts = [];
242
269
  const acc = Object.fromEntries(variants.map((v) => [v, []]));
243
270
  for (let runIdx = 0; runIdx < runs.length; runIdx++) {
@@ -245,47 +272,48 @@ function buildSaturationData(runs, bootstrapSamples = DEFAULT_BOOTSTRAP_SAMPLES,
245
272
  for (const variant of variants) {
246
273
  const newScores = [];
247
274
  for (const entry of run.results ?? []) {
248
- const v = entry.variants?.[variant];
275
+ const v = ownRecordValue(entry.variants, variant);
249
276
  if (!v || typeof v.compositeScore !== 'number' || v.compositeScore <= 0)
250
277
  continue;
251
278
  newScores.push(v.compositeScore);
252
279
  }
253
- acc[variant] = acc[variant].concat(newScores);
280
+ setOwnRecordValue(acc, variant, (ownRecordValue(acc, variant) ?? []).concat(newScores));
254
281
  }
255
282
  // Snapshot cumulative state for this checkpoint.
256
- const checkpointN = acc[variants[0]]?.length ?? 0;
283
+ const checkpointN = ownRecordValue(acc, variants[0])?.length ?? 0;
257
284
  checkpointSampleCounts.push(checkpointN);
258
285
  for (const variant of variants) {
259
- if (!cumulativeByVariant[variant])
260
- cumulativeByVariant[variant] = [];
261
- cumulativeByVariant[variant].push([...acc[variant]]);
286
+ const cumulative = ownRecordValue(cumulativeByVariant, variant)
287
+ ?? setOwnRecordValue(cumulativeByVariant, variant, []);
288
+ cumulative.push([...(ownRecordValue(acc, variant) ?? [])]);
262
289
  // Per-checkpoint trace: bootstrap CI on cumulative scores.
263
- const ci = bootstrapMeanCI(acc[variant], DEFAULT_BOOTSTRAP_ALPHA, bootstrapSamples, seed);
264
- if (!tracesByVariant[variant])
265
- tracesByVariant[variant] = [];
266
- tracesByVariant[variant].push({
267
- n: acc[variant].length,
290
+ const scores = ownRecordValue(acc, variant) ?? [];
291
+ const ci = bootstrapMeanCI(scores, DEFAULT_BOOTSTRAP_ALPHA, bootstrapSamples, seed);
292
+ const traces = ownRecordValue(tracesByVariant, variant)
293
+ ?? setOwnRecordValue(tracesByVariant, variant, []);
294
+ traces.push({
295
+ n: scores.length,
268
296
  mean: ci.estimate,
269
297
  ciLow: ci.low,
270
298
  ciHigh: ci.high,
271
299
  });
272
300
  }
273
301
  }
274
- const verdicts = {};
302
+ const verdicts = Object.create(null);
275
303
  if (runs.length >= 5) {
276
304
  for (const variant of variants) {
277
- const cumulative = cumulativeByVariant[variant];
305
+ const cumulative = ownRecordValue(cumulativeByVariant, variant);
278
306
  if (!cumulative)
279
307
  continue;
280
308
  const r = findSaturationPoint(cumulative, 'bootstrap-ci-width', undefined, undefined, bootstrapSamples, seed);
281
- verdicts[variant] = {
309
+ setOwnRecordValue(verdicts, variant, {
282
310
  saturated: r.saturated,
283
311
  atN: r.atN,
284
312
  confidence: r.confidence,
285
313
  method: r.method,
286
314
  threshold: r.threshold,
287
315
  reason: r.reason,
288
- };
316
+ });
289
317
  }
290
318
  }
291
319
  return {
@@ -299,7 +327,7 @@ export function buildVarianceData(runs, bootstrapSamples = DEFAULT_BOOTSTRAP_SAM
299
327
  return null;
300
328
  }
301
329
  const variants = runs[0].meta.variants || [];
302
- const perVariant = {};
330
+ const perVariant = Object.create(null);
303
331
  for (const variant of variants) {
304
332
  // Composite lives on the legacy flat fields.
305
333
  const composite = buildMetricStats(runs, variant, COMPOSITE_EXTRACTOR);
@@ -319,17 +347,17 @@ export function buildVarianceData(runs, bootstrapSamples = DEFAULT_BOOTSTRAP_SAM
319
347
  if (layerStats)
320
348
  byLayer[key] = layerStats;
321
349
  }
322
- perVariant[variant] = {
350
+ setOwnRecordValue(perVariant, variant, {
323
351
  ...composite,
324
352
  ...(Object.keys(byMetric).length > 0 ? { byMetric } : {}),
325
353
  ...(Object.keys(byLayer).length > 0 ? { byLayer } : {}),
326
- };
354
+ });
327
355
  }
328
356
  const comparisons = [];
329
357
  for (let i = 0; i < variants.length; i++) {
330
358
  for (let j = i + 1; j < variants.length; j++) {
331
- const vA = perVariant[variants[i]];
332
- const vB = perVariant[variants[j]];
359
+ const vA = ownRecordValue(perVariant, variants[i]);
360
+ const vB = ownRecordValue(perVariant, variants[j]);
333
361
  if (!vA || !vB)
334
362
  continue;
335
363
  const compositeComp = buildComparisonMetric(vA.scores, vB.scores, vA.mean, vB.mean);
@@ -367,12 +395,12 @@ export function buildVarianceData(runs, bootstrapSamples = DEFAULT_BOOTSTRAP_SAM
367
395
  ...(saturation ? { saturation } : {}),
368
396
  };
369
397
  }
370
- export async function runBatchEvaluation({ skillDir, model = DEFAULT_MODEL, outputDir = DEFAULT_OUTPUT_DIR, project, owner, tags, noJudge = false, dryRun = false, concurrency = 1, timeoutMs, executorName = 'claude', jobStore = null, persistJob = true, onProgress = null, onSkillProgress = null, skipConnectivity = false, skipDoctor = false, lang = 'zh', mcpConfig, verbose = false, repeat, holdoutRatio, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget, noCache = false, strictBaseline, variantAllowedSkills, }) {
398
+ export async function runBatchEvaluation({ skillDir, model, outputDir = DEFAULT_OUTPUT_DIR, project, owner, tags, noJudge = false, dryRun = false, concurrency = 1, timeoutMs, executorName, jobStore = null, persistJob = true, onProgress = null, onSkillProgress = null, skipConnectivity = false, skipDoctor = false, lang = 'zh', mcpConfig, verbose = false, repeat, holdoutRatio, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget, noCache = false, strictBaseline, variantAllowedSkills, }) {
371
399
  // Same unified judge derivation as runEvaluation (downstream pipeline / report build
372
400
  // still uses single judgeModel + judgeExecutorName per call).
373
401
  const effectiveJudgeModels = judgeModels && judgeModels.length > 0
374
402
  ? judgeModels
375
- : [{ executor: executorName, model: JUDGE_MODEL }];
403
+ : [{ executor: executorName, model }];
376
404
  const judgeModel = effectiveJudgeModels[0].model;
377
405
  const judgeExecutorName = effectiveJudgeModels[0].executor;
378
406
  const skillEntries = discoverBatchSkills(resolve(skillDir));
@@ -1,4 +1,5 @@
1
- import { asErrorLike, DEFAULT_TIMEOUT_MS, errorMessage } from './shared.js';
1
+ import { asErrorLike, DEFAULT_TIMEOUT_MS, errorMessage, readJsonResponse, responseBodyPreview, } from './shared.js';
2
+ import { optionalTokenCount } from '../shared/token-usage.js';
2
3
  export async function anthropicApiExecutor({ model, system, prompt, timeoutMs = DEFAULT_TIMEOUT_MS }) {
3
4
  const apiKey = process.env.ANTHROPIC_API_KEY;
4
5
  if (!apiKey)
@@ -19,17 +20,72 @@ export async function anthropicApiExecutor({ model, system, prompt, timeoutMs =
19
20
  body: JSON.stringify(reqBody),
20
21
  signal: AbortSignal.timeout(timeoutMs),
21
22
  });
22
- const data = await res.json();
23
+ const { data, rawBody } = await readJsonResponse(res);
23
24
  const durationMs = Date.now() - start;
24
25
  if (!res.ok) {
25
- return { ok: false, error: data.error?.message || `API error ${res.status}`, durationMs, durationApiMs: 0, inputTokens: 0, outputTokens: 0, cacheReadTokens: 0, cacheCreationTokens: 0, costUSD: 0, output: null, stopReason: 'error', numTurns: 0 };
26
+ const bodyPreview = responseBodyPreview(rawBody);
27
+ return { ok: false, error: data?.error?.message || `API error ${res.status}${bodyPreview ? `: ${bodyPreview}` : ''}`, durationMs, durationApiMs: 0, inputTokens: 0, outputTokens: 0, cacheReadTokens: 0, cacheCreationTokens: 0, tokenUsageReportedByExecutor: false, costUSD: 0, costReportedByExecutor: false, output: null, stopReason: 'error', numTurns: 0 };
28
+ }
29
+ if (!data) {
30
+ return { ok: false, error: 'Anthropic API returned an empty or non-JSON response', durationMs, durationApiMs: 0, inputTokens: 0, outputTokens: 0, cacheReadTokens: 0, cacheCreationTokens: 0, tokenUsageReportedByExecutor: false, costUSD: 0, costReportedByExecutor: false, output: null, stopReason: 'error', numTurns: 0 };
31
+ }
32
+ const usage = data.usage;
33
+ const inputTokens = optionalTokenCount(usage?.input_tokens);
34
+ const outputTokens = optionalTokenCount(usage?.output_tokens);
35
+ const cacheReadTokens = usage?.cache_read_input_tokens === undefined
36
+ ? 0
37
+ : optionalTokenCount(usage.cache_read_input_tokens);
38
+ const cacheCreationTokens = usage?.cache_creation_input_tokens === undefined
39
+ ? 0
40
+ : optionalTokenCount(usage.cache_creation_input_tokens);
41
+ if (inputTokens === undefined
42
+ || outputTokens === undefined
43
+ || cacheReadTokens === undefined
44
+ || cacheCreationTokens === undefined) {
45
+ return {
46
+ ok: false,
47
+ error: 'Anthropic response contained missing or invalid token usage',
48
+ durationMs,
49
+ durationApiMs: 0,
50
+ inputTokens: 0,
51
+ outputTokens: 0,
52
+ cacheReadTokens: 0,
53
+ cacheCreationTokens: 0,
54
+ tokenUsageReportedByExecutor: false,
55
+ costUSD: 0,
56
+ costReportedByExecutor: false,
57
+ output: null,
58
+ stopReason: 'error',
59
+ numTurns: 1,
60
+ };
61
+ }
62
+ const output = data.content
63
+ ?.filter((block) => block.type === undefined || block.type === 'text')
64
+ .map((block) => block.text || '')
65
+ .join('') ?? '';
66
+ if (!output.trim()) {
67
+ return {
68
+ ok: false,
69
+ error: 'Anthropic response did not contain assistant text',
70
+ durationMs,
71
+ durationApiMs: 0,
72
+ inputTokens,
73
+ outputTokens,
74
+ cacheReadTokens,
75
+ cacheCreationTokens,
76
+ costUSD: 0,
77
+ costReportedByExecutor: false,
78
+ output: null,
79
+ stopReason: 'error',
80
+ numTurns: 1,
81
+ };
26
82
  }
27
- const usage = data.usage || {};
28
83
  return {
29
- ok: true, output: data.content?.map((c) => c.text || '').join('') || '', durationMs, durationApiMs: 0,
30
- inputTokens: usage.input_tokens || 0, outputTokens: usage.output_tokens || 0,
31
- cacheReadTokens: usage.cache_read_input_tokens || 0, cacheCreationTokens: usage.cache_creation_input_tokens || 0,
32
- costUSD: 0, stopReason: data.stop_reason || 'end_turn', numTurns: 1,
84
+ ok: true, output, durationMs, durationApiMs: 0,
85
+ inputTokens, outputTokens,
86
+ cacheReadTokens, cacheCreationTokens,
87
+ costUSD: 0, costReportedByExecutor: false,
88
+ stopReason: data.stop_reason || 'end_turn', numTurns: 1,
33
89
  };
34
90
  }
35
91
  catch (err) {
@@ -37,6 +93,6 @@ export async function anthropicApiExecutor({ model, system, prompt, timeoutMs =
37
93
  const details = asErrorLike(err);
38
94
  const stopReason = details.name === 'TimeoutError' ? 'timeout' : 'error';
39
95
  const error = details.name === 'TimeoutError' ? `API request timed out after ${timeoutMs / 1000}s` : errorMessage(err);
40
- return { ok: false, error, durationMs, durationApiMs: 0, inputTokens: 0, outputTokens: 0, cacheReadTokens: 0, cacheCreationTokens: 0, costUSD: 0, output: null, stopReason, numTurns: 0 };
96
+ return { ok: false, error, durationMs, durationApiMs: 0, inputTokens: 0, outputTokens: 0, cacheReadTokens: 0, cacheCreationTokens: 0, tokenUsageReportedByExecutor: false, costUSD: 0, costReportedByExecutor: false, output: null, stopReason, numTurns: 0 };
41
97
  }
42
98
  }
@@ -1,6 +1,6 @@
1
- import { extractAgentTrace, isClaudeSdkResultMessage } from './claude-sdk-trace.js';
2
1
  import { buildExecEnv, DEFAULT_TIMEOUT_MS, errorMessage, interruptedExecResult, MAX_BUFFER, spawnWithSigintPropagation, timeoutExecResult, } from './shared.js';
3
2
  import { materializeForCliConfigDir } from '../eval-core/mocks-runtime.js';
3
+ import { buildClaudeResult, parseClaudeStreamJson } from './claude-protocol.js';
4
4
  // claude CLI 用 `--disable-slash-commands` (文档:"Disable all skills") +
5
5
  // `--disallowedTools Skill` 实现与 SDK 等价的完全隔离。非空 skill 白名单不再支持
6
6
  // (它无法真正隔离,见 claude-sdk.ts buildSdkIsolationOptions),所以 [name1, ...] 必须 throw。
@@ -19,19 +19,6 @@ function applySkillIsolationToCliArgs(args, allowedSkills) {
19
19
  // 完全隔离:双堵 main session skill 发现 + subagent Skill 工具
20
20
  args.push('--disable-slash-commands', '--disallowedTools', 'Skill');
21
21
  }
22
- function parseStreamJson(stdout) {
23
- const messages = [];
24
- for (const line of stdout.split('\n')) {
25
- const trimmed = line.trim();
26
- if (!trimmed)
27
- continue;
28
- try {
29
- messages.push(JSON.parse(trimmed));
30
- }
31
- catch { /* skip non-JSON lines */ }
32
- }
33
- return messages;
34
- }
35
22
  export async function claudeCliExecutor({ model, system, prompt, cwd, skillDir, timeoutMs = DEFAULT_TIMEOUT_MS, allowedSkills, mocks, mocksBaseDir, mocksStrict, lean, effort }) {
36
23
  const args = ['-p', prompt, '--output-format', 'stream-json', '--verbose', '--model', model,
37
24
  // 评测必须 bypass permission,否则 Bash / Edit / Write 等工具调用会卡在交互式确认。
@@ -86,40 +73,15 @@ export async function claudeCliExecutor({ model, system, prompt, cwd, skillDir,
86
73
  child.stdin?.end();
87
74
  const { stdout } = await done;
88
75
  const durationMs = Date.now() - start;
89
- const messages = parseStreamJson(stdout);
90
- // 提取 result 消息
91
- const resultMsgs = messages.filter(isClaudeSdkResultMessage);
92
- if (resultMsgs.length === 0) {
93
- const ms = captureMockStats();
94
- return {
95
- ok: false, error: 'no result message in stream-json output',
96
- durationMs, durationApiMs: 0,
97
- inputTokens: 0, outputTokens: 0, cacheReadTokens: 0, cacheCreationTokens: 0,
98
- costUSD: 0, output: null, stopReason: 'error', numTurns: 0,
99
- ...(ms && { mockStats: ms }),
100
- };
101
- }
102
- const last = resultMsgs[resultMsgs.length - 1];
103
- const usage = last.usage || {};
104
- // 提取 trace
105
- const trace = extractAgentTrace(messages);
76
+ const parsed = parseClaudeStreamJson(stdout);
106
77
  const ms = captureMockStats();
107
78
  return {
108
- ok: !last.errors?.length && last.subtype !== 'error',
109
- durationMs: last.duration_ms || durationMs,
110
- durationApiMs: last.duration_api_ms || 0,
111
- inputTokens: usage.input_tokens || 0,
112
- outputTokens: usage.output_tokens || 0,
113
- cacheReadTokens: usage.cache_read_input_tokens || 0,
114
- cacheCreationTokens: usage.cache_creation_input_tokens || 0,
115
- costUSD: last.total_cost_usd || 0,
116
- output: last.result || '',
117
- stopReason: last.subtype || 'unknown',
118
- numTurns: last.num_turns || 1,
119
- fullNumTurns: trace.fullNumTurns,
120
- numSubAgents: trace.numSubAgents,
121
- ...(trace.turns.length > 0 && { turns: trace.turns }),
122
- ...(trace.toolCalls.length > 0 && { toolCalls: trace.toolCalls }),
79
+ ...buildClaudeResult({
80
+ messages: parsed.messages,
81
+ malformedLineCount: parsed.malformedLineCount,
82
+ wallClockDurationMs: durationMs,
83
+ source: 'claude stream-json',
84
+ }),
123
85
  ...(ms && { mockStats: ms }),
124
86
  };
125
87
  }
@@ -134,41 +96,16 @@ export async function claudeCliExecutor({ model, system, prompt, cwd, skillDir,
134
96
  const ms = captureMockStats();
135
97
  return { ...interruptedExecResult(durationMs), ...(ms && { mockStats: ms }) };
136
98
  }
137
- // 尝试从 stdout 解析 stream-json(即使进程退出码非 0 也可能有 result)
138
- const messages = parseStreamJson(details.stdout || '');
139
- const resultMsgs = messages.filter(isClaudeSdkResultMessage);
140
- if (resultMsgs.length > 0) {
141
- const last = resultMsgs[resultMsgs.length - 1];
142
- const usage = last.usage || {};
143
- const trace = extractAgentTrace(messages);
144
- const ms = captureMockStats();
145
- return {
146
- ok: false,
147
- error: last.errors?.join('; ') || last.result || errorMessage(err),
148
- durationMs: last.duration_ms || durationMs,
149
- durationApiMs: last.duration_api_ms || 0,
150
- inputTokens: usage.input_tokens || 0,
151
- outputTokens: usage.output_tokens || 0,
152
- cacheReadTokens: usage.cache_read_input_tokens || 0,
153
- cacheCreationTokens: usage.cache_creation_input_tokens || 0,
154
- costUSD: last.total_cost_usd || 0,
155
- output: last.result || null,
156
- stopReason: 'error',
157
- numTurns: last.num_turns || 0,
158
- fullNumTurns: trace.fullNumTurns,
159
- numSubAgents: trace.numSubAgents,
160
- ...(trace.turns.length > 0 && { turns: trace.turns }),
161
- ...(trace.toolCalls.length > 0 && { toolCalls: trace.toolCalls }),
162
- ...(ms && { mockStats: ms }),
163
- };
164
- }
99
+ const parsed = parseClaudeStreamJson(details.stdout || '');
165
100
  const ms = captureMockStats();
166
101
  return {
167
- ok: false,
168
- error: errorMessage(err),
169
- durationMs, durationApiMs: 0,
170
- inputTokens: 0, outputTokens: 0, cacheReadTokens: 0, cacheCreationTokens: 0,
171
- costUSD: 0, output: null, stopReason: 'error', numTurns: 0,
102
+ ...buildClaudeResult({
103
+ messages: parsed.messages,
104
+ malformedLineCount: parsed.malformedLineCount,
105
+ wallClockDurationMs: durationMs,
106
+ source: 'claude stream-json',
107
+ forcedError: details.stderr?.trim() || errorMessage(err),
108
+ }),
172
109
  ...(ms && { mockStats: ms }),
173
110
  };
174
111
  }
@@ -0,0 +1,28 @@
1
+ import type { ExecResult } from '../types/index.js';
2
+ import type { ClaudeSdkBaseMessage, ClaudeSdkResultMessage } from './shared.js';
3
+ export interface ClaudeSdkMeasurements {
4
+ durationMs: number;
5
+ durationApiMs: number;
6
+ inputTokens: number;
7
+ outputTokens: number;
8
+ cacheReadTokens: number;
9
+ cacheCreationTokens: number;
10
+ costUSD: number;
11
+ numTurns: number;
12
+ }
13
+ export declare function normalizeClaudeSdkMeasurements(result: ClaudeSdkResultMessage): ClaudeSdkMeasurements | {
14
+ error: string;
15
+ };
16
+ export interface ClaudeStreamParseResult {
17
+ messages: ClaudeSdkBaseMessage[];
18
+ malformedLineCount: number;
19
+ }
20
+ export declare function parseClaudeStreamJson(stdout: string): ClaudeStreamParseResult;
21
+ export declare function buildClaudeResult(options: {
22
+ messages: ClaudeSdkBaseMessage[];
23
+ wallClockDurationMs: number;
24
+ source: 'claude stream-json' | 'claude-sdk';
25
+ malformedLineCount?: number;
26
+ forcedError?: string;
27
+ messageTimestamps?: number[];
28
+ }): ExecResult;