oh-my-knowledge 0.48.0 → 0.49.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (243) hide show
  1. package/README.md +50 -18
  2. package/README.zh.md +55 -23
  3. package/dist/analysis/coverage-analyzer.d.ts +1 -0
  4. package/dist/analysis/coverage-analyzer.js +125 -62
  5. package/dist/analysis/failure-clusterer.js +2 -1
  6. package/dist/analysis/gap-analyzer.d.ts +2 -2
  7. package/dist/analysis/gap-analyzer.js +13 -3
  8. package/dist/analysis/hedging-classifier.d.ts +2 -2
  9. package/dist/analysis/hedging-classifier.js +3 -4
  10. package/dist/analysis/report-diagnostics.js +9 -7
  11. package/dist/analysis/sample-diagnostics.js +6 -6
  12. package/dist/artifact-graph/doctor.js +15 -7
  13. package/dist/assets/agent-skills/omk/SKILL.md +27 -7
  14. package/dist/assets/agent-skills/omk/references/commands.md +18 -17
  15. package/dist/authoring/evolver.d.ts +10 -6
  16. package/dist/authoring/evolver.js +496 -83
  17. package/dist/authoring/generator.d.ts +3 -3
  18. package/dist/authoring/generator.js +5 -10
  19. package/dist/authoring/sample-fixer.d.ts +8 -6
  20. package/dist/authoring/sample-fixer.js +76 -5
  21. package/dist/cli/commands/doctor.js +31 -14
  22. package/dist/cli/commands/eval/index.d.ts +3 -0
  23. package/dist/cli/commands/eval/index.js +163 -17
  24. package/dist/cli/commands/evolve.d.ts +4 -4
  25. package/dist/cli/commands/evolve.js +27 -13
  26. package/dist/cli/commands/init.js +16 -3
  27. package/dist/cli/commands/observe/inbox.js +28 -21
  28. package/dist/cli/commands/observe/index.js +20 -11
  29. package/dist/cli/commands/observe/ingest.d.ts +3 -0
  30. package/dist/cli/commands/observe/ingest.js +30 -2
  31. package/dist/cli/commands/sample.d.ts +6 -3
  32. package/dist/cli/commands/sample.js +72 -68
  33. package/dist/cli/lib/codex-model-hint.d.ts +9 -0
  34. package/dist/cli/lib/codex-model-hint.js +45 -0
  35. package/dist/cli/lib/generation-failure-hint.d.ts +2 -0
  36. package/dist/cli/lib/generation-failure-hint.js +61 -0
  37. package/dist/cli/lib/i18n-dict/common.d.ts +1 -1
  38. package/dist/cli/lib/i18n-dict/common.js +4 -0
  39. package/dist/cli/lib/i18n-dict/gen.d.ts +1 -1
  40. package/dist/cli/lib/i18n-dict/gen.js +38 -6
  41. package/dist/cli/lib/i18n-dict/help.js +6 -6
  42. package/dist/cli/lib/i18n-dict/init.d.ts +1 -1
  43. package/dist/cli/lib/i18n-dict/init.js +13 -9
  44. package/dist/cli/lib/i18n-dict/run.d.ts +1 -1
  45. package/dist/cli/lib/i18n-dict/run.js +34 -2
  46. package/dist/cli/lib/llm-failure-classifier.d.ts +2 -0
  47. package/dist/cli/lib/llm-failure-classifier.js +8 -0
  48. package/dist/cli/lib/parse-run-config.d.ts +6 -5
  49. package/dist/cli/lib/parse-run-config.js +16 -9
  50. package/dist/cli/lib/runtime-defaults.d.ts +21 -0
  51. package/dist/cli/lib/runtime-defaults.js +79 -0
  52. package/dist/diagnosis/observe-mapper.js +14 -15
  53. package/dist/diagnosis/observe-producer.js +3 -1
  54. package/dist/diagnosis/studio-projection.js +14 -7
  55. package/dist/diagnosis/types.d.ts +2 -0
  56. package/dist/diagnosis/types.js +12 -0
  57. package/dist/doctor/endpoint-rule.js +2 -1
  58. package/dist/eval-core/artifact-file-names.js +18 -1
  59. package/dist/eval-core/artifact-index.d.ts +7 -11
  60. package/dist/eval-core/artifact-index.js +139 -80
  61. package/dist/eval-core/cache.d.ts +12 -3
  62. package/dist/eval-core/cache.js +89 -29
  63. package/dist/eval-core/comparability.js +10 -6
  64. package/dist/eval-core/evaluation-execution.d.ts +2 -1
  65. package/dist/eval-core/evaluation-execution.js +122 -37
  66. package/dist/eval-core/evaluation-job.d.ts +4 -1
  67. package/dist/eval-core/evaluation-job.js +4 -1
  68. package/dist/eval-core/evaluation-reporting.d.ts +15 -13
  69. package/dist/eval-core/evaluation-reporting.js +54 -52
  70. package/dist/eval-core/execution-strategy.d.ts +2 -0
  71. package/dist/eval-core/execution-strategy.js +11 -9
  72. package/dist/eval-core/fact-checker.js +15 -7
  73. package/dist/eval-core/holdout.js +3 -2
  74. package/dist/eval-core/judge-independence.d.ts +2 -2
  75. package/dist/eval-core/mock-hook.cjs +23 -6
  76. package/dist/eval-core/mocks-runtime.js +30 -8
  77. package/dist/eval-core/report-document.d.ts +12 -0
  78. package/dist/eval-core/report-document.js +1151 -0
  79. package/dist/eval-core/report-extensions.d.ts +4 -0
  80. package/dist/eval-core/report-extensions.js +500 -0
  81. package/dist/eval-core/report-file-migration.js +7 -2
  82. package/dist/eval-core/resume-compatibility.d.ts +31 -0
  83. package/dist/eval-core/resume-compatibility.js +141 -0
  84. package/dist/eval-core/sample-fingerprint.d.ts +12 -0
  85. package/dist/eval-core/sample-fingerprint.js +193 -0
  86. package/dist/eval-core/schema.js +86 -31
  87. package/dist/eval-core/verdict.d.ts +8 -4
  88. package/dist/eval-core/verdict.js +24 -10
  89. package/dist/eval-workflows/batch-evaluation-workflow.d.ts +2 -1
  90. package/dist/eval-workflows/batch-evaluation-workflow.js +25 -12
  91. package/dist/eval-workflows/evaluation-pipeline/preflight-warnings.d.ts +10 -5
  92. package/dist/eval-workflows/evaluation-pipeline/preflight-warnings.js +58 -21
  93. package/dist/eval-workflows/evaluation-pipeline/report-finalize.js +3 -1
  94. package/dist/eval-workflows/evaluation-pipeline/run-state.d.ts +4 -1
  95. package/dist/eval-workflows/evaluation-pipeline/run-state.js +4 -1
  96. package/dist/eval-workflows/evaluation-pipeline/test-set-hash.d.ts +6 -5
  97. package/dist/eval-workflows/evaluation-pipeline/test-set-hash.js +17 -10
  98. package/dist/eval-workflows/evaluation-pipeline.js +12 -7
  99. package/dist/eval-workflows/run-evaluation.d.ts +9 -7
  100. package/dist/eval-workflows/run-evaluation.js +79 -51
  101. package/dist/executors/anthropic-api.js +65 -9
  102. package/dist/executors/claude-cli.js +16 -79
  103. package/dist/executors/claude-protocol.d.ts +28 -0
  104. package/dist/executors/claude-protocol.js +180 -0
  105. package/dist/executors/claude-sdk-trace.js +56 -28
  106. package/dist/executors/claude-sdk.d.ts +1 -0
  107. package/dist/executors/claude-sdk.js +39 -93
  108. package/dist/executors/codex-cli-trace.js +166 -31
  109. package/dist/executors/codex-cli.d.ts +6 -8
  110. package/dist/executors/codex-cli.js +49 -151
  111. package/dist/executors/codex-protocol.d.ts +24 -0
  112. package/dist/executors/codex-protocol.js +234 -0
  113. package/dist/executors/codex-sdk.js +68 -120
  114. package/dist/executors/gemini.js +88 -13
  115. package/dist/executors/index.d.ts +2 -3
  116. package/dist/executors/index.js +5 -3
  117. package/dist/executors/openai-api.js +70 -9
  118. package/dist/executors/runtime-fingerprint.js +88 -11
  119. package/dist/executors/script-command.d.ts +8 -0
  120. package/dist/executors/script-command.js +87 -0
  121. package/dist/executors/script.js +202 -29
  122. package/dist/executors/shared.d.ts +35 -3
  123. package/dist/executors/shared.js +113 -15
  124. package/dist/grading/assertions.d.ts +1 -1
  125. package/dist/grading/assertions.js +19 -9
  126. package/dist/grading/diagnostic.d.ts +9 -2
  127. package/dist/grading/diagnostic.js +25 -2
  128. package/dist/grading/index.js +10 -4
  129. package/dist/grading/judge.js +19 -6
  130. package/dist/grading/layered-scores.d.ts +2 -3
  131. package/dist/grading/layered-scores.js +2 -3
  132. package/dist/inputs/load-samples.d.ts +1 -2
  133. package/dist/inputs/load-samples.js +23 -1
  134. package/dist/inputs/mcp-resolver.js +6 -3
  135. package/dist/inputs/sample-document.d.ts +11 -0
  136. package/dist/inputs/sample-document.js +96 -0
  137. package/dist/managed/evidence.d.ts +1 -0
  138. package/dist/managed/evidence.js +1 -1
  139. package/dist/managed/store.js +200 -91
  140. package/dist/observability/codex-trace-adapter.d.ts +5 -0
  141. package/dist/observability/codex-trace-adapter.js +850 -0
  142. package/dist/observability/experience.d.ts +32 -6
  143. package/dist/observability/experience.js +2695 -459
  144. package/dist/observability/feedback-matchers.js +16 -1
  145. package/dist/observability/inbox-view-model.d.ts +1 -1
  146. package/dist/observability/inbox-view-model.js +19 -14
  147. package/dist/observability/inbox.d.ts +7 -1
  148. package/dist/observability/inbox.js +632 -124
  149. package/dist/observability/problem-patterns.js +2 -0
  150. package/dist/observability/review-state.d.ts +6 -0
  151. package/dist/observability/review-state.js +235 -63
  152. package/dist/observability/skill-chain-advisories.js +1 -1
  153. package/dist/observability/skill-chain.js +17 -4
  154. package/dist/observability/skill-health-analyzer.d.ts +32 -7
  155. package/dist/observability/skill-health-analyzer.js +194 -121
  156. package/dist/observability/skill-health-report.d.ts +10 -0
  157. package/dist/observability/skill-health-report.js +620 -0
  158. package/dist/observability/soft-standards/constants.d.ts +0 -1
  159. package/dist/observability/soft-standards/constants.js +0 -1
  160. package/dist/observability/soft-standards/index.d.ts +1 -1
  161. package/dist/observability/soft-standards/index.js +1 -1
  162. package/dist/observability/soft-standards/llm-extractor.js +8 -10
  163. package/dist/observability/soft-standards/skill-standards-store.d.ts +2 -1
  164. package/dist/observability/soft-standards/skill-standards-store.js +59 -18
  165. package/dist/observability/soft-standards/types.d.ts +2 -2
  166. package/dist/observability/trace-adapter.d.ts +12 -7
  167. package/dist/observability/trace-adapter.js +11 -9
  168. package/dist/observability/trace-attribution.d.ts +13 -5
  169. package/dist/observability/trace-attribution.js +315 -21
  170. package/dist/observability/trace-ingestion.d.ts +9 -0
  171. package/dist/observability/trace-ingestion.js +80 -0
  172. package/dist/observability/trace-ir.d.ts +113 -0
  173. package/dist/observability/trace-ir.js +87 -0
  174. package/dist/observability/trace-segmenter.d.ts +19 -6
  175. package/dist/observability/trace-segmenter.js +377 -196
  176. package/dist/observability/trace-session-index.d.ts +19 -0
  177. package/dist/observability/trace-session-index.js +68 -0
  178. package/dist/observability/trace-source.d.ts +12 -4
  179. package/dist/observability/trace-source.js +939 -215
  180. package/dist/renderer/html-renderer.js +37 -6
  181. package/dist/renderer/icons.js +3 -0
  182. package/dist/renderer/observation-inbox-renderer.js +208 -90
  183. package/dist/renderer/skill-detail-renderer.js +452 -109
  184. package/dist/renderer/skill-health-renderer.js +69 -12
  185. package/dist/renderer/summary.js +28 -7
  186. package/dist/renderer/table.js +21 -4
  187. package/dist/renderer/test-view.d.ts +1 -0
  188. package/dist/renderer/test-view.js +44 -9
  189. package/dist/server/indexed-report-store.js +14 -18
  190. package/dist/server/job-store.js +64 -26
  191. package/dist/server/report-server.js +190 -78
  192. package/dist/server/report-store.js +57 -80
  193. package/dist/server/skill-index.js +143 -49
  194. package/dist/server/skill-insights.js +44 -5
  195. package/dist/shared/artifact-graph.d.ts +3 -0
  196. package/dist/shared/artifact-graph.js +224 -0
  197. package/dist/shared/assertion-types.d.ts +8 -0
  198. package/dist/shared/assertion-types.js +46 -0
  199. package/dist/shared/atomic-json.d.ts +8 -0
  200. package/dist/shared/atomic-json.js +33 -0
  201. package/dist/shared/diagnosis-schema.d.ts +9 -0
  202. package/dist/shared/diagnosis-schema.js +181 -0
  203. package/dist/shared/doctor-report.d.ts +3 -0
  204. package/dist/shared/doctor-report.js +103 -0
  205. package/dist/shared/evaluation-job.d.ts +6 -0
  206. package/dist/shared/evaluation-job.js +217 -0
  207. package/dist/shared/executor-result.d.ts +17 -0
  208. package/dist/shared/executor-result.js +221 -0
  209. package/dist/shared/file-lock.d.ts +12 -0
  210. package/dist/shared/file-lock.js +129 -0
  211. package/dist/shared/json-value.d.ts +5 -0
  212. package/dist/shared/json-value.js +36 -0
  213. package/dist/shared/keyed-mutex.d.ts +7 -0
  214. package/dist/shared/keyed-mutex.js +24 -0
  215. package/dist/shared/record-count.d.ts +8 -0
  216. package/dist/shared/record-count.js +43 -0
  217. package/dist/shared/sample-contract.d.ts +3 -0
  218. package/dist/shared/sample-contract.js +332 -0
  219. package/dist/shared/timestamp.d.ts +6 -0
  220. package/dist/shared/timestamp.js +64 -0
  221. package/dist/shared/token-usage.d.ts +19 -0
  222. package/dist/shared/token-usage.js +50 -0
  223. package/dist/shared/tool-call-status.d.ts +8 -0
  224. package/dist/shared/tool-call-status.js +28 -0
  225. package/dist/shared/tool-identity.d.ts +21 -0
  226. package/dist/shared/tool-identity.js +84 -0
  227. package/dist/shared/tool-search.js +73 -16
  228. package/dist/shared/trace-projection.d.ts +5 -0
  229. package/dist/shared/trace-projection.js +20 -0
  230. package/dist/shared/trace-source-kind.d.ts +3 -0
  231. package/dist/shared/trace-source-kind.js +12 -0
  232. package/dist/types/diagnosis.d.ts +2 -0
  233. package/dist/types/eval.d.ts +4 -0
  234. package/dist/types/executor.d.ts +32 -5
  235. package/dist/types/index.d.ts +1 -0
  236. package/dist/types/index.js +1 -0
  237. package/dist/types/judge.d.ts +2 -0
  238. package/dist/types/observability.d.ts +116 -9
  239. package/dist/types/report.d.ts +58 -6
  240. package/dist/types/skill-index.d.ts +7 -0
  241. package/dist/types/trace.d.ts +2 -0
  242. package/dist/types/trace.js +1 -0
  243. package/package.json +9 -5
@@ -4,9 +4,23 @@ import { safeSliceForJson } from '../util/safe-slice.js';
4
4
  import { grade } from '../grading/index.js';
5
5
  import { checkFacts } from './fact-checker.js';
6
6
  import { resolveExecutionStrategy } from './execution-strategy.js';
7
- import { DEFAULT_CACHE_DIR } from './default-dirs.js';
7
+ import { DEFAULT_CACHE_DIR, DEFAULT_ISOLATED_CWD_DIR } from './default-dirs.js';
8
8
  import { getExecutorRuntimeFingerprint } from '../executors/runtime-fingerprint.js';
9
- import { resolve, dirname } from 'node:path';
9
+ import { ownRecordValue, setOwnRecordValue, } from '../shared/record-count.js';
10
+ import { executorResultValidationError, normalizeExecResultToolIdentities, } from '../shared/executor-result.js';
11
+ import { hashSampleExecutionDependencies } from './sample-fingerprint.js';
12
+ import { resolveDiagnosticTarget } from '../grading/diagnostic.js';
13
+ import { dirname, join, resolve } from 'node:path';
14
+ import { mkdir, mkdtemp, rm } from 'node:fs/promises';
15
+ const PREFLIGHT_RUNTIME_LABEL_EXECUTORS = new Set([
16
+ 'claude',
17
+ 'claude-sdk',
18
+ 'codex',
19
+ 'codex-sdk',
20
+ 'gemini',
21
+ 'anthropic-api',
22
+ 'openai-api',
23
+ ]);
10
24
  async function runWithConcurrency(tasks, concurrency, fn) {
11
25
  let index = 0;
12
26
  async function worker() {
@@ -28,7 +42,9 @@ function makeErrorResult(error) {
28
42
  outputTokens: 0,
29
43
  cacheReadTokens: 0,
30
44
  cacheCreationTokens: 0,
45
+ tokenUsageReportedByExecutor: false,
31
46
  costUSD: 0,
47
+ costReportedByExecutor: false,
32
48
  stopReason: 'error',
33
49
  numTurns: 0,
34
50
  error: message,
@@ -37,6 +53,24 @@ function makeErrorResult(error) {
37
53
  function sleep(ms) {
38
54
  return new Promise((resolve) => setTimeout(resolve, ms));
39
55
  }
56
+ async function executeWithAttemptIsolation(executor, input, isolatedCwd) {
57
+ if (!isolatedCwd)
58
+ return executor(input);
59
+ await mkdir(DEFAULT_ISOLATED_CWD_DIR, { recursive: true });
60
+ const runtimeCwd = await mkdtemp(join(DEFAULT_ISOLATED_CWD_DIR, 'attempt-'));
61
+ try {
62
+ return await executor({ ...input, cwd: runtimeCwd });
63
+ }
64
+ finally {
65
+ try {
66
+ await rm(runtimeCwd, { recursive: true, force: true, maxRetries: 3, retryDelay: 50 });
67
+ }
68
+ catch (error) {
69
+ const message = error instanceof Error ? error.message : String(error);
70
+ process.stderr.write(`[omk] 无法清理 baseline 隔离目录 ${runtimeCwd}:${message}\n`);
71
+ }
72
+ }
73
+ }
40
74
  export async function executeTasks({ tasks, executor, executorName, model, noJudge, samplesPath, samplesBaseDir, concurrency, timeoutMs, noCache, verbose, onProgress, retry = 0, existingResults, judgeRepeat = 1, judgeModels, judgeExecutors, lengthDebias = true, budget, effort, noDiagnostic = false, }) {
41
75
  const results = {};
42
76
  let started = 0;
@@ -44,17 +78,26 @@ export async function executeTasks({ tasks, executor, executorName, model, noJud
44
78
  let skipped = 0;
45
79
  let totalCostUSD = 0;
46
80
  let budgetExhausted = false;
81
+ if (!executorName && !noCache) {
82
+ throw new Error('executeTasks requires executorName when cache is enabled; '
83
+ + 'an anonymous executor cannot have a safe cross-run cache identity');
84
+ }
85
+ const effectiveExecutorName = executorName ?? 'custom-executor';
47
86
  // Seed results from previous run (--resume)
48
87
  if (existingResults) {
49
88
  for (const [sampleId, variants] of Object.entries(existingResults)) {
50
- results[sampleId] = { ...variants };
89
+ setOwnRecordValue(results, sampleId, { ...variants });
90
+ for (const result of Object.values(variants)) {
91
+ if (result.ok)
92
+ totalCostUSD += result.costUSD;
93
+ }
51
94
  }
52
95
  }
53
96
  const cacheDir = DEFAULT_CACHE_DIR;
54
97
  const cache = noCache ? null : createCache(cacheDir);
55
98
  async function executeTask(task) {
56
99
  // Skip if already have a successful result (--resume)
57
- if (existingResults?.[task.sample_id]?.[task.variant]?.ok) {
100
+ if (ownRecordValue(ownRecordValue(existingResults ?? {}, task.sample_id) ?? {}, task.variant)?.ok) {
58
101
  skipped++;
59
102
  started++;
60
103
  completed++;
@@ -76,7 +119,6 @@ export async function executeTasks({ tasks, executor, executorName, model, noJud
76
119
  const total = tasks.length;
77
120
  onProgress?.({ phase: 'start', completed: idx, total, sample_id: task.sample_id, variant: task.variant });
78
121
  const executionPlan = resolveExecutionStrategy(task, model, timeoutMs, verbose, effort, samplesBaseDir);
79
- const effectiveExecutorName = executorName || 'claude';
80
122
  const executorRuntime = getExecutorRuntimeFingerprint(effectiveExecutorName, model, {
81
123
  skillDir: executionPlan.input.skillDir,
82
124
  });
@@ -88,7 +130,7 @@ export async function executeTasks({ tasks, executor, executorName, model, noJud
88
130
  const key = cacheKey(model, executionPlan.cacheSystem, executionPlan.input.prompt, executionPlan.input.cwd, task.artifact.allowedSkills, effectiveExecutorName, executorRuntime.fingerprint, executionPlan.input.mocks, executionPlan.input.mocksStrict, effort,
89
131
  // artifact 内容指纹进 key:本地 dir-skill 改 references/ 资产只动 contentHash,system 不变,
90
132
  // 不进 key 会命中旧输出贴到新 artifactHashes(静默污染)。
91
- task.artifact.contentHash);
133
+ task.artifact.contentHash, hashSampleExecutionDependencies(task._sample, samplesBaseDir));
92
134
  const cached = cache?.get(key);
93
135
  const execStart = Date.now();
94
136
  if (cached) {
@@ -97,25 +139,60 @@ export async function executeTasks({ tasks, executor, executorName, model, noJud
97
139
  else {
98
140
  // Execute with retry on failure
99
141
  const maxAttempts = 1 + Math.max(0, retry);
142
+ let attemptCostUSD = 0;
143
+ let attemptCostReported = true;
144
+ let attemptCount = 0;
100
145
  for (let attempt = 1; attempt <= maxAttempts; attempt++) {
101
146
  try {
102
- execResult = await executor(executionPlan.input);
147
+ execResult = await executeWithAttemptIsolation(executor, executionPlan.input, executionPlan.isolatedCwd === true);
103
148
  }
104
149
  catch (err) {
105
150
  execResult = makeErrorResult(err);
106
151
  }
107
- if (execResult.ok || attempt === maxAttempts)
152
+ const validationError = executorResultValidationError(execResult);
153
+ if (validationError) {
154
+ execResult = makeErrorResult(`executor returned invalid result: ${validationError}`);
155
+ }
156
+ else {
157
+ execResult = normalizeExecResultToolIdentities(execResult);
158
+ }
159
+ attemptCount += 1;
160
+ const nextAttemptCost = attemptCostUSD + execResult.costUSD;
161
+ if (Number.isFinite(nextAttemptCost)
162
+ && nextAttemptCost >= 0
163
+ && nextAttemptCost <= Number.MAX_SAFE_INTEGER) {
164
+ attemptCostUSD = nextAttemptCost;
165
+ }
166
+ else {
167
+ // Never turn overflow into a free execution. Saturation keeps persisted
168
+ // numbers valid and remains a conservative lower bound; the completeness
169
+ // flag tells reports that the exact amount is unavailable.
170
+ attemptCostUSD = Number.MAX_SAFE_INTEGER;
171
+ attemptCostReported = false;
172
+ }
173
+ if (execResult.costReportedByExecutor === false)
174
+ attemptCostReported = false;
175
+ const retryWouldExceedBudget = (budget?.perSampleUSD != null
176
+ && attemptCostUSD > budget.perSampleUSD) || (budget?.totalUSD != null
177
+ && totalCostUSD + attemptCostUSD > budget.totalUSD);
178
+ if (execResult.ok || attempt === maxAttempts || retryWouldExceedBudget)
108
179
  break;
109
180
  // Exponential backoff before retry
110
181
  const backoffMs = Math.min(2 ** (attempt - 1) * 1000, 30000);
111
182
  onProgress?.({ phase: 'retry', completed: idx, total, sample_id: task.sample_id, variant: task.variant, attempt, maxAttempts });
112
183
  await sleep(backoffMs);
113
184
  }
185
+ execResult = {
186
+ ...execResult,
187
+ costUSD: attemptCostUSD,
188
+ ...(attemptCostReported ? {} : { costReportedByExecutor: false }),
189
+ ...(attemptCount > 1 ? { attemptCount } : {}),
190
+ };
114
191
  if (cache && execResult.ok)
115
192
  cache.set(key, execResult);
116
193
  }
117
194
  const execMs = Date.now() - execStart;
118
- totalCostUSD += execResult.costUSD;
195
+ totalCostUSD = Math.min(Number.MAX_SAFE_INTEGER, totalCostUSD + execResult.costUSD);
119
196
  if (verbose && onProgress) {
120
197
  onProgress({
121
198
  phase: 'exec_done',
@@ -184,35 +261,26 @@ export async function executeTasks({ tasks, executor, executorName, model, noJud
184
261
  if (execResult.ok && execResult.output && task.cwd) {
185
262
  factCheck = checkFacts(execResult.output, resolve(task.cwd));
186
263
  }
187
- if (!results[task.sample_id])
188
- results[task.sample_id] = {};
264
+ const sampleResults = ownRecordValue(results, task.sample_id)
265
+ ?? setOwnRecordValue(results, task.sample_id, {});
189
266
  const variantResult = buildVariantResult(execResult, gradeResult, { execMs, gradeMs, factCheck });
190
267
  // Diagnostic — 与 judge 完全独立的"哪错了 + skill 怎么改"诊断。
191
268
  // 触发条件:noDiagnostic=false + 至少 1 条 assertion fail + sample 跑成功(有 fullOutput)。
192
269
  //
193
- // executor / model 选择:
194
- // - 优先用 judgeExecutors['claude'](已初始化好) + 'haiku' 模型 — 最便宜的标配。
195
- // - 用户没配 claude judge(只配了 openai / gemini 等)时,沿用第一个 judge 的
196
- // executor + 它对应的 model 名。硬写 'haiku' 会导致非 claude executor 拒绝
197
- // (model not found),诊断整段挂掉。
198
- // - 实在没有 judge executor 配置时回退主 executor(虽然不是 lean 也能跑)。
270
+ // executor / model 选择跟报告契约共用 resolveDiagnosticTarget:
271
+ // 跟随首位 judge;没有 judge 配置时才跟随主执行器。不得暗中偏爱某个 provider。
199
272
  const failedDetails = (gradeResult?.assertions?.details || []).filter((d) => !d.passed);
200
273
  const shouldDiagnose = !noDiagnostic && execResult.ok && failedDetails.length > 0;
201
274
  if (shouldDiagnose) {
275
+ const diagnosticStart = Date.now();
202
276
  try {
203
277
  const { runDiagnostic } = await import('../grading/diagnostic.js');
204
- // diagnostic 优先用 'claude' executor(claude 模型上跑 diagnostic 历史校准最好);
205
- // 没有就用第一个可用的 judge executor 兜底。
206
- const firstJudgeName = Object.keys(judgeExecutors)[0];
207
- const diagExecutorName = ('claude' in judgeExecutors) ? 'claude' : firstJudgeName;
208
- const diagExecutor = diagExecutorName ? judgeExecutors[diagExecutorName] : executor;
209
- // 模型选择:claude executor 走 'haiku' 标配;非 claude executor 沿用它在
210
- // judgeModels 里配的 model(用户已经验证可用)。
211
- let diagModel = 'haiku';
212
- if (diagExecutorName && diagExecutorName !== 'claude') {
213
- const judgeEntry = judgeModels.find((j) => j.executor === diagExecutorName);
214
- if (judgeEntry)
215
- diagModel = judgeEntry.model;
278
+ const diagnosticTarget = resolveDiagnosticTarget(judgeModels, effectiveExecutorName, model);
279
+ const diagExecutor = diagnosticTarget.executor === effectiveExecutorName
280
+ ? executor
281
+ : ownRecordValue(judgeExecutors, diagnosticTarget.executor);
282
+ if (!diagExecutor) {
283
+ throw new Error(`diagnostic executor "${diagnosticTarget.executor}" is not registered`);
216
284
  }
217
285
  const diagnostic = await runDiagnostic({
218
286
  sample: task._sample,
@@ -223,9 +291,12 @@ export async function executeTasks({ tasks, executor, executorName, model, noJud
223
291
  fullOutput: execResult.output || undefined,
224
292
  assertionDetails: gradeResult?.assertions?.details || [],
225
293
  executor: diagExecutor,
226
- model: diagModel,
294
+ model: diagnosticTarget.model,
227
295
  });
228
296
  variantResult.diagnostic = diagnostic;
297
+ if (diagnostic.costReportedByExecutor === false) {
298
+ variantResult.judgeCostReportedByExecutor = false;
299
+ }
229
300
  // diagnostic 成本三层对齐(reviewer PR#95 CR 2026-05-11 P2):
230
301
  // - meta.totalCostUSD 累加(下面 totalCostUSD += 这一行)
231
302
  // - variant summary 的 totalCostUSD / totalDiagnosticCostUSD 由 buildVariantSummary
@@ -240,6 +311,7 @@ export async function executeTasks({ tasks, executor, executorName, model, noJud
240
311
  }
241
312
  }
242
313
  catch (err) {
314
+ variantResult.judgeCostReportedByExecutor = false;
243
315
  // diagnostic 失败不影响主评测,降级成 minimal 错误对象
244
316
  const msg = err instanceof Error ? err.message : String(err);
245
317
  variantResult.diagnostic = {
@@ -252,6 +324,15 @@ export async function executeTasks({ tasks, executor, executorName, model, noJud
252
324
  suggestion: { skill: '', sample: '', none: '' },
253
325
  };
254
326
  }
327
+ finally {
328
+ const diagnosticMs = Date.now() - diagnosticStart;
329
+ variantResult.timing = {
330
+ execMs,
331
+ gradeMs,
332
+ diagnosticMs,
333
+ totalMs: execMs + gradeMs + diagnosticMs,
334
+ };
335
+ }
255
336
  }
256
337
  // per-sample budget enforcement. If a sample's cost or latency
257
338
  // exceeds the per-sample cap, the result is kept (so the user can see
@@ -260,11 +341,12 @@ export async function executeTasks({ tasks, executor, executorName, model, noJud
260
341
  variantResult.ok = false;
261
342
  variantResult.error = `budget overrun: per-sample cost $${variantResult.costUSD.toFixed(4)} > cap $${budget.perSampleUSD.toFixed(4)}`;
262
343
  }
263
- if (budget?.perSampleMs != null && (execMs + (gradeMs ?? 0)) > budget.perSampleMs) {
344
+ const sampleDurationMs = variantResult.timing?.totalMs ?? execMs + gradeMs;
345
+ if (budget?.perSampleMs != null && sampleDurationMs > budget.perSampleMs) {
264
346
  variantResult.ok = false;
265
- variantResult.error = `budget overrun: per-sample latency ${execMs + (gradeMs ?? 0)}ms > cap ${budget.perSampleMs}ms`;
347
+ variantResult.error = `budget overrun: per-sample latency ${sampleDurationMs}ms > cap ${budget.perSampleMs}ms`;
266
348
  }
267
- results[task.sample_id][task.variant] = variantResult;
349
+ setOwnRecordValue(sampleResults, task.variant, variantResult);
268
350
  completed++;
269
351
  onProgress?.({
270
352
  phase: 'done',
@@ -298,7 +380,10 @@ export async function executeTasks({ tasks, executor, executorName, model, noJud
298
380
  }
299
381
  return { results, totalCostUSD, skipped, budgetExhausted };
300
382
  }
301
- export async function preflight(executor, model, timeoutMs = 180000) {
383
+ export function preflightRuntimeLabel(executorName, model) {
384
+ return PREFLIGHT_RUNTIME_LABEL_EXECUTORS.has(executorName) ? `${executorName}:${model}` : `custom:${model}`;
385
+ }
386
+ export async function preflight(executor, model, timeoutMs = 180000, label) {
302
387
  const result = await executor({
303
388
  model,
304
389
  system: '',
@@ -307,7 +392,7 @@ export async function preflight(executor, model, timeoutMs = 180000) {
307
392
  timeoutMs,
308
393
  });
309
394
  if (!result.ok) {
310
- throw new Error(`preflight failed [${model}]: ${result.error}`);
395
+ throw new Error(`preflight failed [${label ?? model}]: ${result.error}`);
311
396
  }
312
397
  }
313
398
  /**
@@ -329,10 +414,10 @@ export async function preflightAllJudges(judgeModels, judgeExecutors, timeoutMs)
329
414
  if (seen.has(key))
330
415
  continue;
331
416
  seen.add(key);
332
- const exec = judgeExecutors[jc.executor];
417
+ const exec = ownRecordValue(judgeExecutors, jc.executor);
333
418
  if (!exec) {
334
419
  throw new Error(`preflight: no executor registered for "${jc.executor}" (judge "${key}"); pipeline must populate judgeExecutors before preflight`);
335
420
  }
336
- await preflight(exec, jc.model, timeoutMs);
421
+ await preflight(exec, jc.model, timeoutMs, preflightRuntimeLabel(jc.executor, jc.model));
337
422
  }
338
423
  }
@@ -1,5 +1,5 @@
1
1
  import type { Artifact, EvaluationErrorCategory, EvaluationJob, EvaluationRequest, EvaluationRun, JudgeConfig } from '../types/index.js';
2
- export declare function buildEvaluationRequest({ samplesPath, skillDir, artifacts, model, executor, noJudge, concurrency, timeoutMs, noCache, dryRun, project, owner, tags, repeat, holdoutRatio, batch, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget, effort, }: {
2
+ export declare function buildEvaluationRequest({ samplesPath, skillDir, artifacts, model, executor, noJudge, concurrency, timeoutMs, noCache, dryRun, project, owner, tags, repeat, holdoutRatio, batch, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget, strictBaseline, effort, retry, noDiagnostic, }: {
3
3
  samplesPath: string;
4
4
  skillDir: string;
5
5
  artifacts: Artifact[];
@@ -22,7 +22,10 @@ export declare function buildEvaluationRequest({ samplesPath, skillDir, artifact
22
22
  bootstrapSamples?: number;
23
23
  lengthDebias?: boolean;
24
24
  budget?: import('../types/index.js').EvalBudget;
25
+ strictBaseline?: boolean;
25
26
  effort?: 'low' | 'medium' | 'high' | 'xhigh' | 'max';
27
+ retry?: number;
28
+ noDiagnostic?: boolean;
26
29
  }): EvaluationRequest;
27
30
  export declare function createEvaluationRun(runId: string, startedAt?: string): {
28
31
  run: EvaluationRun;
@@ -1,7 +1,7 @@
1
1
  function nowIso() {
2
2
  return new Date().toISOString();
3
3
  }
4
- export function buildEvaluationRequest({ samplesPath, skillDir, artifacts, model, executor, noJudge, concurrency, timeoutMs, noCache, dryRun, project, owner, tags, repeat, holdoutRatio, batch, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget, effort, }) {
4
+ export function buildEvaluationRequest({ samplesPath, skillDir, artifacts, model, executor, noJudge, concurrency, timeoutMs, noCache, dryRun, project, owner, tags, repeat, holdoutRatio, batch, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget, strictBaseline, effort, retry, noDiagnostic, }) {
5
5
  return {
6
6
  samplesPath,
7
7
  skillDir,
@@ -25,7 +25,10 @@ export function buildEvaluationRequest({ samplesPath, skillDir, artifacts, model
25
25
  bootstrapSamples,
26
26
  lengthDebias,
27
27
  budget,
28
+ strictBaseline,
28
29
  ...(effort ? { effort } : {}),
30
+ ...(retry && retry > 0 ? { retry } : {}),
31
+ ...(noDiagnostic ? { noDiagnostic: true } : {}),
29
32
  };
30
33
  }
31
34
  export function createEvaluationRun(runId, startedAt = nowIso()) {
@@ -1,15 +1,19 @@
1
- import type { Artifact, Report, Sample, Task, VariantResult, GitInfo, EvaluationJob, EvaluationRequest, EvaluationRun } from '../types/index.js';
1
+ import { getExecutorRuntimeFingerprint } from '../executors/runtime-fingerprint.js';
2
+ import type { Artifact, Report, Sample, Task, VariantResult, GitInfo, EvaluationJob, EvaluationRequest, EvaluationRun, ReportDocument } from '../types/index.js';
2
3
  export declare const DEFAULT_OUTPUT_DIR: string;
3
- export declare const EVALUATION_REPORT_SCHEMA_VERSION = 4;
4
+ export declare const EVALUATION_REPORT_SCHEMA_VERSION = 5;
4
5
  export declare function hashString(str: string): string;
5
- /**
6
- * Stable content hash of a sample. Hashes the prompt + assertions + dimensions/rubric
7
- * (the parts that determine what's being measured). Two samples with the same hash
8
- * across runs measure the same thing; mismatched hashes mean the sample changed.
9
- */
10
- export declare function hashSample(sample: Sample): string;
6
+ export { hashSample } from './sample-fingerprint.js';
11
7
  export declare function getCliVersion(): string;
12
8
  export declare function getGitInfo(): GitInfo | null;
9
+ export declare function buildExecutorRuntimesByVariant({ variants, model, executorName, tasks, artifacts, request, }: {
10
+ variants: string[];
11
+ model: string;
12
+ executorName: string;
13
+ tasks: Task[];
14
+ artifacts: Artifact[];
15
+ request?: Pick<EvaluationRequest, 'skillDir' | 'timeoutMs'>;
16
+ }): Record<string, ReturnType<typeof getExecutorRuntimeFingerprint>>;
13
17
  interface AggregateReportOptions {
14
18
  runId: string;
15
19
  variants: string[];
@@ -18,6 +22,7 @@ interface AggregateReportOptions {
18
22
  noJudge: boolean;
19
23
  executorName: string;
20
24
  samples: Sample[];
25
+ samplesBaseDir?: string;
21
26
  tasks: Task[];
22
27
  results: Record<string, Record<string, VariantResult>>;
23
28
  totalCostUSD: number;
@@ -27,10 +32,8 @@ interface AggregateReportOptions {
27
32
  job?: EvaluationJob;
28
33
  layeredStats?: boolean;
29
34
  }
30
- export declare function aggregateReport({ runId, variants, model, judgeModel, noJudge, executorName, samples, tasks, results, totalCostUSD, artifacts, request, run, job, layeredStats, }: AggregateReportOptions): Report;
31
- export interface PersistableReport {
32
- id: string;
33
- }
35
+ export declare function aggregateReport({ runId, variants, model, judgeModel, noJudge, executorName, samples, samplesBaseDir, tasks, results, totalCostUSD, artifacts, request, run, job, layeredStats, }: AggregateReportOptions): Report;
36
+ export type PersistableReport = ReportDocument;
34
37
  export declare function persistReport(report: PersistableReport, outputDir: string | null): string | null;
35
38
  /**
36
39
  * run id 的时间戳后缀 `YYYYMMDDTHHmmss-rand4`。
@@ -40,4 +43,3 @@ export declare function persistReport(report: PersistableReport, outputDir: stri
40
43
  */
41
44
  export declare function runIdSuffix(): string;
42
45
  export declare function generateRunId(variants: string[]): string;
43
- export {};
@@ -1,17 +1,22 @@
1
- import { readFileSync, writeFileSync, mkdirSync, existsSync } from 'node:fs';
1
+ import { readFileSync, mkdirSync, existsSync } from 'node:fs';
2
2
  import { join, dirname } from 'node:path';
3
3
  import { execFileSync } from 'node:child_process';
4
4
  import { createHash } from 'node:crypto';
5
5
  import { fileURLToPath } from 'node:url';
6
6
  import { DEFAULT_REPORTS_DIR } from './default-dirs.js';
7
7
  import { indexReportWrite } from './artifact-index.js';
8
+ import { parseReportDocument } from './report-document.js';
8
9
  import { randomRunToken, reportFilePath, runTimestamp } from './artifact-file-names.js';
9
10
  import { persistEvalGraphSidecar } from '../artifact-graph/eval.js';
10
11
  import { buildVariantSummary } from './schema.js';
11
12
  import { buildVariantConfig, resolveExecutionStrategy } from './execution-strategy.js';
12
13
  import { getJudgePromptHash } from '../grading/judge.js';
14
+ import { getDiagnosticPromptHash, resolveDiagnosticTarget, } from '../grading/diagnostic.js';
13
15
  import { bootstrapMeanCI, bootstrapPairedDiffCI, DEFAULT_BOOTSTRAP_ALPHA, DEFAULT_BOOTSTRAP_SAMPLES, } from './bootstrap.js';
14
16
  import { getExecutorRuntimeFingerprint } from '../executors/runtime-fingerprint.js';
17
+ import { ownRecordValue, setOwnRecordValue, } from '../shared/record-count.js';
18
+ import { writeJsonFileAtomic } from '../shared/atomic-json.js';
19
+ import { hashSample } from './sample-fingerprint.js';
15
20
  const __dirname = dirname(fileURLToPath(import.meta.url));
16
21
  function findPackageJson(startDir) {
17
22
  let dir = startDir;
@@ -26,39 +31,11 @@ function findPackageJson(startDir) {
26
31
  const PKG = JSON.parse(readFileSync(findPackageJson(__dirname), 'utf-8'));
27
32
  // 写报告的默认目录 = reports 单一来源(default-dirs)。保留 DEFAULT_OUTPUT_DIR 名给既有 16 处 import,值统一,杜绝写/读两端漂移。
28
33
  export const DEFAULT_OUTPUT_DIR = DEFAULT_REPORTS_DIR;
29
- export const EVALUATION_REPORT_SCHEMA_VERSION = 4;
34
+ export const EVALUATION_REPORT_SCHEMA_VERSION = 5;
30
35
  export function hashString(str) {
31
36
  return createHash('sha256').update(str).digest('hex').slice(0, 12);
32
37
  }
33
- /**
34
- * Canonical (key-sorted, recursive) JSON serialization. Required for cross-run hash
35
- * stability — JS object key iteration order is implementation-defined for objects
36
- * built by spread / Object.assign / yaml.parse, so naive JSON.stringify can produce
37
- * different bytes for the "same" sample on different runs.
38
- */
39
- function canonicalStringify(value) {
40
- if (value === null || typeof value !== 'object')
41
- return JSON.stringify(value);
42
- if (Array.isArray(value))
43
- return '[' + value.map(canonicalStringify).join(',') + ']';
44
- const entries = Object.keys(value).sort();
45
- return '{' + entries.map((k) => JSON.stringify(k) + ':' + canonicalStringify(value[k])).join(',') + '}';
46
- }
47
- /**
48
- * Stable content hash of a sample. Hashes the prompt + assertions + dimensions/rubric
49
- * (the parts that determine what's being measured). Two samples with the same hash
50
- * across runs measure the same thing; mismatched hashes mean the sample changed.
51
- */
52
- export function hashSample(sample) {
53
- const stableForm = canonicalStringify({
54
- prompt: sample.prompt,
55
- rubric: sample.rubric ?? null,
56
- dimensions: sample.dimensions ?? null,
57
- assertions: sample.assertions ?? null,
58
- schema: sample.schema ?? null,
59
- });
60
- return hashString(stableForm);
61
- }
38
+ export { hashSample } from './sample-fingerprint.js';
62
39
  export function getCliVersion() {
63
40
  return PKG.version;
64
41
  }
@@ -87,18 +64,18 @@ function commonRuntime(runtimes) {
87
64
  function representativeRuntime(runtimes) {
88
65
  return Object.values(runtimes)[0];
89
66
  }
90
- function buildExecutorRuntimesByVariant({ variants, model, executorName, tasks, artifacts, request, }) {
67
+ export function buildExecutorRuntimesByVariant({ variants, model, executorName, tasks, artifacts, request, }) {
91
68
  const runtimes = {};
92
69
  for (const task of tasks) {
93
- if (runtimes[task.variant])
70
+ if (ownRecordValue(runtimes, task.variant))
94
71
  continue;
95
72
  const executionPlan = resolveExecutionStrategy(task, model, request?.timeoutMs, false);
96
- runtimes[task.variant] = getExecutorRuntimeFingerprint(executorName, model, {
73
+ setOwnRecordValue(runtimes, task.variant, getExecutorRuntimeFingerprint(executorName, model, {
97
74
  skillDir: executionPlan.input.skillDir,
98
- });
75
+ }));
99
76
  }
100
77
  for (const variant of variants) {
101
- if (runtimes[variant])
78
+ if (ownRecordValue(runtimes, variant))
102
79
  continue;
103
80
  const artifact = artifacts.find((a) => a.name === variant);
104
81
  // 与主路径 extractSkillDir 一致:dir-skill 优先隔离副本 execRoot(副本无 node_modules、PATH 不污染);
@@ -109,17 +86,19 @@ function buildExecutorRuntimesByVariant({ variants, model, executorName, tasks,
109
86
  : artifact?.locator
110
87
  ? dirname(artifact.locator)
111
88
  : request?.skillDir);
112
- runtimes[variant] = getExecutorRuntimeFingerprint(executorName, model, {
89
+ setOwnRecordValue(runtimes, variant, getExecutorRuntimeFingerprint(executorName, model, {
113
90
  skillDir: fallbackSkillDir,
114
- });
91
+ }));
115
92
  }
116
93
  return runtimes;
117
94
  }
118
- export function aggregateReport({ runId, variants, model, judgeModel, noJudge, executorName, samples, tasks, results, totalCostUSD, artifacts, request, run, job, layeredStats, }) {
95
+ export function aggregateReport({ runId, variants, model, judgeModel, noJudge, executorName, samples, samplesBaseDir, tasks, results, totalCostUSD, artifacts, request, run, job, layeredStats, }) {
119
96
  const summary = {};
120
97
  for (const variant of variants) {
121
- const entries = Object.values(results).map((result) => result[variant]).filter(Boolean);
122
- summary[variant] = buildVariantSummary(entries);
98
+ const entries = Object.values(results)
99
+ .map((result) => ownRecordValue(result, variant))
100
+ .filter((entry) => Boolean(entry));
101
+ setOwnRecordValue(summary, variant, buildVariantSummary(entries));
123
102
  }
124
103
  // Bootstrap CI (per-variant mean) when --bootstrap requested. Adds bootstrapCI to
125
104
  // each VariantSummary; legacy t-interval (in summary's other fields) is preserved.
@@ -131,7 +110,9 @@ export function aggregateReport({ runId, variants, model, judgeModel, noJudge, e
131
110
  // 当且仅当该样本**无任何可测层**(真·缺测,如纯评委样本且评委失败)。故 `> 0` 过滤精确剔除非测量、
132
111
  // 绝不丢"低分内容"(评委失败已在上游当缺测,不会以 0 进 composite)。下同(control / treatment)。
133
112
  for (const variant of variants) {
134
- const entries = Object.values(results).map((r) => r[variant]).filter(Boolean);
113
+ const entries = Object.values(results)
114
+ .map((r) => ownRecordValue(r, variant))
115
+ .filter((entry) => Boolean(entry));
135
116
  const compositeScores = entries
136
117
  .filter((e) => typeof e.compositeScore === 'number' && e.compositeScore > 0)
137
118
  .map((e) => e.compositeScore);
@@ -155,8 +136,8 @@ export function aggregateReport({ runId, variants, model, judgeModel, noJudge, e
155
136
  const treatmentName = variants[i];
156
137
  const pairs = [];
157
138
  for (const r of sampleRecords) {
158
- const c = r[controlName];
159
- const t = r[treatmentName];
139
+ const c = ownRecordValue(r, controlName);
140
+ const t = ownRecordValue(r, treatmentName);
160
141
  const a = c && typeof c.compositeScore === 'number' && c.compositeScore > 0 ? c.compositeScore : undefined;
161
142
  const b = t && typeof t.compositeScore === 'number' && t.compositeScore > 0 ? t.compositeScore : undefined;
162
143
  if (a !== undefined && b !== undefined)
@@ -188,7 +169,7 @@ export function aggregateReport({ runId, variants, model, judgeModel, noJudge, e
188
169
  // contentHash 落在同一空间——证据可绑定的前提,也修掉「只哈 SKILL.md 正文、改资产指纹不变」的资产瞎。
189
170
  // baseline / 无 skill 记 'no-skill'。
190
171
  const artifactHashes = Object.fromEntries(artifacts.map((artifact) => [artifact.name, artifact.contentHash ?? 'no-skill']));
191
- const sampleHashes = Object.fromEntries(samples.map((s) => [s.sample_id, hashSample(s)]));
172
+ const sampleHashes = Object.fromEntries(samples.map((sample) => [sample.sample_id, hashSample(sample, samplesBaseDir)]));
192
173
  const judgeRepeat = request?.judgeRepeat && request.judgeRepeat > 1 ? request.judgeRepeat : undefined;
193
174
  const runtimeOptions = { skillDir: request?.skillDir };
194
175
  const executorRuntimes = buildExecutorRuntimesByVariant({ variants, model, executorName, tasks, artifacts, request });
@@ -204,6 +185,17 @@ export function aggregateReport({ runId, variants, model, judgeModel, noJudge, e
204
185
  model: jc.model,
205
186
  ...(noJudge ? {} : { runtime: getExecutorRuntimeFingerprint(jc.executor, jc.model, runtimeOptions) }),
206
187
  }));
188
+ const diagnosticEnabled = request?.noDiagnostic !== true;
189
+ const diagnosticTarget = resolveDiagnosticTarget(requestJudges, executorName, model);
190
+ const diagnostic = diagnosticEnabled
191
+ ? {
192
+ enabled: true,
193
+ executor: diagnosticTarget.executor,
194
+ model: diagnosticTarget.model,
195
+ runtime: getExecutorRuntimeFingerprint(diagnosticTarget.executor, diagnosticTarget.model),
196
+ promptHash: getDiagnosticPromptHash(),
197
+ }
198
+ : { enabled: false };
207
199
  // length-debias is on by default; the request only sets it
208
200
  // false when the user passed --no-debias-length. The judgePromptHash differs between
209
201
  // the length-debias-on and -off prompt variants so readers can detect the divergence.
@@ -235,6 +227,7 @@ export function aggregateReport({ runId, variants, model, judgeModel, noJudge, e
235
227
  artifactHashes,
236
228
  sampleHashes,
237
229
  ...(noJudge ? {} : { judgePromptHash: getJudgePromptHash(lengthDebiasOn) }),
230
+ diagnostic,
238
231
  executorRuntime,
239
232
  executorRuntimes,
240
233
  judgeModels: judgeModelsMeta,
@@ -259,16 +252,22 @@ export function aggregateReport({ runId, variants, model, judgeModel, noJudge, e
259
252
  sample_id,
260
253
  variants: variantData,
261
254
  })),
262
- // 用例设计快照,供单测视角渲染。只挑渲染需要的字段,跳过 cwd / allowedTools /
263
- // expectedTools / dimensions / environment(对单测视图无附加价值)。Sample 字段
264
- // 全选会让 report 体积接近翻倍,选子集 size 增长 ~10-20%。
255
+ // 用例设计快照,供单测视角与证据审计使用。执行/评分语义字段必须保留:
256
+ // cwd / environment / mocksStrict / allowedTools 等会改变真实构造,不能只留一个
257
+ // 不可解释的 sampleHash。纯扩展字段仍不盲目全选,控制报告体积。
265
258
  sampleSnapshots: Object.fromEntries(samples.map((s) => [s.sample_id, {
266
259
  sample_id: s.sample_id,
267
260
  prompt: s.prompt,
261
+ ...(s.cwd ? { cwd: s.cwd } : {}),
268
262
  ...(s.rubric ? { rubric: s.rubric } : {}),
269
263
  ...(s.context ? { context: s.context } : {}),
264
+ ...(s.dimensions && Object.keys(s.dimensions).length > 0 ? { dimensions: s.dimensions } : {}),
270
265
  ...(s.assertions && s.assertions.length > 0 ? { assertions: s.assertions } : {}),
271
266
  ...(s.mocks && s.mocks.length > 0 ? { mocks: s.mocks } : {}),
267
+ ...(s.mocksStrict !== undefined ? { mocksStrict: s.mocksStrict } : {}),
268
+ ...(s.environment ? { environment: s.environment } : {}),
269
+ ...(s.allowedTools && s.allowedTools.length > 0 ? { allowedTools: s.allowedTools } : {}),
270
+ ...(s.expectedTools && s.expectedTools.length > 0 ? { expectedTools: s.expectedTools } : {}),
272
271
  ...(s.capability && s.capability.length > 0 ? { capability: s.capability } : {}),
273
272
  ...(s.difficulty ? { difficulty: s.difficulty } : {}),
274
273
  ...(s.construct ? { construct: s.construct } : {}),
@@ -279,7 +278,7 @@ export function aggregateReport({ runId, variants, model, judgeModel, noJudge, e
279
278
  };
280
279
  }
281
280
  function isEvaluationReport(report) {
282
- return report['kind'] === 'evaluation';
281
+ return report.kind === 'evaluation';
283
282
  }
284
283
  function persistEvalGraphSidecarSafely(report, outputDir, sourcePath) {
285
284
  if (!isEvaluationReport(report))
@@ -298,11 +297,14 @@ export function persistReport(report, outputDir) {
298
297
  if (!existsSync(outputDir))
299
298
  mkdirSync(outputDir, { recursive: true });
300
299
  const filePath = reportFilePath(outputDir, report.id);
301
- writeFileSync(filePath, JSON.stringify(report, null, 2));
302
- persistEvalGraphSidecarSafely(report, outputDir, filePath);
300
+ const parsed = parseReportDocument(report, report.id, report.id);
301
+ if (!parsed)
302
+ throw new Error('invalid report');
303
+ writeJsonFileAtomic(filePath, parsed);
304
+ persistEvalGraphSidecarSafely(parsed, outputDir, filePath);
303
305
  // 产物发现索引:报告落项目本地后,best-effort 追加全局轻卡片,让 omk studio 跨项目聚合成机器级总览。
304
306
  // 永不抛、永不阻断报告落盘(正文是 source of truth)。
305
- indexReportWrite(report, filePath, outputDir);
307
+ indexReportWrite(parsed, filePath, outputDir);
306
308
  return filePath;
307
309
  }
308
310
  /**
@@ -3,6 +3,8 @@ export interface ExecutionPlan {
3
3
  strategy: ExecutionStrategyKind;
4
4
  cacheSystem: string;
5
5
  input: ExecutorInput;
6
+ /** Replace the logical cwd with a fresh empty directory for each attempt. */
7
+ isolatedCwd?: boolean;
6
8
  }
7
9
  export declare function resolveArtifactExecutionStrategy(artifact: Artifact): ExecutionStrategyKind;
8
10
  export declare function resolveExperimentType(artifact: Artifact): ExperimentType;