oh-my-knowledge 0.48.0 → 0.49.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (243) hide show
  1. package/README.md +50 -18
  2. package/README.zh.md +55 -23
  3. package/dist/analysis/coverage-analyzer.d.ts +1 -0
  4. package/dist/analysis/coverage-analyzer.js +125 -62
  5. package/dist/analysis/failure-clusterer.js +2 -1
  6. package/dist/analysis/gap-analyzer.d.ts +2 -2
  7. package/dist/analysis/gap-analyzer.js +13 -3
  8. package/dist/analysis/hedging-classifier.d.ts +2 -2
  9. package/dist/analysis/hedging-classifier.js +3 -4
  10. package/dist/analysis/report-diagnostics.js +9 -7
  11. package/dist/analysis/sample-diagnostics.js +6 -6
  12. package/dist/artifact-graph/doctor.js +15 -7
  13. package/dist/assets/agent-skills/omk/SKILL.md +27 -7
  14. package/dist/assets/agent-skills/omk/references/commands.md +18 -17
  15. package/dist/authoring/evolver.d.ts +10 -6
  16. package/dist/authoring/evolver.js +496 -83
  17. package/dist/authoring/generator.d.ts +3 -3
  18. package/dist/authoring/generator.js +5 -10
  19. package/dist/authoring/sample-fixer.d.ts +8 -6
  20. package/dist/authoring/sample-fixer.js +76 -5
  21. package/dist/cli/commands/doctor.js +31 -14
  22. package/dist/cli/commands/eval/index.d.ts +3 -0
  23. package/dist/cli/commands/eval/index.js +163 -17
  24. package/dist/cli/commands/evolve.d.ts +4 -4
  25. package/dist/cli/commands/evolve.js +27 -13
  26. package/dist/cli/commands/init.js +16 -3
  27. package/dist/cli/commands/observe/inbox.js +28 -21
  28. package/dist/cli/commands/observe/index.js +20 -11
  29. package/dist/cli/commands/observe/ingest.d.ts +3 -0
  30. package/dist/cli/commands/observe/ingest.js +30 -2
  31. package/dist/cli/commands/sample.d.ts +6 -3
  32. package/dist/cli/commands/sample.js +72 -68
  33. package/dist/cli/lib/codex-model-hint.d.ts +9 -0
  34. package/dist/cli/lib/codex-model-hint.js +45 -0
  35. package/dist/cli/lib/generation-failure-hint.d.ts +2 -0
  36. package/dist/cli/lib/generation-failure-hint.js +61 -0
  37. package/dist/cli/lib/i18n-dict/common.d.ts +1 -1
  38. package/dist/cli/lib/i18n-dict/common.js +4 -0
  39. package/dist/cli/lib/i18n-dict/gen.d.ts +1 -1
  40. package/dist/cli/lib/i18n-dict/gen.js +38 -6
  41. package/dist/cli/lib/i18n-dict/help.js +6 -6
  42. package/dist/cli/lib/i18n-dict/init.d.ts +1 -1
  43. package/dist/cli/lib/i18n-dict/init.js +13 -9
  44. package/dist/cli/lib/i18n-dict/run.d.ts +1 -1
  45. package/dist/cli/lib/i18n-dict/run.js +34 -2
  46. package/dist/cli/lib/llm-failure-classifier.d.ts +2 -0
  47. package/dist/cli/lib/llm-failure-classifier.js +8 -0
  48. package/dist/cli/lib/parse-run-config.d.ts +6 -5
  49. package/dist/cli/lib/parse-run-config.js +16 -9
  50. package/dist/cli/lib/runtime-defaults.d.ts +21 -0
  51. package/dist/cli/lib/runtime-defaults.js +79 -0
  52. package/dist/diagnosis/observe-mapper.js +14 -15
  53. package/dist/diagnosis/observe-producer.js +3 -1
  54. package/dist/diagnosis/studio-projection.js +14 -7
  55. package/dist/diagnosis/types.d.ts +2 -0
  56. package/dist/diagnosis/types.js +12 -0
  57. package/dist/doctor/endpoint-rule.js +2 -1
  58. package/dist/eval-core/artifact-file-names.js +18 -1
  59. package/dist/eval-core/artifact-index.d.ts +7 -11
  60. package/dist/eval-core/artifact-index.js +139 -80
  61. package/dist/eval-core/cache.d.ts +12 -3
  62. package/dist/eval-core/cache.js +89 -29
  63. package/dist/eval-core/comparability.js +10 -6
  64. package/dist/eval-core/evaluation-execution.d.ts +2 -1
  65. package/dist/eval-core/evaluation-execution.js +122 -37
  66. package/dist/eval-core/evaluation-job.d.ts +4 -1
  67. package/dist/eval-core/evaluation-job.js +4 -1
  68. package/dist/eval-core/evaluation-reporting.d.ts +15 -13
  69. package/dist/eval-core/evaluation-reporting.js +54 -52
  70. package/dist/eval-core/execution-strategy.d.ts +2 -0
  71. package/dist/eval-core/execution-strategy.js +11 -9
  72. package/dist/eval-core/fact-checker.js +15 -7
  73. package/dist/eval-core/holdout.js +3 -2
  74. package/dist/eval-core/judge-independence.d.ts +2 -2
  75. package/dist/eval-core/mock-hook.cjs +23 -6
  76. package/dist/eval-core/mocks-runtime.js +30 -8
  77. package/dist/eval-core/report-document.d.ts +12 -0
  78. package/dist/eval-core/report-document.js +1151 -0
  79. package/dist/eval-core/report-extensions.d.ts +4 -0
  80. package/dist/eval-core/report-extensions.js +500 -0
  81. package/dist/eval-core/report-file-migration.js +7 -2
  82. package/dist/eval-core/resume-compatibility.d.ts +31 -0
  83. package/dist/eval-core/resume-compatibility.js +141 -0
  84. package/dist/eval-core/sample-fingerprint.d.ts +12 -0
  85. package/dist/eval-core/sample-fingerprint.js +193 -0
  86. package/dist/eval-core/schema.js +86 -31
  87. package/dist/eval-core/verdict.d.ts +8 -4
  88. package/dist/eval-core/verdict.js +24 -10
  89. package/dist/eval-workflows/batch-evaluation-workflow.d.ts +2 -1
  90. package/dist/eval-workflows/batch-evaluation-workflow.js +25 -12
  91. package/dist/eval-workflows/evaluation-pipeline/preflight-warnings.d.ts +10 -5
  92. package/dist/eval-workflows/evaluation-pipeline/preflight-warnings.js +58 -21
  93. package/dist/eval-workflows/evaluation-pipeline/report-finalize.js +3 -1
  94. package/dist/eval-workflows/evaluation-pipeline/run-state.d.ts +4 -1
  95. package/dist/eval-workflows/evaluation-pipeline/run-state.js +4 -1
  96. package/dist/eval-workflows/evaluation-pipeline/test-set-hash.d.ts +6 -5
  97. package/dist/eval-workflows/evaluation-pipeline/test-set-hash.js +17 -10
  98. package/dist/eval-workflows/evaluation-pipeline.js +12 -7
  99. package/dist/eval-workflows/run-evaluation.d.ts +9 -7
  100. package/dist/eval-workflows/run-evaluation.js +79 -51
  101. package/dist/executors/anthropic-api.js +65 -9
  102. package/dist/executors/claude-cli.js +16 -79
  103. package/dist/executors/claude-protocol.d.ts +28 -0
  104. package/dist/executors/claude-protocol.js +180 -0
  105. package/dist/executors/claude-sdk-trace.js +56 -28
  106. package/dist/executors/claude-sdk.d.ts +1 -0
  107. package/dist/executors/claude-sdk.js +39 -93
  108. package/dist/executors/codex-cli-trace.js +166 -31
  109. package/dist/executors/codex-cli.d.ts +6 -8
  110. package/dist/executors/codex-cli.js +49 -151
  111. package/dist/executors/codex-protocol.d.ts +24 -0
  112. package/dist/executors/codex-protocol.js +234 -0
  113. package/dist/executors/codex-sdk.js +68 -120
  114. package/dist/executors/gemini.js +88 -13
  115. package/dist/executors/index.d.ts +2 -3
  116. package/dist/executors/index.js +5 -3
  117. package/dist/executors/openai-api.js +70 -9
  118. package/dist/executors/runtime-fingerprint.js +88 -11
  119. package/dist/executors/script-command.d.ts +8 -0
  120. package/dist/executors/script-command.js +87 -0
  121. package/dist/executors/script.js +202 -29
  122. package/dist/executors/shared.d.ts +35 -3
  123. package/dist/executors/shared.js +113 -15
  124. package/dist/grading/assertions.d.ts +1 -1
  125. package/dist/grading/assertions.js +19 -9
  126. package/dist/grading/diagnostic.d.ts +9 -2
  127. package/dist/grading/diagnostic.js +25 -2
  128. package/dist/grading/index.js +10 -4
  129. package/dist/grading/judge.js +19 -6
  130. package/dist/grading/layered-scores.d.ts +2 -3
  131. package/dist/grading/layered-scores.js +2 -3
  132. package/dist/inputs/load-samples.d.ts +1 -2
  133. package/dist/inputs/load-samples.js +23 -1
  134. package/dist/inputs/mcp-resolver.js +6 -3
  135. package/dist/inputs/sample-document.d.ts +11 -0
  136. package/dist/inputs/sample-document.js +96 -0
  137. package/dist/managed/evidence.d.ts +1 -0
  138. package/dist/managed/evidence.js +1 -1
  139. package/dist/managed/store.js +200 -91
  140. package/dist/observability/codex-trace-adapter.d.ts +5 -0
  141. package/dist/observability/codex-trace-adapter.js +850 -0
  142. package/dist/observability/experience.d.ts +32 -6
  143. package/dist/observability/experience.js +2695 -459
  144. package/dist/observability/feedback-matchers.js +16 -1
  145. package/dist/observability/inbox-view-model.d.ts +1 -1
  146. package/dist/observability/inbox-view-model.js +19 -14
  147. package/dist/observability/inbox.d.ts +7 -1
  148. package/dist/observability/inbox.js +632 -124
  149. package/dist/observability/problem-patterns.js +2 -0
  150. package/dist/observability/review-state.d.ts +6 -0
  151. package/dist/observability/review-state.js +235 -63
  152. package/dist/observability/skill-chain-advisories.js +1 -1
  153. package/dist/observability/skill-chain.js +17 -4
  154. package/dist/observability/skill-health-analyzer.d.ts +32 -7
  155. package/dist/observability/skill-health-analyzer.js +194 -121
  156. package/dist/observability/skill-health-report.d.ts +10 -0
  157. package/dist/observability/skill-health-report.js +620 -0
  158. package/dist/observability/soft-standards/constants.d.ts +0 -1
  159. package/dist/observability/soft-standards/constants.js +0 -1
  160. package/dist/observability/soft-standards/index.d.ts +1 -1
  161. package/dist/observability/soft-standards/index.js +1 -1
  162. package/dist/observability/soft-standards/llm-extractor.js +8 -10
  163. package/dist/observability/soft-standards/skill-standards-store.d.ts +2 -1
  164. package/dist/observability/soft-standards/skill-standards-store.js +59 -18
  165. package/dist/observability/soft-standards/types.d.ts +2 -2
  166. package/dist/observability/trace-adapter.d.ts +12 -7
  167. package/dist/observability/trace-adapter.js +11 -9
  168. package/dist/observability/trace-attribution.d.ts +13 -5
  169. package/dist/observability/trace-attribution.js +315 -21
  170. package/dist/observability/trace-ingestion.d.ts +9 -0
  171. package/dist/observability/trace-ingestion.js +80 -0
  172. package/dist/observability/trace-ir.d.ts +113 -0
  173. package/dist/observability/trace-ir.js +87 -0
  174. package/dist/observability/trace-segmenter.d.ts +19 -6
  175. package/dist/observability/trace-segmenter.js +377 -196
  176. package/dist/observability/trace-session-index.d.ts +19 -0
  177. package/dist/observability/trace-session-index.js +68 -0
  178. package/dist/observability/trace-source.d.ts +12 -4
  179. package/dist/observability/trace-source.js +939 -215
  180. package/dist/renderer/html-renderer.js +37 -6
  181. package/dist/renderer/icons.js +3 -0
  182. package/dist/renderer/observation-inbox-renderer.js +208 -90
  183. package/dist/renderer/skill-detail-renderer.js +452 -109
  184. package/dist/renderer/skill-health-renderer.js +69 -12
  185. package/dist/renderer/summary.js +28 -7
  186. package/dist/renderer/table.js +21 -4
  187. package/dist/renderer/test-view.d.ts +1 -0
  188. package/dist/renderer/test-view.js +44 -9
  189. package/dist/server/indexed-report-store.js +14 -18
  190. package/dist/server/job-store.js +64 -26
  191. package/dist/server/report-server.js +190 -78
  192. package/dist/server/report-store.js +57 -80
  193. package/dist/server/skill-index.js +143 -49
  194. package/dist/server/skill-insights.js +44 -5
  195. package/dist/shared/artifact-graph.d.ts +3 -0
  196. package/dist/shared/artifact-graph.js +224 -0
  197. package/dist/shared/assertion-types.d.ts +8 -0
  198. package/dist/shared/assertion-types.js +46 -0
  199. package/dist/shared/atomic-json.d.ts +8 -0
  200. package/dist/shared/atomic-json.js +33 -0
  201. package/dist/shared/diagnosis-schema.d.ts +9 -0
  202. package/dist/shared/diagnosis-schema.js +181 -0
  203. package/dist/shared/doctor-report.d.ts +3 -0
  204. package/dist/shared/doctor-report.js +103 -0
  205. package/dist/shared/evaluation-job.d.ts +6 -0
  206. package/dist/shared/evaluation-job.js +217 -0
  207. package/dist/shared/executor-result.d.ts +17 -0
  208. package/dist/shared/executor-result.js +221 -0
  209. package/dist/shared/file-lock.d.ts +12 -0
  210. package/dist/shared/file-lock.js +129 -0
  211. package/dist/shared/json-value.d.ts +5 -0
  212. package/dist/shared/json-value.js +36 -0
  213. package/dist/shared/keyed-mutex.d.ts +7 -0
  214. package/dist/shared/keyed-mutex.js +24 -0
  215. package/dist/shared/record-count.d.ts +8 -0
  216. package/dist/shared/record-count.js +43 -0
  217. package/dist/shared/sample-contract.d.ts +3 -0
  218. package/dist/shared/sample-contract.js +332 -0
  219. package/dist/shared/timestamp.d.ts +6 -0
  220. package/dist/shared/timestamp.js +64 -0
  221. package/dist/shared/token-usage.d.ts +19 -0
  222. package/dist/shared/token-usage.js +50 -0
  223. package/dist/shared/tool-call-status.d.ts +8 -0
  224. package/dist/shared/tool-call-status.js +28 -0
  225. package/dist/shared/tool-identity.d.ts +21 -0
  226. package/dist/shared/tool-identity.js +84 -0
  227. package/dist/shared/tool-search.js +73 -16
  228. package/dist/shared/trace-projection.d.ts +5 -0
  229. package/dist/shared/trace-projection.js +20 -0
  230. package/dist/shared/trace-source-kind.d.ts +3 -0
  231. package/dist/shared/trace-source-kind.js +12 -0
  232. package/dist/types/diagnosis.d.ts +2 -0
  233. package/dist/types/eval.d.ts +4 -0
  234. package/dist/types/executor.d.ts +32 -5
  235. package/dist/types/index.d.ts +1 -0
  236. package/dist/types/index.js +1 -0
  237. package/dist/types/judge.d.ts +2 -0
  238. package/dist/types/observability.d.ts +116 -9
  239. package/dist/types/report.d.ts +58 -6
  240. package/dist/types/skill-index.d.ts +7 -0
  241. package/dist/types/trace.d.ts +2 -0
  242. package/dist/types/trace.js +1 -0
  243. package/package.json +9 -5
@@ -9,10 +9,15 @@
9
9
  * 三域共用 `artifactIndexDir(domain)` + tmp-rename 原子写 + 指纹缓存读 的同一套机制,各域只差「投影成卡片」
10
10
  * 与「卡片还原成 snapshot」两段域特定逻辑。
11
11
  */
12
- import { existsSync, mkdirSync, readdirSync, readFileSync, writeFileSync, renameSync, unlinkSync, statSync } from 'node:fs';
13
- import { join, resolve } from 'node:path';
12
+ import { existsSync, readdirSync, readFileSync, unlinkSync, statSync } from 'node:fs';
13
+ import { dirname, join, resolve } from 'node:path';
14
14
  import { DEFAULT_ARTIFACT_INDEX_DIR } from './default-dirs.js';
15
15
  import { globalReportsDir, globalDoctorsDir, globalObserveHealthDir } from './measurement-dirs.js';
16
+ import { setOwnRecordValue, sumRecordCounts } from '../shared/record-count.js';
17
+ import { writeJsonFileAtomic } from '../shared/atomic-json.js';
18
+ import { isRfc3339Timestamp } from '../shared/timestamp.js';
19
+ import { reportFilePath, safeArtifactFileStem } from './artifact-file-names.js';
20
+ import { parseReportDocument, parseReportIndexCard } from './report-document.js';
16
21
  // ── 通用机制(域无关)──────────────────────────────────────────────────────────
17
22
  /** 索引根:`OMK_ARTIFACT_INDEX_DIR` 覆盖(测试隔离,仿 OMK_TREES_DIR),默认 state/artifact-index。 */
18
23
  function artifactIndexRoot() {
@@ -32,22 +37,29 @@ export function shouldIndexReport(outputDir) {
32
37
  return shouldIndexDir(outputDir, globalReportsDir());
33
38
  }
34
39
  function safeFileName(id) {
35
- return id.replaceAll(/[/\\:*?"<>|]/g, '_');
40
+ return safeArtifactFileStem(id);
41
+ }
42
+ function isCanonicalCardId(value) {
43
+ return typeof value === 'string' && value.length > 0 && safeFileName(value) === value;
44
+ }
45
+ function isCanonicalCardPath(path, id) {
46
+ return typeof path === 'string'
47
+ && resolve(path) === path
48
+ && reportFilePath(dirname(path), id) === path;
36
49
  }
37
50
  // 卡片读侧小 guard:索引是可重生 scratch,坏卡片(脏文件 / 别域误落 / 字段缺失)读侧从严跳过,
38
51
  // 不让 undefined / 非法枚举 / NaN 当可信输入污染 studio。
39
52
  const isFiniteNumber = (v) => typeof v === 'number' && Number.isFinite(v);
53
+ const isNonNegativeInteger = (v) => Number.isSafeInteger(v) && v >= 0;
54
+ const isRate = (v) => isFiniteNumber(v) && v >= 0 && v <= 1;
40
55
  const isDoctorStatus = (v) => v === 'pass' || v === 'warn' || v === 'fail';
41
56
  const isHealthBand = (v) => v === 'green' || v === 'yellow' || v === 'red';
42
57
  const isConfidence = (v) => v === undefined || v === 'high' || v === 'low' || v === 'underpowered';
43
58
  /** 原子写一张卡片(tmp+rename,防半截 JSON 被 reader 读到)。 */
44
59
  function writeCard(domain, id, card) {
45
- const indexDir = artifactIndexDir(domain);
46
- mkdirSync(indexDir, { recursive: true });
47
- const target = join(indexDir, `${safeFileName(id)}.json`);
48
- const tmp = `${target}.tmp.${process.pid}.${Math.random().toString(36).slice(2)}`;
49
- writeFileSync(tmp, JSON.stringify(card, null, 2));
50
- renameSync(tmp, target);
60
+ if (!isCanonicalCardId(id))
61
+ throw new Error('invalid artifact index id');
62
+ writeJsonFileAtomic(join(artifactIndexDir(domain), `${id}.json`), card);
51
63
  }
52
64
  /** 读某域全部卡片,逐条用 `valid` 谓词过滤坏文件 / 缺字段(索引是 scratch,读侧从严)。 */
53
65
  function readArtifactCards(domain, valid) {
@@ -62,12 +74,14 @@ function readArtifactCards(domain, valid) {
62
74
  return [];
63
75
  }
64
76
  const out = [];
65
- for (const f of files) {
77
+ for (const f of files.sort()) {
66
78
  if (!f.endsWith('.json') || f.includes('.json.tmp.'))
67
79
  continue;
68
80
  try {
69
81
  const c = JSON.parse(readFileSync(join(dir, f), 'utf-8'));
70
- if (valid(c))
82
+ if (valid(c)
83
+ && isCanonicalCardId(c.id)
84
+ && f === `${c.id}.json`)
71
85
  out.push(c);
72
86
  }
73
87
  catch { /* skip corrupt card */ }
@@ -76,8 +90,10 @@ function readArtifactCards(domain, valid) {
76
90
  }
77
91
  /** 删某域某 id 的卡片。幂等、best-effort。返回卡片是否曾存在。 */
78
92
  function removeArtifactCard(domain, id) {
93
+ if (!isCanonicalCardId(id))
94
+ return false;
79
95
  try {
80
- const p = join(artifactIndexDir(domain), `${safeFileName(id)}.json`);
96
+ const p = join(artifactIndexDir(domain), `${id}.json`);
81
97
  if (!existsSync(p))
82
98
  return false;
83
99
  unlinkSync(p);
@@ -103,33 +119,20 @@ function liveCards(cards) {
103
119
  * 缓存失效 → live 过滤生效。读卡片只为拿 path,字段宽松。
104
120
  */
105
121
  export function cardTargetSentinel(domain) {
106
- const dir = artifactIndexDir(domain);
107
- if (!existsSync(dir))
108
- return '';
109
- let files;
110
- try {
111
- files = readdirSync(dir);
112
- }
113
- catch {
114
- return '';
115
- }
122
+ const cards = domain === 'report'
123
+ ? listReportCards()
124
+ : domain === 'doctor'
125
+ ? listDoctorCards()
126
+ : listObserveCards();
116
127
  const parts = [];
117
- for (const f of files) {
118
- if (!f.endsWith('.json') || f.includes('.json.tmp.'))
119
- continue;
128
+ for (const card of cards) {
120
129
  try {
121
- const c = JSON.parse(readFileSync(join(dir, f), 'utf-8'));
122
- if (typeof c.path !== 'string')
123
- continue;
124
- try {
125
- const s = statSync(c.path);
126
- parts.push(`${f}=${s.mtimeMs}:${s.size}`);
127
- }
128
- catch {
129
- parts.push(`${f}=gone`);
130
- }
130
+ const stat = statSync(card.path);
131
+ parts.push(`${card.id}.json=${stat.mtimeMs}:${stat.size}`);
132
+ }
133
+ catch {
134
+ parts.push(`${card.id}.json=gone`);
131
135
  }
132
- catch { /* skip corrupt */ }
133
136
  }
134
137
  return parts.sort().join(',');
135
138
  }
@@ -137,8 +140,8 @@ export function cardTargetSentinel(domain) {
137
140
  /** 报告投影成卡片(剥掉 results 重体)。 */
138
141
  function reportCard(report, sourcePath) {
139
142
  return report.kind === 'evaluation'
140
- ? { domain: 'report', id: report.id, path: sourcePath, kind: 'evaluation', meta: report.meta, summary: report.summary }
141
- : { domain: 'report', id: report.id, path: sourcePath, kind: 'batch-evaluation', meta: report.meta, items: report.items };
143
+ ? { domain: 'report', id: report.id, path: resolve(sourcePath), kind: 'evaluation', meta: report.meta, summary: report.summary }
144
+ : { domain: 'report', id: report.id, path: resolve(sourcePath), kind: 'batch-evaluation', meta: report.meta, items: report.items };
142
145
  }
143
146
  /**
144
147
  * persistReport 的索引钩子:报告落盘后 best-effort 追加卡片。永不抛、永不阻断报告落盘。
@@ -148,10 +151,10 @@ export function indexReportWrite(report, sourcePath, outputDir) {
148
151
  try {
149
152
  if (!shouldIndexReport(outputDir))
150
153
  return;
151
- const doc = report;
152
- if (doc.kind !== 'evaluation' && doc.kind !== 'batch-evaluation')
154
+ const parsed = parseReportDocument(report, report.id, report.id);
155
+ if (!parsed)
153
156
  return;
154
- writeCard('report', report.id, reportCard(doc, sourcePath));
157
+ writeCard('report', report.id, reportCard(parsed, sourcePath));
155
158
  }
156
159
  catch {
157
160
  // 索引可重建,失败静默(可选 stderr warn);正文已落盘不受影响。
@@ -159,24 +162,12 @@ export function indexReportWrite(report, sourcePath, outputDir) {
159
162
  }
160
163
  /** 读 report 域全部卡片(跳过坏文件 / 缺字段 / 坏 kind)。 */
161
164
  export function listReportCards() {
162
- return readArtifactCards('report', (c) => {
163
- const card = c;
164
- // kind 白名单:cardToReportDocument 对任何非 evaluation 一律按 batch 投影,坏 kind(拼错 / 'doctor')
165
- // 会污染机器级 list 成空 batch 报告。索引是可重生 scratch,读侧从严跳过坏卡片。
166
- const kindOk = card?.kind === 'evaluation' || card?.kind === 'batch-evaluation';
167
- return !!card && card.domain === 'report' && kindOk && typeof card.id === 'string' && typeof card.path === 'string' && !!card.meta;
168
- });
165
+ return readArtifactCards('report', (value) => parseReportIndexCard(value) !== null);
169
166
  }
170
167
  /** report 卡片(过滤悬空真身):供 studio 机器级 list / findBy 展示。 */
171
168
  export function listLiveReportCards() {
172
169
  return liveCards(listReportCards());
173
170
  }
174
- /** 卡片 → ReportDocument(results:[]):供 studio list / buildSkillIndex / trend 消费(它们不读 results)。 */
175
- export function cardToReportDocument(card) {
176
- return card.kind === 'evaluation'
177
- ? { kind: 'evaluation', id: card.id, meta: card.meta, summary: card.summary ?? {}, results: [] }
178
- : { kind: 'batch-evaluation', id: card.id, mode: 'skill', meta: card.meta, items: card.items ?? [] };
179
- }
180
171
  /** 删 report 域某 id 的卡片(DELETE 报告时连卡片一起删,使其从机器级 list 消失)。幂等、best-effort。 */
181
172
  export function removeReportCard(id) {
182
173
  return removeArtifactCard('report', id);
@@ -186,7 +177,11 @@ export function indexDoctorWrite(card, outputDir) {
186
177
  try {
187
178
  if (!shouldIndexDir(outputDir, globalDoctorsDir()))
188
179
  return;
189
- writeCard('doctor', card.id, { domain: 'doctor', ...card });
180
+ writeCard('doctor', card.id, {
181
+ domain: 'doctor',
182
+ ...card,
183
+ path: resolve(card.path),
184
+ });
190
185
  }
191
186
  catch { /* 索引可重建,失败静默 */ }
192
187
  }
@@ -194,26 +189,30 @@ export function indexDoctorWrite(card, outputDir) {
194
189
  export function listDoctorCards() {
195
190
  return readArtifactCards('doctor', (c) => {
196
191
  const card = c;
197
- return !!card && card.domain === 'doctor' && typeof card.id === 'string' && typeof card.path === 'string'
198
- && typeof card.skillName === 'string' && typeof card.reportId === 'string' && typeof card.timestamp === 'string'
192
+ return !!card && card.domain === 'doctor' && isCanonicalCardId(card.id)
193
+ && isCanonicalCardPath(card.path, card.id)
194
+ && typeof card.skillName === 'string' && card.skillName.length > 0
195
+ && typeof card.reportId === 'string' && card.reportId.length > 0
196
+ && isRfc3339Timestamp(card.timestamp)
199
197
  && isDoctorStatus(card.status)
200
- && isFiniteNumber(card.passCount) && isFiniteNumber(card.warnCount) && isFiniteNumber(card.failCount);
198
+ && isNonNegativeInteger(card.passCount)
199
+ && isNonNegativeInteger(card.warnCount)
200
+ && isNonNegativeInteger(card.failCount)
201
+ && (() => {
202
+ try {
203
+ sumRecordCounts(card.passCount, card.warnCount, card.failCount);
204
+ return true;
205
+ }
206
+ catch {
207
+ return false;
208
+ }
209
+ })();
201
210
  });
202
211
  }
203
212
  /** doctor 卡片(过滤悬空真身):供 buildSkillIndex 机器级合并。 */
204
213
  export function listLiveDoctorCards() {
205
214
  return liveCards(listDoctorCards());
206
215
  }
207
- /** doctor 卡片 → (skillName, SkillDoctorSnapshot)。results:[](逐规则详情需回源项目看)。 */
208
- export function cardToDoctorSnapshot(card) {
209
- return {
210
- skillName: card.skillName,
211
- snap: {
212
- reportId: card.reportId, timestamp: card.timestamp, status: card.status,
213
- passCount: card.passCount, warnCount: card.warnCount, failCount: card.failCount, results: [],
214
- },
215
- };
216
- }
217
216
  /** 删 doctor 域某 id(文件 stem)的卡片。doctor 历史按 50/skill prune,删正文时必须连卡片一起删,
218
217
  * 否则被 prune 掉的报告会经卡片合并在本项目 studio「复活」。 */
219
218
  export function removeDoctorCard(id) {
@@ -226,13 +225,21 @@ export function indexObserveWrite(report, sourcePath, outputDir, id) {
226
225
  return;
227
226
  const bySkill = {};
228
227
  for (const [name, h] of Object.entries(report.bySkill || {})) {
229
- bySkill[name] = {
230
- toolFailureRate: h.toolFailureRate, segmentCount: h.segmentCount,
231
- gap: { weightedGapRate: h.gap?.weightedGapRate ?? 0 }, confidence: h.confidence,
232
- };
228
+ setOwnRecordValue(bySkill, name, {
229
+ toolFailureRate: h.toolFailureRate,
230
+ toolFailureCount: h.toolFailureCount,
231
+ toolCallCount: h.toolCallCount,
232
+ toolResolvedCount: h.toolResolvedCount,
233
+ toolCancelledCount: h.toolCancelledCount,
234
+ toolUnknownCount: h.toolUnknownCount,
235
+ segmentCount: h.segmentCount,
236
+ gap: { weightedGapRate: h.gap?.weightedGapRate ?? 0 },
237
+ confidence: h.confidence,
238
+ stability: h.stability,
239
+ });
233
240
  }
234
241
  const card = {
235
- domain: 'observe-health', id, path: sourcePath,
242
+ domain: 'observe-health', id, path: resolve(sourcePath),
236
243
  meta: { generatedAt: report.meta.generatedAt, sessionCount: report.meta.sessionCount, segmentCount: report.meta.segmentCount },
237
244
  overall: { healthBand: report.overall.healthBand, confidence: report.overall.confidence },
238
245
  bySkill,
@@ -245,26 +252,78 @@ export function indexObserveWrite(report, sourcePath, outputDir, id) {
245
252
  export function listObserveCards() {
246
253
  return readArtifactCards('observe-health', (c) => {
247
254
  const card = c;
248
- if (!card || card.domain !== 'observe-health' || typeof card.id !== 'string' || typeof card.path !== 'string')
255
+ if (!card
256
+ || card.domain !== 'observe-health'
257
+ || !isCanonicalCardId(card.id)
258
+ || !isCanonicalCardPath(card.path, card.id))
249
259
  return false;
250
260
  const meta = card.meta;
251
261
  const overall = card.overall;
252
262
  const bySkill = card.bySkill;
253
- if (!meta || !isFiniteNumber(meta.sessionCount) || !isFiniteNumber(meta.segmentCount) || typeof meta.generatedAt !== 'string')
263
+ if (!meta
264
+ || !isNonNegativeInteger(meta.sessionCount)
265
+ || !isNonNegativeInteger(meta.segmentCount)
266
+ || !isRfc3339Timestamp(meta.generatedAt))
254
267
  return false;
255
268
  if (!overall || !isHealthBand(overall.healthBand) || !isConfidence(overall.confidence))
256
269
  return false;
257
- if (!bySkill || typeof bySkill !== 'object')
270
+ if (!bySkill || typeof bySkill !== 'object' || Array.isArray(bySkill))
258
271
  return false;
259
272
  // 每个 skill 的标量也校验:坏 toolFailureRate / segmentCount / gap.weightedGapRate 会让 bandFromObserveHealth /
260
273
  // 趋势 / 健康快照出 NaN(buildSkillIndex 直接取 h.gap?.weightedGapRate ?? 0 进 gapRate、参与阈值判断)。
261
- for (const h of Object.values(bySkill)) {
262
- if (!h || !isFiniteNumber(h.toolFailureRate) || !isFiniteNumber(h.segmentCount) || !isConfidence(h.confidence))
274
+ const segmentCounts = [];
275
+ for (const [skillName, h] of Object.entries(bySkill)) {
276
+ if (skillName.length === 0)
277
+ return false;
278
+ if (!h || !isRate(h.toolFailureRate) || !isNonNegativeInteger(h.segmentCount) || !isConfidence(h.confidence))
279
+ return false;
280
+ segmentCounts.push(h.segmentCount);
281
+ if (h.toolFailureCount !== undefined && !isNonNegativeInteger(h.toolFailureCount))
282
+ return false;
283
+ if (h.toolCallCount !== undefined && !isNonNegativeInteger(h.toolCallCount))
284
+ return false;
285
+ if (h.toolResolvedCount !== undefined && !isNonNegativeInteger(h.toolResolvedCount))
286
+ return false;
287
+ if (h.toolCancelledCount !== undefined && !isNonNegativeInteger(h.toolCancelledCount))
288
+ return false;
289
+ if (h.toolUnknownCount !== undefined && !isNonNegativeInteger(h.toolUnknownCount))
263
290
  return false;
264
- if (h.gap !== undefined && h.gap.weightedGapRate !== undefined && !isFiniteNumber(h.gap.weightedGapRate))
291
+ if (h.toolCallCount === undefined
292
+ && (h.toolResolvedCount !== undefined
293
+ || h.toolCancelledCount !== undefined
294
+ || h.toolUnknownCount !== undefined))
295
+ return false;
296
+ if (h.toolCallCount !== undefined
297
+ && ((h.toolResolvedCount ?? 0) > h.toolCallCount
298
+ || (h.toolCancelledCount ?? 0) > (h.toolResolvedCount ?? h.toolCallCount)
299
+ || (h.toolUnknownCount ?? 0) > h.toolCallCount
300
+ || (h.toolResolvedCount !== undefined
301
+ && h.toolUnknownCount !== undefined
302
+ && h.toolResolvedCount + h.toolUnknownCount !== h.toolCallCount)))
303
+ return false;
304
+ if (h.toolFailureCount !== undefined
305
+ && h.toolCallCount !== undefined) {
306
+ const resolved = h.toolResolvedCount ?? h.toolCallCount;
307
+ const comparable = resolved - (h.toolCancelledCount ?? 0);
308
+ const expectedFailureRate = comparable > 0
309
+ ? Number((h.toolFailureCount / comparable).toFixed(4))
310
+ : 0;
311
+ if (h.toolFailureCount > comparable
312
+ || Math.abs(h.toolFailureRate - expectedFailureRate) > 0.0001)
313
+ return false;
314
+ }
315
+ if (h.stability !== undefined
316
+ && !['stable', 'unstable', 'very-unstable', 'unknown'].includes(h.stability))
317
+ return false;
318
+ if (h.gap !== undefined && h.gap.weightedGapRate !== undefined && !isRate(h.gap.weightedGapRate))
265
319
  return false;
266
320
  }
267
- return true;
321
+ try {
322
+ return sumRecordCounts(...segmentCounts) === meta.segmentCount;
323
+ }
324
+ catch {
325
+ return false;
326
+ }
268
327
  });
269
328
  }
270
329
  /** observe 卡片(过滤悬空真身):供 listAnalyses / buildSkillIndex 机器级合并。 */
@@ -2,8 +2,9 @@
2
2
  * Executor result cache.
3
3
  *
4
4
  * Caches successful executor results to disk to avoid redundant API calls.
5
- * Cache key v6 = sha256(model + system + prompt + cwd + allowedSkills + executor +
6
- * runtime + mocks + mocksStrict + effort + artifactContentHash).
5
+ * Cache key v9 = sha256(model + system + prompt + cwd + allowedSkills + executor +
6
+ * runtime + mocks + mocksStrict + effort + artifactContentHash +
7
+ * sampleExecutionDependencyHash).
7
8
  * Loaded into memory on init, flushed to disk on save().
8
9
  *
9
10
  * Prefix bumps intentionally invalidate old entries when construct-validity
@@ -20,6 +21,12 @@
20
21
  * 不动 system → 旧 key 会命中旧输出、贴到新 artifactHashes 上,形成静默测量污染。把
21
22
  * contentHash 纳入 key,资产变即重跑。git skill 的 contentHash 只随 SKILL.md 变(其资产不
22
23
  * 暴露给 executor、本就不该触发重跑),口径自洽
24
+ * - v7: executor 返回值进入统一契约校验,且所有未知成本路径显式记录 provenance。
25
+ * 旧缓存缺少这些语义,不能在新报告中继续复用。
26
+ * - v8: 所有 executor 的工具身份在统一边界归一化。旧缓存里的 provider-native
27
+ * 工具名不能继续参与 source-neutral 工具断言与分布统计。
28
+ * - v9: sample mock 的 `return_file` 内容指纹进入 key。只哈声明路径会在 fixture
29
+ * 内容变化后错误复用旧执行结果。
23
30
  */
24
31
  import type { ExecutorCache } from '../types/index.js';
25
32
  export declare function createCache(cacheDir: string): ExecutorCache;
@@ -35,4 +42,6 @@ mocksStrict?: boolean,
35
42
  effort?: string,
36
43
  /** artifact 内容指纹(整树 / 单文件哈)。本地 dir-skill 改 references/ 资产只动此值、不动 system,
37
44
  * 不进 key 会让改资产后命中旧输出 → 静默污染。空(baseline / 无 skill)等价无指纹。 */
38
- artifactContentHash?: string): string;
45
+ artifactContentHash?: string,
46
+ /** External sample files that can alter executor output, currently mock return_file fixtures. */
47
+ sampleExecutionDependencyHash?: string): string;
@@ -2,8 +2,9 @@
2
2
  * Executor result cache.
3
3
  *
4
4
  * Caches successful executor results to disk to avoid redundant API calls.
5
- * Cache key v6 = sha256(model + system + prompt + cwd + allowedSkills + executor +
6
- * runtime + mocks + mocksStrict + effort + artifactContentHash).
5
+ * Cache key v9 = sha256(model + system + prompt + cwd + allowedSkills + executor +
6
+ * runtime + mocks + mocksStrict + effort + artifactContentHash +
7
+ * sampleExecutionDependencyHash).
7
8
  * Loaded into memory on init, flushed to disk on save().
8
9
  *
9
10
  * Prefix bumps intentionally invalidate old entries when construct-validity
@@ -20,10 +21,19 @@
20
21
  * 不动 system → 旧 key 会命中旧输出、贴到新 artifactHashes 上,形成静默测量污染。把
21
22
  * contentHash 纳入 key,资产变即重跑。git skill 的 contentHash 只随 SKILL.md 变(其资产不
22
23
  * 暴露给 executor、本就不该触发重跑),口径自洽
24
+ * - v7: executor 返回值进入统一契约校验,且所有未知成本路径显式记录 provenance。
25
+ * 旧缓存缺少这些语义,不能在新报告中继续复用。
26
+ * - v8: 所有 executor 的工具身份在统一边界归一化。旧缓存里的 provider-native
27
+ * 工具名不能继续参与 source-neutral 工具断言与分布统计。
28
+ * - v9: sample mock 的 `return_file` 内容指纹进入 key。只哈声明路径会在 fixture
29
+ * 内容变化后错误复用旧执行结果。
23
30
  */
24
- import { readFileSync, writeFileSync, mkdirSync, existsSync } from 'node:fs';
31
+ import { readFileSync, mkdirSync, existsSync } from 'node:fs';
25
32
  import { join } from 'node:path';
26
33
  import { createHash } from 'node:crypto';
34
+ import { executorResultValidationError, normalizeExecResultToolIdentities, parseExecResult, } from '../shared/executor-result.js';
35
+ import { writeJsonFileAtomic } from '../shared/atomic-json.js';
36
+ import { withFileLock } from '../shared/file-lock.js';
27
37
  const CACHE_FILE = 'executor-cache.json';
28
38
  /** v5 保留 turns / toolCalls 后单 entry 可达 5–50 KB,长期使用会无界膨胀。
29
39
  * Map iteration 是插入序,set() 时若 key 已存在先 delete 再 set 把它移到末尾 → 实现 LRU。
@@ -46,22 +56,9 @@ export function createCache(cacheDir) {
46
56
  mkdirSync(cacheDir, { recursive: true });
47
57
  const filePath = join(cacheDir, CACHE_FILE);
48
58
  const cap = resolveCacheCap();
49
- // 用 Map 替代 Record:Map 保证 iteration 顺序 = 插入顺序,LRU 淘汰用得上。
50
- // 老 JSON cache 文件仍按 Record<string, ExecResult> 写,反序列化时 Object.entries
51
- // 转 Map(Object 字段顺序在主流引擎里也是插入序,所以 LRU 信号不丢)。
52
- const store = new Map();
53
- if (existsSync(filePath)) {
54
- try {
55
- const raw = JSON.parse(readFileSync(filePath, 'utf-8'));
56
- for (const [k, v] of Object.entries(raw))
57
- store.set(k, v);
58
- // 加载完也 enforce 一次 cap,处理上次 process exit 前没保存到的极端膨胀。
59
- evictUntilWithinCap(store, cap);
60
- }
61
- catch {
62
- // 老 cache 坏了忽略 — 重跑会重建。
63
- }
64
- }
59
+ const store = readCacheStore(filePath, cap);
60
+ const pendingWrites = new Map();
61
+ const touchedKeys = new Set();
65
62
  let dirty = false;
66
63
  return {
67
64
  get(key) {
@@ -71,6 +68,7 @@ export function createCache(cacheDir) {
71
68
  // LRU touch:命中时移到末尾(最新)。这样 evict 时永远从最旧端拿。
72
69
  store.delete(key);
73
70
  store.set(key, v);
71
+ touchedKeys.add(key);
74
72
  dirty = true;
75
73
  return v;
76
74
  },
@@ -78,21 +76,51 @@ export function createCache(cacheDir) {
78
76
  // 保留完整 ExecResult(含 turns / toolCalls):工具类 assertion (tool_called /
79
77
  // tool_input_contains / tools_called)和 diagnostic 要看 trace,砍掉的话 cached
80
78
  // rerun 进 grade() 时工具断言为空、diagnostic 没真实证据,跟 cold run 不一致。
79
+ const validationError = executorResultValidationError(value);
80
+ if (validationError)
81
+ throw new Error(`invalid executor cache entry: ${validationError}`);
82
+ const normalized = normalizeExecResultToolIdentities(value);
81
83
  if (store.has(key))
82
84
  store.delete(key);
83
- store.set(key, { ...value });
85
+ store.set(key, normalized);
86
+ pendingWrites.delete(key);
87
+ pendingWrites.set(key, normalized);
88
+ touchedKeys.delete(key);
84
89
  evictUntilWithinCap(store, cap);
90
+ for (const pendingKey of pendingWrites.keys()) {
91
+ if (!store.has(pendingKey))
92
+ pendingWrites.delete(pendingKey);
93
+ }
85
94
  dirty = true;
86
95
  },
87
96
  save() {
88
97
  if (!dirty)
89
98
  return;
90
- // 序列化时用普通 object,跟老 cache 文件格式兼容(读老报告不破)。
91
- // Map iteration 是插入序,生成的 object 字段顺序也保留 LRU 状态,下次加载继续生效。
92
- const obj = {};
93
- for (const [k, v] of store)
94
- obj[k] = v;
95
- writeFileSync(filePath, JSON.stringify(obj, null, 2));
99
+ withFileLock(`${filePath}.lock`, () => {
100
+ // Another eval process may have saved after this cache instance loaded.
101
+ // Merge only local writes and LRU touches into the latest disk state;
102
+ // replacing it with the whole stale in-memory snapshot loses entries.
103
+ const merged = readCacheStore(filePath, cap);
104
+ for (const key of touchedKeys) {
105
+ const value = merged.get(key);
106
+ if (!value)
107
+ continue;
108
+ merged.delete(key);
109
+ merged.set(key, value);
110
+ }
111
+ for (const [key, value] of pendingWrites) {
112
+ if (merged.has(key))
113
+ merged.delete(key);
114
+ merged.set(key, value);
115
+ }
116
+ evictUntilWithinCap(merged, cap);
117
+ writeCacheStore(filePath, merged);
118
+ store.clear();
119
+ for (const [key, value] of merged)
120
+ store.set(key, value);
121
+ }, { label: 'executor cache' });
122
+ pendingWrites.clear();
123
+ touchedKeys.clear();
96
124
  dirty = false;
97
125
  },
98
126
  size() {
@@ -100,6 +128,36 @@ export function createCache(cacheDir) {
100
128
  },
101
129
  };
102
130
  }
131
+ function readCacheStore(filePath, cap) {
132
+ // Map guarantees iteration order, which is the persisted LRU order.
133
+ const store = new Map();
134
+ if (!existsSync(filePath))
135
+ return store;
136
+ try {
137
+ const raw = JSON.parse(readFileSync(filePath, 'utf-8'));
138
+ if (!raw || typeof raw !== 'object' || Array.isArray(raw)) {
139
+ throw new Error('invalid executor cache root');
140
+ }
141
+ for (const [key, value] of Object.entries(raw)) {
142
+ const parsed = parseExecResult(value);
143
+ if (parsed)
144
+ store.set(key, parsed);
145
+ }
146
+ evictUntilWithinCap(store, cap);
147
+ }
148
+ catch {
149
+ // A corrupt cache is disposable evidence acceleration, never report data.
150
+ store.clear();
151
+ }
152
+ return store;
153
+ }
154
+ function writeCacheStore(filePath, store) {
155
+ // Keep the historical JSON object format. Property order carries LRU state.
156
+ const serialized = {};
157
+ for (const [key, value] of store)
158
+ serialized[key] = value;
159
+ writeJsonFileAtomic(filePath, serialized);
160
+ }
103
161
  function evictUntilWithinCap(store, cap) {
104
162
  if (!Number.isFinite(cap))
105
163
  return;
@@ -122,7 +180,9 @@ mocksStrict,
122
180
  effort,
123
181
  /** artifact 内容指纹(整树 / 单文件哈)。本地 dir-skill 改 references/ 资产只动此值、不动 system,
124
182
  * 不进 key 会让改资产后命中旧输出 → 静默污染。空(baseline / 无 skill)等价无指纹。 */
125
- artifactContentHash) {
183
+ artifactContentHash,
184
+ /** External sample files that can alter executor output, currently mock return_file fixtures. */
185
+ sampleExecutionDependencyHash) {
126
186
  // allowedSkills 序列化:undefined → "" / [] → "[]" / [...] → 排序后 JSON。
127
187
  // 排序保证 ["a","b"] 和 ["b","a"] 命中同一缓存(语义等价)。
128
188
  const isoStr = allowedSkills === undefined
@@ -138,8 +198,8 @@ artifactContentHash) {
138
198
  // executor + runtime + effort 进 cache key:同 model 名走不同 executor 或同 executor
139
199
  // 换 binary/SDK 版本时输出可能不同,旧 cache 不可复用。
140
200
  const hash = createHash('sha256')
141
- .update(`${model || ''}\n${system || ''}\n${prompt || ''}\n${cwd || ''}\n${isoStr}\n${executor || ''}\n${runtimeFingerprint || ''}\n${mockStr}\n${strictStr}\n${effortStr}\n${artifactContentHash || ''}`)
201
+ .update(`${model || ''}\n${system || ''}\n${prompt || ''}\n${cwd || ''}\n${isoStr}\n${executor || ''}\n${runtimeFingerprint || ''}\n${mockStr}\n${strictStr}\n${effortStr}\n${artifactContentHash || ''}\n${sampleExecutionDependencyHash || ''}`)
142
202
  .digest('hex')
143
203
  .slice(0, 16);
144
- return `v6:${hash}`;
204
+ return `v9:${hash}`;
145
205
  }
@@ -1,3 +1,4 @@
1
+ import { ownRecordValue } from '../shared/record-count.js';
1
2
  function stableStringify(value) {
2
3
  if (value === null || typeof value !== 'object')
3
4
  return JSON.stringify(value);
@@ -88,7 +89,7 @@ function runtimeMapKeys(meta) {
88
89
  function reportExecutorRuntimeWarnings(report, warnings) {
89
90
  const runtimes = report.meta.executorRuntimes;
90
91
  if (runtimes && Object.keys(runtimes).length > 0) {
91
- const missing = report.meta.variants.filter((variant) => !runtimes[variant]);
92
+ const missing = report.meta.variants.filter((variant) => !ownRecordValue(runtimes, variant));
92
93
  if (missing.length > 0) {
93
94
  push(warnings, 'executor_runtime_missing', `报告缺少部分 variant 的 executor runtime 指纹: ${missing.join(', ')}。`, `Report is missing executor runtime fingerprints for variants: ${missing.join(', ')}.`);
94
95
  }
@@ -116,12 +117,14 @@ function countSampleHashMismatches(a, b) {
116
117
  let common = 0;
117
118
  const ids = new Set([...Object.keys(ah), ...Object.keys(bh)]);
118
119
  for (const id of ids) {
119
- if (ah[id] == null || bh[id] == null) {
120
+ const aHash = ownRecordValue(ah, id);
121
+ const bHash = ownRecordValue(bh, id);
122
+ if (aHash == null || bHash == null) {
120
123
  missing++;
121
124
  continue;
122
125
  }
123
126
  common++;
124
- if (ah[id] !== bh[id])
127
+ if (aHash !== bHash)
125
128
  mismatched++;
126
129
  }
127
130
  return { mismatched, missing, common };
@@ -175,13 +178,14 @@ export function crossReportComparabilityWarnings(before, after) {
175
178
  const hasExecutorRuntimeMap = bExecutorRuntimeKeys.length > 0 || aExecutorRuntimeKeys.length > 0;
176
179
  if (hasExecutorRuntimeMap) {
177
180
  const keys = sorted([...new Set([...b.variants, ...a.variants, ...bExecutorRuntimeKeys, ...aExecutorRuntimeKeys])]);
178
- const missing = keys.filter((key) => !b.executorRuntimes?.[key] || !a.executorRuntimes?.[key]);
181
+ const missing = keys.filter((key) => !ownRecordValue(b.executorRuntimes ?? {}, key)
182
+ || !ownRecordValue(a.executorRuntimes ?? {}, key));
179
183
  if (missing.length > 0) {
180
184
  push(warnings, 'executor_runtime_missing', `至少一份报告缺少 per-variant executor runtime 指纹: ${missing.join(', ')}。`, `At least one report is missing per-variant executor runtime fingerprints: ${missing.join(', ')}.`);
181
185
  }
182
186
  for (const key of keys) {
183
- const beforeRuntime = b.executorRuntimes?.[key];
184
- const afterRuntime = a.executorRuntimes?.[key];
187
+ const beforeRuntime = ownRecordValue(b.executorRuntimes ?? {}, key);
188
+ const afterRuntime = ownRecordValue(a.executorRuntimes ?? {}, key);
185
189
  if (beforeRuntime?.fingerprint && afterRuntime?.fingerprint && beforeRuntime.fingerprint !== afterRuntime.fingerprint) {
186
190
  push(warnings, 'executor_runtime_mismatch', `variant ${key} executor runtime 指纹不同: ${runtimeLabel(beforeRuntime)} → ${runtimeLabel(afterRuntime)}。`, `Variant ${key} executor runtime fingerprint changed: ${runtimeLabel(beforeRuntime)} → ${runtimeLabel(afterRuntime)}.`);
187
191
  }
@@ -49,7 +49,8 @@ export declare function executeTasks({ tasks, executor, executorName, model, noJ
49
49
  skipped: number;
50
50
  budgetExhausted: boolean;
51
51
  }>;
52
- export declare function preflight(executor: ExecutorFn, model: string, timeoutMs?: number): Promise<void>;
52
+ export declare function preflightRuntimeLabel(executorName: string, model: string): string;
53
+ export declare function preflight(executor: ExecutorFn, model: string, timeoutMs?: number, label?: string): Promise<void>;
53
54
  /**
54
55
  * Preflight every unique `(executor, model)` judge in `judgeModels`. Used by the
55
56
  * eval pipeline to fail fast when any ensemble member is misconfigured (404 model,