@citeark/agent 0.3.20

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (347) hide show
  1. package/LICENSE +202 -0
  2. package/README.md +128 -0
  3. package/data/dataset-source-registry.v1.json +300 -0
  4. package/dist/arkgraph/boot.js +6 -0
  5. package/dist/arkgraph/index.html +1 -0
  6. package/dist/arkgraph/viewer.css +1 -0
  7. package/dist/arkgraph/viewer.en.css +1 -0
  8. package/dist/arkgraph/viewer.en.js +49 -0
  9. package/dist/arkgraph/viewer.en.js.LEGAL.txt +56 -0
  10. package/dist/arkgraph/viewer.js +49 -0
  11. package/dist/arkgraph/viewer.js.LEGAL.txt +56 -0
  12. package/docker/claude-code/Dockerfile +97 -0
  13. package/docker/claude-code/codex-pro-relay.mjs +466 -0
  14. package/docker/claude-code/runtime-contract-check.mjs +79 -0
  15. package/docs/arkgraph-reading.md +79 -0
  16. package/docs/configuration.md +100 -0
  17. package/docs/integration.md +92 -0
  18. package/docs/maturity-plan.md +27 -0
  19. package/docs/npm-release.md +44 -0
  20. package/docs/paper-reading.md +40 -0
  21. package/docs/research-plan-granularity.md +27 -0
  22. package/docs/terminal.md +49 -0
  23. package/examples/toy-evaluation/compile-task.json +27 -0
  24. package/examples/toy-evaluation/paper.md +5 -0
  25. package/examples/toy-evaluation/repository/README.md +9 -0
  26. package/examples/toy-evaluation/repository/checkpoint.json +4 -0
  27. package/examples/toy-evaluation/repository/evaluate.py +17 -0
  28. package/examples/toy-evaluation/task.json +81 -0
  29. package/package.json +59 -0
  30. package/prompts/compile-research.md +58 -0
  31. package/prompts/execute-contract.md +72 -0
  32. package/prompts/execute-workspace-simple.md +51 -0
  33. package/prompts/execute-workspace.md +34 -0
  34. package/prompts/prepare-reproduction.md +82 -0
  35. package/prompts/repair-research.md +45 -0
  36. package/protocol/CAP.md +129 -0
  37. package/protocol/LICENSE +12 -0
  38. package/protocol/MAPPINGS.md +72 -0
  39. package/protocol/README.md +38 -0
  40. package/protocol/conformance-v2.0-alpha.1.json +36 -0
  41. package/protocol/examples/arkgraph/checkpoint-evaluation.json +309 -0
  42. package/protocol/examples/arkgraph/fixtures.mjs +49 -0
  43. package/protocol/examples/arkgraph/paper-free.json +291 -0
  44. package/protocol/examples/arkgraph/partial-failure.json +344 -0
  45. package/protocol/examples/arkgraph/training-evaluation.json +443 -0
  46. package/protocol/profiles/agent-trace.md +16 -0
  47. package/protocol/profiles/computational-run.md +16 -0
  48. package/protocol/profiles/core.md +15 -0
  49. package/protocol/profiles/public-bundle.md +18 -0
  50. package/protocol/profiles/reproduction.md +29 -0
  51. package/protocol/profiles/research-compilation.md +44 -0
  52. package/protocol/profiles/research-plan.md +39 -0
  53. package/protocol/profiles/restricted-evidence.md +15 -0
  54. package/runtime/bootstrap-autodl-runtime.sh +314 -0
  55. package/runtime/create-runtime-venv.sh +41 -0
  56. package/runtime/install-local-cpu-runtime.sh +23 -0
  57. package/runtime/install-scientific-runtime.sh +153 -0
  58. package/runtime/mineru/parse.py +62 -0
  59. package/runtime/mineru/requirements.txt +4 -0
  60. package/runtime/requirements-baseline.txt +38 -0
  61. package/schemas/cap/v2/activity.schema.json +47 -0
  62. package/schemas/cap/v2/agent.schema.json +32 -0
  63. package/schemas/cap/v2/assertion.schema.json +110 -0
  64. package/schemas/cap/v2/descriptor.schema.json +243 -0
  65. package/schemas/cap/v2/entity.schema.json +64 -0
  66. package/schemas/cap/v2/manifest.schema.json +67 -0
  67. package/schemas/cap/v2/relation.schema.json +82 -0
  68. package/schemas/compute-catalog.schema.json +63 -0
  69. package/schemas/compute-decision.schema.json +27 -0
  70. package/schemas/execution-contract.schema.json +1024 -0
  71. package/schemas/research-card.schema.json +30 -0
  72. package/schemas/research-inventory-draft.schema.json +366 -0
  73. package/schemas/research.schema.json +1044 -0
  74. package/schemas/result.schema.json +173 -0
  75. package/schemas/verification-policy.schema.json +47 -0
  76. package/schemas/verified-conclusion.schema.json +58 -0
  77. package/schemas/workspace-summary.schema.json +24 -0
  78. package/scripts/build-arkgraph-view.mjs +12 -0
  79. package/scripts/check-execution-feasibility.mjs +24 -0
  80. package/scripts/check-syntax.mjs +15 -0
  81. package/scripts/deterministic-asset-preparation.py +438 -0
  82. package/scripts/package-cap.mjs +23 -0
  83. package/scripts/package-local-agent.mjs +23 -0
  84. package/scripts/preview-arkgraph.mjs +25 -0
  85. package/scripts/replay-research-compiler-candidate.mjs +134 -0
  86. package/scripts/review-compiler-sources.mjs +44 -0
  87. package/scripts/run-asset-preparation.sh +17 -0
  88. package/scripts/run-research-plan.mjs +98 -0
  89. package/scripts/validate-asset-preparation.py +290 -0
  90. package/scripts/verify-local-runtime.mjs +57 -0
  91. package/scripts/verify-npm-package.mjs +57 -0
  92. package/src/adapters/paper2agent.mjs +107 -0
  93. package/src/assets/cache.mjs +159 -0
  94. package/src/assets/compute.mjs +98 -0
  95. package/src/assets/executor.mjs +145 -0
  96. package/src/assets/lifecycle.mjs +213 -0
  97. package/src/assets/manifest.mjs +242 -0
  98. package/src/assets/opportunistic-preparation.mjs +81 -0
  99. package/src/assets/plan.mjs +411 -0
  100. package/src/assets/prompts.mjs +29 -0
  101. package/src/assets/public-asset-probe.mjs +525 -0
  102. package/src/assets/qualification.mjs +119 -0
  103. package/src/assets/readiness.mjs +130 -0
  104. package/src/assets/reproduction-admission.mjs +355 -0
  105. package/src/assets/requirements.mjs +152 -0
  106. package/src/assets/source-grounding.mjs +341 -0
  107. package/src/assets/source-policy.mjs +118 -0
  108. package/src/autodl/client.mjs +260 -0
  109. package/src/autodl/ssh.mjs +380 -0
  110. package/src/autodl/tools.mjs +129 -0
  111. package/src/cap/redaction.mjs +38 -0
  112. package/src/cap/v2/archive.mjs +152 -0
  113. package/src/cap/v2/attestation.mjs +204 -0
  114. package/src/cap/v2/canonical-json.mjs +114 -0
  115. package/src/cap/v2/compilation-artifact.mjs +240 -0
  116. package/src/cap/v2/core.mjs +282 -0
  117. package/src/cap/v2/measurement-assessment-records.mjs +23 -0
  118. package/src/cap/v2/pipeline-artifact.mjs +922 -0
  119. package/src/cap/v2/read.mjs +41 -0
  120. package/src/cap/v2/reassessment-artifact.mjs +383 -0
  121. package/src/cap/v2/research-artifact.mjs +231 -0
  122. package/src/cap/v2/research-map-records.mjs +46 -0
  123. package/src/cap/v2/research-object-records.mjs +163 -0
  124. package/src/cap/v2/research-records.mjs +187 -0
  125. package/src/cap/v2/verify.mjs +642 -0
  126. package/src/cli.mjs +1146 -0
  127. package/src/compute/autodl-pro-compiler.mjs +347 -0
  128. package/src/compute/autodl-pro-executor.mjs +459 -0
  129. package/src/compute/autodl-pro-job.mjs +843 -0
  130. package/src/compute/autodl-pro-network.mjs +295 -0
  131. package/src/compute/autodl-pro-remote.mjs +810 -0
  132. package/src/compute/autodl-pro-staging.mjs +117 -0
  133. package/src/compute/campaign.mjs +110 -0
  134. package/src/compute/catalog.mjs +123 -0
  135. package/src/compute/checkpoint-protocol.mjs +154 -0
  136. package/src/compute/codex-account-lock.mjs +111 -0
  137. package/src/compute/codex-account-session.mjs +107 -0
  138. package/src/compute/compiler-profile.mjs +38 -0
  139. package/src/compute/compiler-router.mjs +23 -0
  140. package/src/compute/coordinator-recovery.mjs +210 -0
  141. package/src/compute/executor-router.mjs +29 -0
  142. package/src/compute/gcp-batch-compiler.mjs +685 -0
  143. package/src/compute/gcp-batch-executor.mjs +1215 -0
  144. package/src/compute/gcp-batch-failure.mjs +92 -0
  145. package/src/compute/gcp-batch-job.mjs +527 -0
  146. package/src/compute/gcp-batch-lifecycle.mjs +81 -0
  147. package/src/compute/gcp-checkpoint-worker.mjs +1633 -0
  148. package/src/compute/local-codex-compiler.mjs +52 -0
  149. package/src/compute/measurement-hardware.mjs +128 -0
  150. package/src/compute/remote-attempt.mjs +226 -0
  151. package/src/compute/requirements.mjs +124 -0
  152. package/src/compute/research-phases.mjs +48 -0
  153. package/src/compute/scheduler.mjs +452 -0
  154. package/src/compute/shared-workloads.mjs +26 -0
  155. package/src/compute/stage-archive.mjs +79 -0
  156. package/src/contracts/campaign-contract.mjs +52 -0
  157. package/src/contracts/execution-contract.mjs +819 -0
  158. package/src/contracts/execution-mode.mjs +19 -0
  159. package/src/contracts/execution-timeouts.mjs +45 -0
  160. package/src/contracts/execution-workload.mjs +68 -0
  161. package/src/contracts/preflight-schema.mjs +25 -0
  162. package/src/contracts/public-contract.mjs +63 -0
  163. package/src/contracts/subject-tags.mjs +31 -0
  164. package/src/dashboard/data.mjs +898 -0
  165. package/src/dashboard/server.mjs +79 -0
  166. package/src/dashboard/static/dashboard.css +366 -0
  167. package/src/dashboard/static/dashboard.js +560 -0
  168. package/src/dashboard/static/index.html +85 -0
  169. package/src/deployment/community-policy.mjs +9 -0
  170. package/src/deployment/environment.mjs +112 -0
  171. package/src/deployment/guided.mjs +98 -0
  172. package/src/deployment/handoff.mjs +102 -0
  173. package/src/deployment/local-contract.mjs +31 -0
  174. package/src/deployment/local.mjs +100 -0
  175. package/src/deployment/prepare.mjs +46 -0
  176. package/src/deployment/recipe.mjs +108 -0
  177. package/src/deployment/supplement.mjs +51 -0
  178. package/src/deployment/terminal.mjs +43 -0
  179. package/src/diagnosis/renderer.mjs +75 -0
  180. package/src/diagnosis/target-failure.mjs +46 -0
  181. package/src/evidence/parser-registry.mjs +54 -0
  182. package/src/evidence/parsers/fasttext-classification.mjs +82 -0
  183. package/src/evidence/parsers/json-scalar.mjs +96 -0
  184. package/src/evidence/parsers/simcse-senteval.mjs +104 -0
  185. package/src/evidence/parsers/starspace-classification.mjs +78 -0
  186. package/src/evidence/registry.mjs +147 -0
  187. package/src/execution/runner-audit.mjs +473 -0
  188. package/src/gcp/auth.mjs +106 -0
  189. package/src/gcp/batch-client.mjs +120 -0
  190. package/src/gcp/resource-discovery.mjs +177 -0
  191. package/src/gcp/rest.mjs +82 -0
  192. package/src/gcp/secret-manager.mjs +34 -0
  193. package/src/gcp/signed-url.mjs +133 -0
  194. package/src/gcp/storage.mjs +220 -0
  195. package/src/graph/command.mjs +41 -0
  196. package/src/graph/execution.mjs +97 -0
  197. package/src/graph/model.mjs +37 -0
  198. package/src/graph/presentation.mjs +110 -0
  199. package/src/graph/query.mjs +159 -0
  200. package/src/graph/research-relations.mjs +69 -0
  201. package/src/graph/source-page.mjs +12 -0
  202. package/src/graph/source-preview.mjs +34 -0
  203. package/src/graph/validate.mjs +76 -0
  204. package/src/job.mjs +496 -0
  205. package/src/network/autodl-routing-proxy.mjs +462 -0
  206. package/src/network/egress-proxy.mjs +158 -0
  207. package/src/observability/event-contract.mjs +230 -0
  208. package/src/observability/pipeline-monitor.mjs +166 -0
  209. package/src/pipeline/orchestrator.mjs +1281 -0
  210. package/src/pipeline/recovery-error.mjs +11 -0
  211. package/src/pipeline/replay.mjs +304 -0
  212. package/src/pipeline/shared-execution.mjs +115 -0
  213. package/src/pipeline/stage-checkpoint.mjs +86 -0
  214. package/src/pipeline/stage-recovery.mjs +101 -0
  215. package/src/pipeline/targets.mjs +110 -0
  216. package/src/process.mjs +143 -0
  217. package/src/protocol.mjs +312 -0
  218. package/src/provider/codex-account.mjs +44 -0
  219. package/src/provider/codex-completion.mjs +49 -0
  220. package/src/provider/completion.mjs +292 -0
  221. package/src/provider/model-client.mjs +44 -0
  222. package/src/provider/model-route.mjs +29 -0
  223. package/src/provider/openrouter-readiness.mjs +189 -0
  224. package/src/provider/reader-bridge.mjs +35 -0
  225. package/src/provider/relay.mjs +263 -0
  226. package/src/provider/runtime-auth.mjs +40 -0
  227. package/src/public/cap.d.mts +90 -0
  228. package/src/public/cap.mjs +12 -0
  229. package/src/public/contracts.d.mts +2 -0
  230. package/src/public/host.mjs +171 -0
  231. package/src/public/operations.d.mts +11 -0
  232. package/src/public/presentation.d.mts +4 -0
  233. package/src/records/views.mjs +26 -0
  234. package/src/remote/command.mjs +178 -0
  235. package/src/remote/ssh.mjs +59 -0
  236. package/src/repository-origin.mjs +81 -0
  237. package/src/reproduction/evidence-feedback.mjs +96 -0
  238. package/src/reproduction/incomplete-initialization.mjs +25 -0
  239. package/src/reproduction/lifecycle.mjs +253 -0
  240. package/src/reproduction/plan.mjs +132 -0
  241. package/src/reproduction/prompts.mjs +70 -0
  242. package/src/reproduction/runner.mjs +188 -0
  243. package/src/reproduction/summary.mjs +130 -0
  244. package/src/reproduction/workspace-mode.mjs +7 -0
  245. package/src/research/automatic-admission.mjs +156 -0
  246. package/src/research/compiler-coverage.mjs +85 -0
  247. package/src/research/compiler-failure.mjs +24 -0
  248. package/src/research/compiler-normalization-guards.mjs +112 -0
  249. package/src/research/compiler-repair.mjs +3 -0
  250. package/src/research/compiler.mjs +853 -0
  251. package/src/research/continuation-selection.mjs +26 -0
  252. package/src/research/execution-graph-context.mjs +43 -0
  253. package/src/research/experiment-importance.mjs +15 -0
  254. package/src/research/inventory-handoff.mjs +104 -0
  255. package/src/research/inventory-revisions.mjs +32 -0
  256. package/src/research/mineru-local.mjs +73 -0
  257. package/src/research/paper-command.mjs +19 -0
  258. package/src/research/paper-markdown.mjs +180 -0
  259. package/src/research/paper-source-map.mjs +69 -0
  260. package/src/research/planning-policy.mjs +88 -0
  261. package/src/research/reference-materials.mjs +11 -0
  262. package/src/research/reproduction-scope.mjs +30 -0
  263. package/src/research/research-map.mjs +94 -0
  264. package/src/research/research-objects.mjs +88 -0
  265. package/src/research/source-discovery.mjs +646 -0
  266. package/src/research/source-observations.mjs +75 -0
  267. package/src/research/source-review-cli-mcp.mjs +26 -0
  268. package/src/research/source-review-input.mjs +209 -0
  269. package/src/research/source-review-local-codex.mjs +36 -0
  270. package/src/research/source-review-model.mjs +70 -0
  271. package/src/research/source-review.mjs +173 -0
  272. package/src/research/structure.mjs +3163 -0
  273. package/src/research-card/renderer.mjs +277 -0
  274. package/src/research-card/verified-conclusion.mjs +143 -0
  275. package/src/results/output-registry.mjs +183 -0
  276. package/src/runtime/claude-code.mjs +52 -0
  277. package/src/runtime/codex-capacity-retry.mjs +87 -0
  278. package/src/runtime/codex.mjs +64 -0
  279. package/src/runtime/config.mjs +157 -0
  280. package/src/runtime/final-output.mjs +40 -0
  281. package/src/runtime/index.mjs +21 -0
  282. package/src/runtime/local-codex.mjs +74 -0
  283. package/src/runtime/opencode.mjs +95 -0
  284. package/src/runtime/prompt.mjs +13 -0
  285. package/src/sandbox/docker.mjs +363 -0
  286. package/src/settings/command.mjs +297 -0
  287. package/src/settings/store.mjs +119 -0
  288. package/src/telemetry/pricing.mjs +68 -0
  289. package/src/telemetry/usage.mjs +265 -0
  290. package/src/terminal/events.mjs +97 -0
  291. package/src/terminal/input.mjs +40 -0
  292. package/src/terminal/plain.mjs +40 -0
  293. package/src/terminal/remote-stream.mjs +22 -0
  294. package/src/terminal/screen.mjs +214 -0
  295. package/src/terminal/transcript.mjs +69 -0
  296. package/src/util.mjs +107 -0
  297. package/src/verification/ai-assessor.mjs +534 -0
  298. package/src/verification/claim-evaluator.mjs +242 -0
  299. package/src/verification/evidence-context.mjs +165 -0
  300. package/src/verification/evidence-reader.mjs +95 -0
  301. package/src/verification/integrity.mjs +570 -0
  302. package/src/verification/tolerance.mjs +32 -0
  303. package/src/workloads/cpu-research-preparation.mjs +56 -0
  304. package/src/workloads/definition.mjs +74 -0
  305. package/src/workloads/phase-aware-reproduction.mjs +46 -0
  306. package/src/workloads/reproduction.mjs +85 -0
  307. package/src/workspace/command.mjs +242 -0
  308. package/src/workspace/control.mjs +49 -0
  309. package/src/workspace/entry.mjs +28 -0
  310. package/src/workspace/input.mjs +93 -0
  311. package/src/workspace/interactive.mjs +94 -0
  312. package/src/workspace/jobs.mjs +418 -0
  313. package/src/workspace/session.mjs +97 -0
  314. package/src/workspace/worker.mjs +137 -0
  315. package/ui/arkgraph/ambient-motion.mjs +10 -0
  316. package/ui/arkgraph/app.jsx +153 -0
  317. package/ui/arkgraph/boot.js +6 -0
  318. package/ui/arkgraph/camera-motion.mjs +20 -0
  319. package/ui/arkgraph/context-reveal.mjs +39 -0
  320. package/ui/arkgraph/details.css +3 -0
  321. package/ui/arkgraph/entry.jsx +28 -0
  322. package/ui/arkgraph/experiment-curves.mjs +17 -0
  323. package/ui/arkgraph/experiment-selection.mjs +15 -0
  324. package/ui/arkgraph/experiment-style.css +26 -0
  325. package/ui/arkgraph/experiment-ui.jsx +32 -0
  326. package/ui/arkgraph/frame.html +1 -0
  327. package/ui/arkgraph/graph-gestures.mjs +62 -0
  328. package/ui/arkgraph/label-layout.mjs +57 -0
  329. package/ui/arkgraph/locales/en.json +229 -0
  330. package/ui/arkgraph/locales/source-types.json +15 -0
  331. package/ui/arkgraph/localization-build.mjs +27 -0
  332. package/ui/arkgraph/material-build.mjs +23 -0
  333. package/ui/arkgraph/material-colors.mjs +39 -0
  334. package/ui/arkgraph/material-style.css +15 -0
  335. package/ui/arkgraph/open-graph.jsx +326 -0
  336. package/ui/arkgraph/outline.jsx +49 -0
  337. package/ui/arkgraph/package-lock.json +888 -0
  338. package/ui/arkgraph/package.json +17 -0
  339. package/ui/arkgraph/reading-layout.mjs +130 -0
  340. package/ui/arkgraph/reading-presentation.mjs +73 -0
  341. package/ui/arkgraph/record-detail.css +51 -0
  342. package/ui/arkgraph/record-details.jsx +29 -0
  343. package/ui/arkgraph/research-types.mjs +31 -0
  344. package/ui/arkgraph/selection-mark.jsx +6 -0
  345. package/ui/arkgraph/soft-spine.mjs +26 -0
  346. package/ui/arkgraph/steering-style.css +187 -0
  347. package/ui/arkgraph/style.css +272 -0
@@ -0,0 +1,3163 @@
1
+ import { validateResearchMap } from './research-map.mjs';
2
+ import { validateResearchObjects } from './research-objects.mjs';
3
+ import { normalizeExperimentImportance } from "./experiment-importance.mjs";
4
+ import { REPRODUCTION_ROUTES, reproductionScopeIssues } from "./reproduction-scope.mjs";
5
+ import { revisedInventoryClaims, sourceClaim } from "./inventory-revisions.mjs";
6
+ import { measurementRequiresHardwareProtocol, measurementHardwareIssues } from "../compute/measurement-hardware.mjs";
7
+ import { publicContractPolicyIssues, researchPublicContract } from "../contracts/public-contract.mjs";
8
+ import { continuationRepairTargets } from "./continuation-selection.mjs";
9
+ import { executionWorkloadIssues } from "../contracts/execution-workload.mjs";
10
+ import { automaticClaimCoverage } from "./automatic-admission.mjs";
11
+ import path from "node:path";
12
+ import { spawnSync } from "node:child_process";
13
+
14
+ import { CiteArkError, isRecord, readJson, safeRelativePath, sha256Value, writeJson } from "../util.mjs";
15
+ import { validateEvidenceParserDescriptor } from "../evidence/parser-registry.mjs";
16
+ import { validateDeclaredComputeRequirement } from "../compute/requirements.mjs";
17
+ import { EXECUTION_MODES } from "../contracts/execution-mode.mjs";
18
+ import { executionTimeoutCapMinutes } from "../contracts/execution-timeouts.mjs";
19
+ import {
20
+ CURRENT_PREFLIGHT_SCHEMA_VERSION,
21
+ describeSupportedPreflightSchemaVersions,
22
+ MAX_PREFLIGHT_MINUTES,
23
+ MIN_PUBLIC_ASSET_PREFLIGHT_MINUTES,
24
+ PREFLIGHT_CHECK_KINDS,
25
+ supportsPreflightSchemaVersion,
26
+ } from "../contracts/preflight-schema.mjs";
27
+ import {
28
+ IMPLEMENTATION_ORIGINS,
29
+ RECONSTRUCTION_REPOSITORY,
30
+ } from "../repository-origin.mjs";
31
+ import {
32
+ ASSET_ACQUISITION_MODES,
33
+ ASSET_SCIENTIFIC_ROLES,
34
+ assetLocalPath,
35
+ } from "../assets/plan.mjs";
36
+ import {
37
+ ASSET_REQUIREMENT_KINDS,
38
+ evaluateAssetRequirementCoverage,
39
+ normalizeAssetRequirements,
40
+ } from "../assets/requirements.mjs";
41
+ import {
42
+ commandVerifiesDependencies,
43
+ explicitlyUnderspecifiedHardwareProtocol,
44
+ hardwareSensitiveMetric,
45
+ isSimpleScriptEntrypoint,
46
+ parserPreflightTargetLeaks,
47
+ sanitizeParserPreflightCommand,
48
+ } from "./compiler-normalization-guards.mjs";
49
+
50
+ export const RESEARCH_SCHEMA_VERSION = "1.0";
51
+ export {
52
+ MAX_PREFLIGHT_MINUTES,
53
+ MIN_PUBLIC_ASSET_PREFLIGHT_MINUTES,
54
+ PREFLIGHT_CHECK_KINDS,
55
+ };
56
+ const CLAIM_TYPES = new Set(["finding", "method", "measurement", "limitation"]);
57
+ const SOURCE_KINDS = new Set(["paper", "repository", "dataset", "supplement", "checkpoint"]);
58
+ const REPRODUCTION_LEVELS = new Set(REPRODUCTION_ROUTES);
59
+ const REPRODUCTION_DISPOSITIONS = new Set(["planned", "deferred", "blocked", "not_applicable"]);
60
+ export const RECONSTRUCTION_FIDELITIES = new Set(["exact", "faithful", "approximate", "proxy"]);
61
+ const PREFLIGHT_ASSET_KINDS = new Set([
62
+ "dataset",
63
+ "checkpoint",
64
+ "model",
65
+ "repository",
66
+ "other",
67
+ ]);
68
+ const PREFLIGHT_ASSET_AVAILABILITY = new Set([
69
+ "bundled",
70
+ "public_verified",
71
+ "public_unverified",
72
+ "private",
73
+ "unknown",
74
+ ]);
75
+ const COVERAGE_STATUSES = new Set(["represented", "deferred", "blocked", "not_applicable"]);
76
+ import { SUBJECT_TAGS } from '../contracts/subject-tags.mjs';
77
+ export { SUBJECT_TAGS };
78
+
79
+ export async function loadResearchStructure(target) {
80
+ return normalizeResearchStructure(await readJson(target));
81
+ }
82
+
83
+ export async function normalizeResearchFile(inputPath, outputPath) {
84
+ const research = await loadResearchStructure(inputPath);
85
+ await writeJson(outputPath, research);
86
+ return research;
87
+ }
88
+
89
+ /**
90
+ * Accept a best-effort Research Compiler draft and deterministically turn local
91
+ * structural mistakes into warnings, conservative claim dispositions, or
92
+ * dropped non-executable experiments. The returned research object still
93
+ * passes the strict Artifact validator; this tolerance is only for the
94
+ * compiler boundary, never for third-party Artifact verification.
95
+ */
96
+ export function normalizeCompilerResearchDraft(input, context = {}, { collectIssues = false } = {}) {
97
+ if (!isRecord(input)) throw new CiteArkError("research.json 的根节点必须是对象");
98
+ const warnings = [];
99
+ const draft = structuredClone(input);
100
+ draft.schemaVersion = RESEARCH_SCHEMA_VERSION;
101
+ if (context.reproductionScope !== undefined) draft.reproductionScope = context.reproductionScope;
102
+ else delete draft.reproductionScope;
103
+ draft.work = compilerWork(draft.work, context, warnings);
104
+ draft.sources = compilerSources(draft.sources, context, warnings);
105
+ const sourceIds = new Set(draft.sources.map((source) => source.id));
106
+ const paperSourceId = draft.sources.find((source) => source.kind === "paper")?.id
107
+ ?? draft.sources[0]?.id;
108
+ draft.claims = compilerClaims(draft.claims, {
109
+ sourceIds,
110
+ paperSourceId,
111
+ warnings,
112
+ });
113
+ const inventoryOnly = context.compilationStage === "inventory";
114
+ if (inventoryOnly) draft.compilationStage = "inventory";
115
+ else delete draft.compilationStage;
116
+ draft.experiments = inventoryOnly ? [] : compilerExperiments(draft.experiments, {
117
+ claims: draft.claims,
118
+ researchObjects: draft.researchObjects,
119
+ repositoryIdentity: context.repositoryIdentity,
120
+ executionAllowed: context.licensePolicy?.executable !== false,
121
+ reproductionScope: context.reproductionScope,
122
+ continuation: context.continuation,
123
+ warnings,
124
+ });
125
+ if (context.licensePolicy?.executable === false) {
126
+ for (const claim of draft.claims) if (claim.reportedMeasurements.length) {
127
+ claim.reproduction = { status: "blocked", reason: "The recorded license policy does not allow execution.",
128
+ blocker: { kind: "access", evidence: "Runner-owned licensePolicy.executable=false" } };
129
+ }
130
+ }
131
+ if (inventoryOnly) {
132
+ for (const claim of draft.claims) claim.reproduction = { status: "deferred", reason: "Source inventory recorded; verification route and feasibility belong to reproduction preparation." };
133
+ } else reconcileCompilerDispositions(draft.claims, draft.experiments, warnings, context.computeContext);
134
+ draft.experiments = draft.experiments.filter((experiment) => experiment.claimIds.length > 0);
135
+ draft.coverage = compilerCoverage(draft.coverage, {
136
+ claims: draft.claims,
137
+ sourceIds,
138
+ paperSourceId,
139
+ warnings,
140
+ });
141
+ draft.lineage = compilerLineage(draft.lineage, {
142
+ sources: draft.sources,
143
+ claims: draft.claims,
144
+ experiments: draft.experiments,
145
+ warnings,
146
+ });
147
+ draft.license = compilerLicense(draft.license, context.licensePolicy, warnings);
148
+ draft.provenance = compilerProvenance(draft.provenance, context, warnings);
149
+ const allWarnings = [...new Set(warnings)];
150
+ const uniqueWarnings = allWarnings.length > 100
151
+ ? [
152
+ ...allWarnings.slice(0, 99),
153
+ `${allWarnings.length - 99} additional normalization warnings were omitted from this record.`,
154
+ ]
155
+ : allWarnings;
156
+ if (uniqueWarnings.length) draft.provenance.normalizationWarnings = uniqueWarnings;
157
+ else delete draft.provenance.normalizationWarnings;
158
+ try {
159
+ return {
160
+ research: normalizeResearchStructure(draft, { requireAuthors: true }),
161
+ warnings: uniqueWarnings,
162
+ structureIssues: [],
163
+ };
164
+ } catch (error) {
165
+ if (!collectIssues) throw error;
166
+ // Keep structure failures blocking while also collecting feasibility errors.
167
+ return { research: draft, warnings: uniqueWarnings, structureIssues: [error.message] };
168
+ }
169
+ }
170
+
171
+ /**
172
+ * Inspect a compiler draft for trust-boundary and execution-agent diagnostics.
173
+ * The caller retries only identity, scope and evidence-integrity violations.
174
+ * Operational findings remain useful audit guidance, but an agent-planned
175
+ * contract carries them forward instead of demanding another broad rewrite.
176
+ */
177
+ export function compilerReconstructionRetryIssues(research, context = {}) {
178
+ const issues = [];
179
+ const allExperiments = Array.isArray(research?.experiments) ? research.experiments : [];
180
+ const experiments = continuationFailedTargetExperiments(
181
+ allExperiments,
182
+ context.continuation,
183
+ );
184
+ const missingContinuationTargets = continuationTargetCoverageFailures(
185
+ allExperiments,
186
+ context.continuation,
187
+ );
188
+ if (missingContinuationTargets.length) {
189
+ const details = missingContinuationTargets.slice(0, 8)
190
+ .map(({ claimId, experimentId, disposition }) =>
191
+ `${disposition}:${[claimId, experimentId].filter(Boolean).join("/") || "unknown target"}`)
192
+ .join(", ");
193
+ issues.push(
194
+ `Continuation-target coverage retry: ${details} were recorded by the parent campaign but are absent from this repaired plan. A continuation may change resources, preflight, orchestration, or evidence generation for failed targets, but it must preserve every completed or failed experiment id and its claim binding. Completed targets must remain addressable under their original identity so the worker imports their evidence instead of rerunning a renamed duplicate; failed targets remain mandatory until fresh execution succeeds. Do not delete a difficult target, replace it with an already completed supporting experiment, rename a completed target, or mark a required target blocked merely to pass validation.`,
195
+ );
196
+ }
197
+ const hasOfficialRepository = Boolean(
198
+ context.repositoryIdentity?.available !== false
199
+ && textValue(context.repositoryIdentity?.url)
200
+ && textValue(context.repositoryIdentity?.commit),
201
+ );
202
+ const missingOfficialEntrypoints = officialEntrypointIdentityFailures(experiments, context);
203
+ if (missingOfficialEntrypoints.length) {
204
+ const details = missingOfficialEntrypoints.slice(0, 8)
205
+ .map(({ experimentId, references }) => `${experimentId}: ${references.join(", ")}`)
206
+ .join("; ");
207
+ issues.push(
208
+ `Official-entrypoint identity retry: ${details} are declared as official scientific entrypoints but are absent from the runner-owned snapshot of the fixed repository commit. An official plan may invoke only source files that actually exist in that snapshot. Inspect the repository inventory and replace invented adapters with the real checked-in command(s); capture and transform their raw stdout or source artifacts into /job/output evidence without changing the scientific command. If a report-defining implementation truly is absent, use a separately declared CiteArk reconstruction from the clean scaffold or leave that measurement blocked. Audit matching preflight commands and requiredCommandFragments at the same time; never describe a generated wrapper as checked in or official.`,
209
+ );
210
+ }
211
+ const reconstructionOriginLeaks = reconstructionOriginIsolationFailures(experiments, context);
212
+ if (reconstructionOriginLeaks.length) {
213
+ const details = reconstructionOriginLeaks.slice(0, 8)
214
+ .map(({ experimentId, references }) => `${experimentId}: ${references.join(", ")}`)
215
+ .join("; ");
216
+ issues.push(
217
+ `Reconstruction-origin isolation retry: ${details} are fixed-author-snapshot files referenced by experiments declared as citeark_reconstruction. An independent reconstruction runs from CiteArk's clean scaffold and must not copy, import, patch, or invoke the author snapshot. If a generated loop, evaluator driver, or evidence adapter calls a checked-in author implementation, keep that helper operational and declare the scientific experiment official with the real checked-in command visible in protocol.entrypoint and requiredCommandFragments. Otherwise implement the paper-described method independently under /job/workspace/repository without any author-source dependency, lower fidelity for material deviations, and keep missing conditions blocked when independence cannot be established. Audit protocol instructions, preflight checks, and command fragments together; changing only the repository label does not repair provenance.`,
218
+ );
219
+ }
220
+ const incompletePreflights = experiments
221
+ .filter((experiment) =>
222
+ experiment?.reconstructionFidelity !== "proxy"
223
+ && experiment?.protocol?.preflight?.source !== "declared",
224
+ );
225
+ if (incompletePreflights.length) {
226
+ issues.push(
227
+ `Execution preflight retry: ${incompletePreflights.slice(0, 8).map((experiment) => experiment.id).join(", ")} do not have a compiler-declared structured preflight. For every automatically runnable experiment, declare protocol.preflight with a bounded maxMinutes, runtime/dependency/entrypoint/parser/smoke-test checks, a real dataset_loader check for each required dataset, every required dataset/checkpoint/model asset and its current availability, and exact pass criteria. Do not defer known Python, dependency, checkpoint, dataset-loader, or parser requirements to the expensive execution phase.`,
228
+ );
229
+ }
230
+ const missingDurationEstimates = experiments.flatMap((experiment) =>
231
+ missingRunnableDurationEstimate(experiment),
232
+ );
233
+ if (missingDurationEstimates.length) {
234
+ const details = missingDurationEstimates.slice(0, 8)
235
+ .map(({ experimentId, missing }) => `${experimentId}: ${missing.join(" and ")}`)
236
+ .join("; ");
237
+ issues.push(
238
+ `Compute-estimate retry: ${details} do not have a finite positive duration estimate for every scheduler-eligible accelerator. Estimate the complete workload, including setup and evidence publication, from its declared units and a measured, documented, or explicitly assumed conservative rate. Explain the arithmetic, hardware, basis and uncertainty in compute.rationale. A rough transparent estimate is required; null, omission, the runtime ceiling, or a cap-aware placeholder is not an estimate. Retain a separate finite execution timeout and a bounded real-device probe that can recalibrate the estimate before expensive work.`,
239
+ );
240
+ }
241
+ const invalidExecutionTimeouts = experiments.flatMap((experiment) =>
242
+ runnableExecutionTimeoutIssues(experiment),
243
+ );
244
+ if (invalidExecutionTimeouts.length) {
245
+ const details = invalidExecutionTimeouts.slice(0, 8)
246
+ .map(({ experimentId, reason }) => `${experimentId}: ${reason}`)
247
+ .join("; ");
248
+ issues.push(
249
+ `Execution-timeout retry: ${details}. Every automatically runnable experiment must declare environment.timeoutMinutes explicitly. Treat it as a generous anomaly circuit breaker rather than the expected duration: keep it above the longest scheduler-eligible estimate plus a conservative reserve, but within the platform fidelity cap. Do not shorten the scientific workload merely to fit a timeout.`,
250
+ );
251
+ }
252
+ const mixedVariantExperiments = experiments.flatMap((experiment) => {
253
+ if (experiment?.reconstructionFidelity === "proxy") return [];
254
+ const variants = experimentReportedModelVariants(research, experiment);
255
+ return variants.length > 1 ? [{ experiment, variants }] : [];
256
+ });
257
+ if (mixedVariantExperiments.length) {
258
+ const details = mixedVariantExperiments.slice(0, 8)
259
+ .map(({ experiment, variants }) => `${experiment.id} binds ${variants.join(" / ")}`)
260
+ .join("; ");
261
+ issues.push(
262
+ `Model-variant isolation retry: ${details}. Split distinct paper-reported model variants into separately bound experiments with distinct entrypoint commands and evidence paths, then verify that every variant has a real official or explicitly reconstructed implementation path. A selector, boolean, loop label, or second evidence filename is not a model variant when it does not change the scientific computation. If the fixed official repository omits one variant, do not relabel the available implementation or duplicate its output; either ground a CiteArk reconstruction in the paper's algorithm and lower fidelity as needed, or leave that measurement blocked with the concrete missing condition.`,
263
+ );
264
+ }
265
+ const selfDisqualifiedExperiments = experiments.filter((experiment) =>
266
+ experiment?.reconstructionFidelity !== "proxy"
267
+ && Array.isArray(experiment?.measurements)
268
+ && experiment.measurements.length > 0
269
+ && explicitlyDisclaimsReportedScope(experiment),
270
+ );
271
+ if (selfDisqualifiedExperiments.length) {
272
+ issues.push(
273
+ `Scope-equivalence retry: ${selfDisqualifiedExperiments.slice(0, 8).map((experiment) => experiment.id).join(", ")} bind paper-reported measurements while their own title or protocol instructions describe the run as a smoke/plumbing check, not equivalent to the paper evaluation, or outside the reported scope. A non-proxy experiment cannot bind a reported measurement after explicitly disavowing its dataset, sample count, split, repetition, aggregation, or evaluation scope. Keep valid structural checks as separate supporting experiments, but either execute the report-defining scope for the measurement or mark the reduced run proxy and leave the paper measurement unsatisfied. Never compare a one-sample or synthetic-input agreement rate with a paper's full-dataset aggregate merely because both can equal 100%.`,
274
+ );
275
+ }
276
+ const reportedMeasurements = new Map(
277
+ (Array.isArray(research?.claims) ? research.claims : []).flatMap((claim) =>
278
+ (Array.isArray(claim?.reportedMeasurements) ? claim.reportedMeasurements : [])
279
+ .map((measurement) => [measurement.id, measurement]),
280
+ ),
281
+ );
282
+ const embeddedReportedValues = experiments.flatMap((experiment) =>
283
+ (Array.isArray(experiment?.measurements) ? experiment.measurements : []).flatMap((measurement) => {
284
+ const reported = reportedMeasurements.get(measurement?.reportedMeasurementId);
285
+ return entrypointEmbedsReportedValue(experiment?.protocol?.entrypoint, measurement, reported)
286
+ ? [{ experiment, measurement }]
287
+ : [];
288
+ }),
289
+ );
290
+ if (embeddedReportedValues.length) {
291
+ const details = embeddedReportedValues.slice(0, 8)
292
+ .map(({ experiment, measurement }) => `${experiment.id}/${measurement.reportedMeasurementId}`)
293
+ .join(", ");
294
+ issues.push(
295
+ `Evidence-integrity retry: ${details} embed the paper-reported measurement value directly in protocol.entrypoint under the parsed metric key. An entrypoint must derive raw evidence from executed model, dataset, evaluator, or source artifacts; it cannot print, assign, or serialize the target value as a literal. Parser fixtures may demonstrate a JSON shape in preflight, but the measurement evidence path must be produced from independently executed outputs. Remove the literal result and implement a real aggregation over raw per-sample or per-run artifacts, or leave the measurement unsatisfied.`,
296
+ );
297
+ }
298
+ const placeholderEntrypoints = experiments.filter((experiment) =>
299
+ experiment?.reconstructionFidelity !== "proxy"
300
+ && entrypointIsProsePlaceholder(experiment?.protocol?.entrypoint),
301
+ );
302
+ if (placeholderEntrypoints.length) {
303
+ issues.push(
304
+ `Entrypoint-executability retry: ${placeholderEntrypoints.slice(0, 8).map((experiment) => experiment.id).join(", ")} use a prose-printing placeholder as the immutable scientific command. Importing function names without calling them, then printing instructions such as “execute/evaluate/write ...”, performs no model, dataset, cycle, aggregation, or evidence work. Replace it with a command that actually invokes the fixed implementation, enumerates the declared scope, computes the bound measurement, and writes the parser evidence unchanged at execution time. An execution Agent must never be asked to invent the missing scientific driver.`,
305
+ );
306
+ }
307
+ const alreadyInvalidEntrypointIds = new Set([
308
+ ...placeholderEntrypoints.map((experiment) => experiment.id),
309
+ ...embeddedReportedValues.map(({ experiment }) => experiment.id),
310
+ ]);
311
+ const inconsistentEntrypointFragments = experiments.flatMap((experiment) => {
312
+ const entrypoint = textValue(experiment?.protocol?.entrypoint) ?? "";
313
+ if (
314
+ !entrypoint
315
+ || simpleEntrypointReference(entrypoint)
316
+ || alreadyInvalidEntrypointIds.has(experiment.id)
317
+ ) return [];
318
+ const fragments = Array.isArray(experiment?.protocol?.requiredCommandFragments)
319
+ ? experiment.protocol.requiredCommandFragments.map(textValue).filter(Boolean)
320
+ : [];
321
+ const missing = fragments.filter((fragment) => !entrypoint.includes(fragment));
322
+ return missing.length ? [{ experimentId: experiment.id, missing }] : [];
323
+ });
324
+ if (inconsistentEntrypointFragments.length) {
325
+ const details = inconsistentEntrypointFragments.slice(0, 8)
326
+ .map(({ experimentId, missing }) => `${experimentId}: ${missing.slice(0, 8).join(", ")}`)
327
+ .join("; ");
328
+ issues.push(
329
+ `Entrypoint-executability retry: ${details} are declared in requiredCommandFragments but are absent from the complete immutable protocol.entrypoint. For a full command, every required scientific fragment must be present in that command; a token used only by parser, smoke, duration, or another preflight fixture is not a scientific entrypoint fragment. Remove preflight-only fragments, or repair the immutable entrypoint when the missing operation is genuinely part of the paper measurement. Keep legacy path-only entrypoints separate because their recorded invocation may supply fixed arguments.`,
330
+ );
331
+ }
332
+ const invalidInlinePythonEntrypoints = inlinePythonSyntaxFailures(experiments);
333
+ if (invalidInlinePythonEntrypoints.length) {
334
+ issues.push(
335
+ `Entrypoint-syntax retry: ${invalidInlinePythonEntrypoints.slice(0, 8).map((experiment) => experiment.id).join(", ")} contain inline Python that does not parse before any scientific work begins. Validate the exact code passed to python -c with ast.parse or compile during compilation, then keep that same immutable command in the plan. In particular, do not mix a one-line for-suite with a subsequently indented nested loop; use a syntactically valid multiline block or a checked-in/generated repository script with a compile-only preflight. Never defer a deterministic syntax error to the expensive execution Agent.`,
336
+ );
337
+ }
338
+ const nonRepresentativeSmokeTests = smokePreflightFailures(experiments);
339
+ if (nonRepresentativeSmokeTests.length) {
340
+ issues.push(
341
+ `Execution preflight retry: ${nonRepresentativeSmokeTests.slice(0, 8).map((experiment) => experiment.id).join(", ")} declare corpus-scale neural work but use a constant, always-true, or non-scientific smoke command. The smoke test must execute a bounded representative path through the fixed implementation on one real or identity-verified fixture input, including model/evaluator loading, the decisive encode/decode/infer/evaluate operation, and evidence-shape generation. Assertions such as \`assert 2620 > 0\`, \`... or True\`, or parser-only fixtures do not prove end-to-end plumbing and must not defer the first real invocation to the full experiment.`,
342
+ );
343
+ }
344
+ const unverifiedOfficialCliInterfaces = officialCliInterfaceFailures(experiments);
345
+ if (unverifiedOfficialCliInterfaces.length) {
346
+ const details = unverifiedOfficialCliInterfaces.slice(0, 8)
347
+ .map(({ experimentId, script, options }) => `${experimentId}: ${script} ${options.join(" ")}`)
348
+ .join("; ");
349
+ issues.push(
350
+ `Entrypoint-executability retry: ${details} pass long options to an official repository script without an entrypoint preflight that invokes that exact script's --help interface and verifies every declared option. File existence does not prove a CLI accepts invented flags. Inspect the fixed script; if it exposes those options, add a non-scientific --help/interface check. If it does not, replace the command with a real inline driver that calls the checked-in functions and performs the complete declared scope, or leave the measurement blocked. Never defer an unsupported immutable command to the execution Agent.`,
351
+ );
352
+ }
353
+ const nonPortableAssetPaths = experiments.filter((experiment) =>
354
+ experimentUsesUnavailableDataRoot(experiment),
355
+ );
356
+ if (nonPortableAssetPaths.length) {
357
+ issues.push(
358
+ `Asset-path portability retry: ${nonPortableAssetPaths.slice(0, 8).map((experiment) => experiment.id).join(", ")} reference /data in immutable entrypoint or preflight commands, but CiteArk execution grants the unprivileged Agent a writable /job/assets root and does not promise a writable or mounted /data directory. Bind every acquired dataset, checkpoint, and model beneath /job/assets, use those exact paths in the immutable command and checks, and keep caches under /job/runtime-cache. Do not rely on root-created compatibility symlinks.`,
359
+ );
360
+ }
361
+ const mismatchedAssetPaths = experiments.flatMap(assetPathBindingFailures);
362
+ if (mismatchedAssetPaths.length) {
363
+ const details = mismatchedAssetPaths.slice(0, 8)
364
+ .map(({ experimentId, path: assetPath }) => `${experimentId}: ${assetPath}`)
365
+ .join("; ");
366
+ issues.push(
367
+ `Asset-path binding retry: ${details} reference an asset path that is not the immutable localPath derived from the declared asset kind and id. Use /job/assets/<kind>/<asset-id> consistently in the scientific entrypoint and every preflight command. Do not rely on the execution Agent to create compatibility symlinks for a mismatched compiler path.`,
368
+ );
369
+ }
370
+ const invalidParserFixtures = parserPreflightFixtureFailures(
371
+ experiments,
372
+ reportedMeasurements,
373
+ );
374
+ if (invalidParserFixtures.length) {
375
+ const details = invalidParserFixtures.slice(0, 8)
376
+ .map(({ experimentId, reasons }) => `${experimentId}: ${reasons.join(" and ")}`)
377
+ .join("; ");
378
+ issues.push(
379
+ `Parser-evidence generation retry: ${details}. A parser preflight must create its own tiny synthetic sentinel fixture in the same command before reading it, and the sentinel must not contain or compare against the paper-reported target value. Keep fixtures outside the measurement evidence path, validate only shape/unit/parser plumbing, and require the scientific entrypoint to derive final evidence from executed outputs.`,
380
+ );
381
+ }
382
+ const prematureFastTextPartitions = experiments.filter((experiment) => {
383
+ const entrypoint = textValue(experiment?.protocol?.entrypoint) ?? "";
384
+ const usesFastTextParser = (Array.isArray(experiment?.measurements)
385
+ ? experiment.measurements
386
+ : []).some((measurement) => measurement?.parser?.id === "fasttext.classification.test");
387
+ return usesFastTextParser
388
+ && /\^N\[\[:space:\]\][\s\S]{0,80}\{[^}]*\bexit\b[^}]*\}/i.test(entrypoint);
389
+ });
390
+ if (prematureFastTextPartitions.length) {
391
+ issues.push(
392
+ `Parser-evidence generation retry: ${prematureFastTextPartitions.slice(0, 8).map((experiment) => experiment.id).join(", ")} partition fastText native output by exiting on the N sample-count line before P@1 and R@1 are preserved. The registered fasttext.classification.test parser requires the complete native metric block at every parser.evidencePath. Change the deterministic partitioner to stop only after the required metric lines, and make declared preflight run the exact splitter on a representative N/P@1/R@1 fixture followed by the registered parser. Testing the parser alone on a hand-written file does not validate evidence generation.`,
393
+ );
394
+ }
395
+ const outputEntrypoints = experiments.filter((experiment) =>
396
+ experiment?.implementationOrigin === "citeark_reconstruction"
397
+ && /(?:^|[\s"'])\/job\/output\/[A-Za-z0-9._/-]+\.(?:py|sh|js|mjs|cjs|rb|pl|jl|r)(?=$|[\s"'])/i
398
+ .test(textValue(experiment?.protocol?.entrypoint) ?? ""),
399
+ );
400
+ if (outputEntrypoints.length) {
401
+ issues.push(
402
+ `Entrypoint-location retry: ${outputEntrypoints.slice(0, 8).map((experiment) => experiment.id).join(", ")} place generated scientific code under /job/output. Reconstruction entrypoints must live under /job/workspace/repository so source changes, checkpoints, and integrity policy share one auditable path; reserve /job/output for evidence, manifests, results, and other run products. Rewrite the immutable command to execute the repository-relative reconstruction script and keep parser evidence under /job/output.`,
403
+ );
404
+ }
405
+ const nonPortableRecursiveGlobs = experiments.filter((experiment) => {
406
+ const entrypoint = textValue(experiment?.protocol?.entrypoint) ?? "";
407
+ return /(?:^|[;&|]\s*|\b)(?:\/bin\/)?sh\s+-c\b/i.test(entrypoint)
408
+ && /(?:^|[\s"'])[^\s"']*\*\*\//.test(entrypoint);
409
+ });
410
+ if (nonPortableRecursiveGlobs.length) {
411
+ issues.push(
412
+ `Shell-portability retry: ${nonPortableRecursiveGlobs.slice(0, 8).map((experiment) => experiment.id).join(", ")} use recursive ** globs under sh -c. POSIX sh does not enable recursive globbing, so the immutable command can pass a literal nonexistent path or silently cover the wrong corpus. Replace it with deterministic null-safe find -print0 or pathlib.Path.rglob enumeration, sort identities, and make preflight verify the exact declared sample count before expensive execution.`,
413
+ );
414
+ }
415
+ const repeatedInterpreterCorpusLoops = experiments.filter((experiment) => {
416
+ const entrypoint = textValue(experiment?.protocol?.entrypoint) ?? "";
417
+ if (!/\bfor\s+[A-Za-z_][A-Za-z0-9_]*\s+in\b[\s\S]*\bpython(?:3)?\b/i.test(entrypoint)) {
418
+ return false;
419
+ }
420
+ const durations = Object.values(experiment?.compute?.estimatedDurationMinutes ?? {})
421
+ .map(Number)
422
+ .filter(Number.isFinite);
423
+ return durations.some((duration) => duration >= 30);
424
+ });
425
+ if (repeatedInterpreterCorpusLoops.length) {
426
+ issues.push(
427
+ `Corpus-execution retry: ${repeatedInterpreterCorpusLoops.slice(0, 8).map((experiment) => experiment.id).join(", ")} launch a fresh Python interpreter inside a long corpus loop. This commonly reloads the model or evaluator once per sample and makes the declared full-scope duration unreliable. Use one persistent official model/evaluator process with deterministic enumeration and auditable per-sample outputs, or provide a bounded startup/throughput benchmark proving the complete immutable loop fits its declared duration and budget. Prefer compatible accelerator execution when the fixed implementation supports it and CPU is materially slower.`,
428
+ );
429
+ }
430
+ const unbenchmarkedLargeWorkloads = experiments.filter((experiment) =>
431
+ requiresDurationFeasibilityBenchmark(experiment)
432
+ && !hasDurationFeasibilityCheck(experiment),
433
+ );
434
+ const missingWorkloads = experiments.filter((experiment) => requiresDurationFeasibilityBenchmark(experiment) && !experiment.protocol?.workload);
435
+ if (missingWorkloads.length) issues.push(`Workload-boundary retry: ${missingWorkloads.map((experiment) => experiment.id).join(", ")} must declare protocol.workload with full execution unit counts and groups covering different methods and material cost strata. Preserve every dataset, seed, epoch and aggregation requirement. A probe must execute a whole representative unit; the execution Agent measures it and forecasts all remaining units before committing to the full run.`);
436
+ if (unbenchmarkedLargeWorkloads.length) {
437
+ issues.push(
438
+ `Duration-feasibility retry: ${unbenchmarkedLargeWorkloads.slice(0, 8).map((experiment) => experiment.id).join(", ")} multiply corpus items, cycles, repetitions, epochs, steps, or trials into a large execution workload without a runnable duration preflight. Add a bounded representative benchmark that uses the fixed implementation and allocated device, records measured setup and per-unit throughput, extrapolates the complete immutable workload, and requires it to fit the contract timeout after reserving at least five minutes for evidence and result publication. If the forecast does not fit, redesign only the operational execution plan—such as safe batching, durable per-item progress, a compatible faster profile, or an honestly longer Research Plan budget—without reducing the scientific sample count, cycles, repetitions, aggregation, or other reported scope.`,
439
+ );
440
+ }
441
+ const computeEntrypointMismatches = experiments.filter((experiment) =>
442
+ experiment?.reconstructionFidelity !== "proxy"
443
+ && entrypointForcesCpuWhileDeclaringGpu(experiment),
444
+ );
445
+ if (computeEntrypointMismatches.length) {
446
+ issues.push(
447
+ `Compute-boundary retry: ${computeEntrypointMismatches.slice(0, 8).map((experiment) => experiment.id).join(", ")} declare GPU resources or GPU duration estimates while the immutable entrypoint hard-codes CPU execution. The scheduler can only honor the executable contract it receives. Either make the command select the allocated CUDA device and declare the accelerator required when the complete workload depends on it, or remove the GPU declaration and prove with a real CPU duration benchmark that the complete immutable run fits. Do not describe GPU as preferred while silently forcing the scientific command onto CPU.`,
448
+ );
449
+ }
450
+ const underconstrainedAccelerators = experiments.filter((experiment) =>
451
+ experiment?.reconstructionFidelity !== "proxy"
452
+ && largeWorkloadNeedsDeclaredAccelerator(experiment),
453
+ );
454
+ if (underconstrainedAccelerators.length) {
455
+ issues.push(
456
+ `Compute-boundary retry: ${underconstrainedAccelerators.slice(0, 8).map((experiment) => experiment.id).join(", ")} describe a large workload whose own estimates make GPU materially faster, but leave the accelerator optional with CPU fallback enabled. That contract is eligible for Standard CPU scheduling. If the promised duration relies on GPU throughput, set accelerator to required, disable CPU fallback, request a compatible GPU, and make the entrypoint use the allocated device. Keep accelerator optional only when a runnable CPU benchmark independently proves the complete workload fits the CPU timeout and budget.`,
457
+ );
458
+ }
459
+ const cycleCountMismatches = repeatedCycleCountFailures(experiments);
460
+ if (cycleCountMismatches.length) {
461
+ const details = cycleCountMismatches.slice(0, 8)
462
+ .map(({ experimentId, declaredCycles, loopCycles }) =>
463
+ `${experimentId}: declares ${declaredCycles} cycles but executes ${loopCycles}`)
464
+ .join("; ");
465
+ issues.push(
466
+ `Scope-equivalence retry: ${details}. The initial encoding establishes the baseline and is not a decode-re-encode cycle. Execute exactly the declared number of decode-re-encode transitions before comparing with the baseline; do not subtract one from the paper's cycle count or label a shorter loop as the reported scope.`,
467
+ );
468
+ }
469
+ const invalidFinalCycleComparisons = repeatedCycleFinalComparisonFailures(
470
+ experiments,
471
+ reportedMeasurements,
472
+ );
473
+ if (invalidFinalCycleComparisons.length) {
474
+ const details = invalidFinalCycleComparisons.slice(0, 8)
475
+ .map(({ experimentId, cycles }) => `${experimentId}: final comparison after ${cycles} cycles`)
476
+ .join("; ");
477
+ issues.push(
478
+ `Scope-equivalence retry: ${details} establish the baseline inside the repeated cycle loop or average agreement across intermediate cycles. A paper measurement defined as agreement between the first encoding and cycle N requires one baseline encoding before the loop, exactly N decode-re-encode transitions, and one final token-stream comparison per sample after the loop. Do not include the tautological baseline comparison or cycles 1..N-1 in the reported aggregate.`,
479
+ );
480
+ }
481
+ const selectionMismatches = reportedSelectionFailures(experiments, reportedMeasurements);
482
+ if (selectionMismatches.length) {
483
+ const details = selectionMismatches.slice(0, 8)
484
+ .map(({ experimentId, selection }) => `${experimentId}: ${selection}`)
485
+ .join("; ");
486
+ issues.push(
487
+ `Scope-equivalence retry: ${details}. The immutable entrypoint does not implement or consume evidence for the paper-declared sample selection. Sorting a corpus and taking the first N items is not a seeded random, reader/speaker-balanced, or stratified panel. Implement the exact deterministic seed and grouping algorithm in the command, or consume a digest-bound selection manifest whose construction and identities are declared and verified; otherwise lower the claim to the actually executed selection and leave the paper measurement unsatisfied.`,
488
+ );
489
+ }
490
+ const collapsedCorpusScopes = experiments.filter((experiment) =>
491
+ corpusScopeCollapsedToSingleExplicitInput(experiment, reportedMeasurements),
492
+ );
493
+ if (collapsedCorpusScopes.length) {
494
+ issues.push(
495
+ `Scope-equivalence retry: ${collapsedCorpusScopes.slice(0, 8).map((experiment) => experiment.id).join(", ")} declare a multi-sample corpus measurement but their immutable entrypoint processes only one explicit input file and contains no corpus or manifest enumeration. A declared sampleCount, dataset split, or aggregate cannot be satisfied by repeatedly invoking one fixture. Deterministically enumerate and verify the complete declared identities, process each distinct sample, and aggregate per-sample evidence; otherwise lower the experiment to the actually executed one-sample scope and leave the paper measurement unsatisfied.`,
496
+ );
497
+ }
498
+ const invalidSiSnrFormulas = siSnrFormulaFailures(experiments);
499
+ if (invalidSiSnrFormulas.length) {
500
+ issues.push(
501
+ `Evidence-integrity retry: ${invalidSiSnrFormulas.slice(0, 8).map((experiment) => experiment.id).join(", ")} implement SI-SNR without projecting the reconstructed estimate onto the clean reference. Computing the projection from reference-minus-estimate noise, or subtracting that projection from the reference, is a different quantity. Use a fixed trusted evaluator or compute target = <estimate, reference> / ||reference||^2 * reference, residual = estimate - target, then 10*log10(||target||^2 / ||residual||^2); add an analytical scale-invariance sanity check that exercises the same evaluator before corpus execution.`,
502
+ );
503
+ }
504
+ const missingSiSnrInvariantChecks = siSnrInvariantPreflightFailures(experiments);
505
+ if (missingSiSnrInvariantChecks.length) {
506
+ issues.push(
507
+ `Execution preflight retry: ${missingSiSnrInvariantChecks.slice(0, 8).map((experiment) => experiment.id).join(", ")} compute SI-SNR but their smoke preflight does not verify scale invariance with the same evaluator used by the corpus path. The smoke preflight and corpus entrypoint are separate command strings: each must define or import and call the same named evaluator. Repeating the arithmetic inline in the corpus path while naming the evaluator only in the smoke command does not satisfy this requirement. Use a non-colinear analytical reference/estimate pair, call the same named evaluator once on that pair and once on a clearly named non-unit scaled estimate such as scaled_estimate or analytic_scaled_estimate while keeping the reference fixed, then assert the two finite SI-SNR values agree within a small numerical tolerance. Scaling both reference and estimate together, using estimate = scalar * reference, or checking only finiteness is degenerate and cannot distinguish SI-SNR from an incorrectly oriented projection.`,
508
+ );
509
+ }
510
+ const leakedSiSnrInvariantFixtures = siSnrEntrypointInvariantFixtureFailures(experiments);
511
+ if (leakedSiSnrInvariantFixtures.length) {
512
+ issues.push(
513
+ `Execution preflight retry: ${leakedSiSnrInvariantFixtures.slice(0, 8).map((experiment) => experiment.id).join(", ")} copy the analytical SI-SNR scale-invariance fixture into the immutable scientific entrypoint or requiredCommandFragments. The named scaled estimate and tolerance assertion are smoke-only plumbing checks, not part of the corpus measurement or its command provenance. Keep the invariant in protocol.preflight smoke_test, remove its fixture variables and assertions from protocol.entrypoint, and remove preflight-only tokens such as scaled_estimate from requiredCommandFragments; do not satisfy fragment consistency by moving a smoke fixture into the scientific command.`,
514
+ );
515
+ }
516
+ const activeClaimIds = new Set(experiments.flatMap((experiment) =>
517
+ Array.isArray(experiment?.claimIds) ? experiment.claimIds : []));
518
+ const blockedClaims = (Array.isArray(research?.claims) ? research.claims : []).filter((claim) =>
519
+ ["blocked", "deferred"].includes(claim?.reproduction?.status)
520
+ && (!context.continuation || activeClaimIds.has(claim?.id)),
521
+ );
522
+ const compilerResourceBlocks = blockedClaims.filter((claim) => {
523
+ const reason = textValue(claim?.reproduction?.reason) ?? "";
524
+ return /(?:\b(?:compiler|compilation|current|available)\b[\s\S]{0,120}\b(?:cpu|host|job|budget|resources?)\b|\b(?:cpu|host|job|budget|resources?)\b[\s\S]{0,120}\b(?:compiler|compilation)\b)/i
525
+ .test(reason);
526
+ });
527
+ if (compilerResourceBlocks.length) {
528
+ issues.push(
529
+ `Compute-boundary retry: ${compilerResourceBlocks.slice(0, 8).map((claim) => claim.id).join(", ")} use the Research Compiler's CPU host, compilation budget, or inability to run the experiment during compilation as a scientific resource blocker. Compilation and experiment execution are separate allocations. Declare the complete experiment's CPU/GPU requirements and duration so the scheduler can choose GCP Standard or AutoDL; block only after identifying a concrete execution-profile, lifetime, storage, or experiment-budget limit.`,
530
+ );
531
+ }
532
+ const preverifiedPublicAssetBlocks = hasOfficialRepository
533
+ ? blockedClaims.filter((claim) => {
534
+ const reason = textValue(claim?.reproduction?.reason) ?? "";
535
+ return /(?:\b(?:checkpoint|model|dataset|evaluator|training (?:code|script))\b[\s\S]{0,180}\b(?:not bundled|no (?:checkpoint )?bytes|unverified at compil|pre-known digest|not (?:publicly )?(?:available|released|published)|unavailable|missing)|\bpublic\b[\s\S]{0,180}\b(?:unverified at compil|not bundled|missing digest|unavailable)|\b(?:not released|not published|no public)\b[\s\S]{0,120}\b(?:checkpoint|model|dataset|evaluator|code|script))/i
536
+ .test(reason);
537
+ })
538
+ : [];
539
+ if (preverifiedPublicAssetBlocks.length) {
540
+ issues.push(
541
+ `Public-asset preflight retry: ${preverifiedPublicAssetBlocks.slice(0, 8).map((claim) => claim.id).join(", ")} treat missing bundled bytes, a missing pre-known digest, an unavailable author script, or public assets being unverified during compilation as a blocker. Inspect the fixed paper and repository for the exact condition first. For grounded public checkpoints, datasets, and evaluators, declare public_unverified assets and make execution preflight resolve the transport, reject error/authentication content, and record byte size, SHA-256, format, and scientific identity. If author code is absent but the paper specifies a reconstructable protocol, create an isolated citeark_reconstruction experiment instead. Block only when the required scientific input or decision rule itself cannot be recovered, and name that exact blocker.`,
542
+ );
543
+ }
544
+ const incompleteCoverage = context.continuation
545
+ ? []
546
+ : (Array.isArray(research?.claims) ? research.claims : []).flatMap((claim) => {
547
+ if (
548
+ claim?.reproduction?.status === "not_applicable"
549
+ || !Array.isArray(claim?.reportedMeasurements)
550
+ || claim.reportedMeasurements.length === 0
551
+ ) return [];
552
+ const coverage = claimCatalogCoverage(
553
+ claim,
554
+ Array.isArray(research?.experiments) ? research.experiments : [],
555
+ );
556
+ return (coverage.bound.length > 0 || claim.reproduction?.status === "deferred") && coverage.missingIds.length > 0
557
+ ? [{ claim, coverage }]
558
+ : [];
559
+ });
560
+ if (incompleteCoverage.length) {
561
+ const details = incompleteCoverage.slice(0, 8).map(({ claim, coverage }) =>
562
+ `${claim.id} is missing ${coverage.missingIds.join(", ")}`,
563
+ ).join("; ");
564
+ const remainder = incompleteCoverage.length > 8
565
+ ? `; and ${incompleteCoverage.length - 8} more claims`
566
+ : "";
567
+ issues.push(
568
+ `Scientific coverage retry: the draft leaves empirical claim measurements unscheduled (${details}${remainder}). First check each measurement against the exact claim statement and cited source: an unrelated table column is not automatically a separate claim or a required condition of this one; bind its evidence to the scientifically meaningful proposition it tests, using a source-grounded inventory revision when necessary. Then schedule a scientifically coherent experiment for every feasible missing measurement, including claims with zero experiments. Separate experiments when their scientific questions, methods, decision rules or independently selectable routes differ. Preserve dataset, split and seed axes inside coherent method batches; a loop coordinate alone does not require another experiment, step or output object. Do not merge distinct questions merely to satisfy the schema. If a remaining measurement is genuinely impossible to execute, state the concrete hard blocker and leave the partial experiments intact instead of deleting valid work.`,
569
+ );
570
+ }
571
+ if (context.licensePolicy?.executable === false) return issues;
572
+ // A continuation is a bounded repair campaign over a prior immutable plan.
573
+ // Preserve honest blocked coverage instead of expanding the campaign into
574
+ // unrelated claims while repairing the diagnosed failed targets.
575
+ if (context.continuation) return issues;
576
+ const partiallyCoveredClaimIds = new Set(incompleteCoverage.filter(({ coverage }) => coverage.bound.length > 0).map(({ claim }) => claim.id));
577
+ const unresolved = (Array.isArray(research?.claims) ? research.claims : []).filter((claim) =>
578
+ claim?.type !== "limitation"
579
+ && (
580
+ claim?.type === "finding"
581
+ || claim?.type === "measurement"
582
+ || (Array.isArray(claim?.reportedMeasurements) && claim.reportedMeasurements.length > 0)
583
+ )
584
+ && !(Array.isArray(claim.reproduction?.derivedFromClaimIds)
585
+ && claim.reproduction.derivedFromClaimIds.length > 0)
586
+ && claim.reproduction?.status !== "planned"
587
+ && !partiallyCoveredClaimIds.has(claim.id),
588
+ );
589
+ if (!unresolved.length) return issues;
590
+ const ids = unresolved.slice(0, 8).map((claim) => claim.id).join(", ");
591
+ const remainder = unresolved.length > 8 ? ` and ${unresolved.length - 8} more` : "";
592
+ const officialAssetGuidance = hasOfficialRepository
593
+ ? " The fixed official repository is available. An official public checkpoint, dataset, or evaluator must not be treated as unavailable merely because its bytes were not bundled or its digest was not known at compile time: declare it public_unverified, make preflight fetch it from the paper- or repository-grounded URL, record the resolved URL, byte size, SHA-256, archive/member identity, and format checks in the acquisition manifest, and stop before the expensive phase only if that verification actually fails. Prioritize a direct experiment on the paper's central empirical claim; a static configuration, parameter-count, or formula check is useful secondary evidence but is not a substitute for the headline result when the official public assets appear reachable. If the fixed official snapshot demonstrably omits or breaks the implementation for a report-defining condition, keep runnable official conditions as official experiments and independently reconstruct only the missing condition from the clean CiteArk scaffold with implementationOrigin citeark_reconstruction. Never copy, relabel, or silently repair author code as an independent reconstruction, and disclose every reconstruction assumption and material deviation."
594
+ : "";
595
+ issues.push(
596
+ `Direct-evidence retry: paper-grounded empirical claims remain unscheduled (${ids}${remainder}). Make one more attempt to design an exact, faithful, or honest approximate experiment that directly measures the paper-bound outcome.${officialAssetGuidance} Pin every recoverable material condition and disclose every substitution. Missing author code or an incompletely specified implementation is not by itself a blocker: independently reconstruct the method when the paper gives enough information to define a meaningful test. Proxy experiments remain researcher-selected because they test a substitute rather than the paper-bound measurement. Keep a claim blocked only when a concrete required scientific input, safe decision rule, access right, or feasible execution profile is unavailable; state that exact blocker.`,
597
+ );
598
+ return issues;
599
+ }
600
+
601
+ function assetPathBindingFailures(experiment) {
602
+ const assets = Array.isArray(experiment?.protocol?.preflight?.assets)
603
+ ? experiment.protocol.preflight.assets
604
+ : [];
605
+ if (!assets.length) return [];
606
+ const allowed = assets.map((asset) => assetLocalPath(asset?.kind ?? "other", asset?.id ?? "asset"));
607
+ const texts = [
608
+ experiment?.protocol?.entrypoint,
609
+ ...(Array.isArray(experiment?.protocol?.preflight?.checks)
610
+ ? experiment.protocol.preflight.checks.map((check) => check?.command)
611
+ : []),
612
+ ].filter((value) => typeof value === "string");
613
+ const paths = [...new Set(texts.flatMap((value) =>
614
+ [...value.matchAll(/\/job\/assets\/[A-Za-z0-9._/-]+/g)].map((match) =>
615
+ match[0].replace(/[),;'"\]}]+$/, ""),
616
+ )))];
617
+ return paths
618
+ .filter((candidate) => !allowed.some((root) => candidate === root || candidate.startsWith(`${root}/`)))
619
+ .map((candidate) => ({ experimentId: experiment?.id ?? "unknown", path: candidate }));
620
+ }
621
+
622
+ export function blockingCompilerReconstructionIssues(issues, {
623
+ agentPlanned = true,
624
+ } = {}) {
625
+ const candidates = Array.isArray(issues) ? issues : [];
626
+ const trustBoundary = /^(?:Continuation-repair boundary|Continuation-target coverage|Reconstruction-origin isolation) retry:/;
627
+ const quoteBoundary = /^Compute-estimate retry:/;
628
+ if (agentPlanned) return candidates.filter((issue) =>
629
+ trustBoundary.test(issue) || quoteBoundary.test(issue));
630
+
631
+ // Textual scope, model-label and result-literal heuristics are advisories in
632
+ // both modes: they cannot establish scientific violations.
633
+ // Legacy fixed-entrypoint plans cannot repair operational defects while
634
+ // running. Keep the old compile-time gate only for that compatibility mode.
635
+ const fixedEntrypointBoundary = /^(?:Official-entrypoint identity|Execution preflight|Entrypoint-executability|Entrypoint-syntax|Asset-path portability|Asset-path binding|Parser-evidence generation|Shell-portability|Corpus-execution|Duration-feasibility|Compute-boundary|Public-asset preflight) retry:/;
636
+ return candidates.filter((issue) =>
637
+ trustBoundary.test(issue) || quoteBoundary.test(issue) || fixedEntrypointBoundary.test(issue),
638
+ );
639
+ }
640
+
641
+ function inlinePythonSyntaxFailures(experiments) {
642
+ return experiments.filter((experiment) => {
643
+ if (experiment?.reconstructionFidelity === "proxy") return false;
644
+ const entrypoint = textValue(experiment?.protocol?.entrypoint) ?? "";
645
+ const match = /^\s*python(?:3(?:\.\d+)?)?\s+-c\s+(["'])([\s\S]*)\1\s*$/i.exec(entrypoint);
646
+ if (!match) return false;
647
+ const quote = match[1];
648
+ const source = quote === '"'
649
+ ? match[2].replace(/\\(["\\$`\n])/g, "$1")
650
+ : match[2];
651
+ const parsed = spawnSync(
652
+ "python3",
653
+ [
654
+ "-c",
655
+ [
656
+ "import ast,sys",
657
+ "pending=[sys.stdin.read()]",
658
+ "seen=set()",
659
+ "while pending:",
660
+ " source=pending.pop()",
661
+ " if source in seen: continue",
662
+ " seen.add(source)",
663
+ " tree=ast.parse(source)",
664
+ " for node in ast.walk(tree):",
665
+ " if isinstance(node,ast.Call) and isinstance(node.func,ast.Name) and node.func.id=='exec' and node.args and isinstance(node.args[0],ast.Constant) and isinstance(node.args[0].value,str): pending.append(node.args[0].value)",
666
+ ].join("\n"),
667
+ ],
668
+ {
669
+ input: source,
670
+ encoding: "utf8",
671
+ timeout: 5_000,
672
+ maxBuffer: 64 * 1024,
673
+ },
674
+ );
675
+ return parsed.error || parsed.status !== 0;
676
+ });
677
+ }
678
+
679
+ function inlinePythonExecutableSources(entrypoint) {
680
+ const value = textValue(entrypoint) ?? "";
681
+ const match = /^\s*python(?:3(?:\.\d+)?)?\s+-c\s+(["'])([\s\S]*)\1\s*$/i.exec(value);
682
+ if (!match) return [value];
683
+ const quote = match[1];
684
+ const source = quote === '"'
685
+ ? match[2].replace(/\\(["\\$`\n])/g, "$1")
686
+ : match[2];
687
+ const parsed = spawnSync(
688
+ "python3",
689
+ [
690
+ "-c",
691
+ [
692
+ "import ast,json,sys",
693
+ "pending=[sys.stdin.read()]",
694
+ "seen=set()",
695
+ "sources=[]",
696
+ "while pending:",
697
+ " source=pending.pop()",
698
+ " if source in seen: continue",
699
+ " seen.add(source)",
700
+ " sources.append(source)",
701
+ " try: tree=ast.parse(source)",
702
+ " except SyntaxError: continue",
703
+ " for node in ast.walk(tree):",
704
+ " if isinstance(node,ast.Call) and isinstance(node.func,ast.Name) and node.func.id=='exec' and node.args and isinstance(node.args[0],ast.Constant) and isinstance(node.args[0].value,str): pending.append(node.args[0].value)",
705
+ "print(json.dumps(sources))",
706
+ ].join("\n"),
707
+ ],
708
+ {
709
+ input: source,
710
+ encoding: "utf8",
711
+ timeout: 5_000,
712
+ maxBuffer: 256 * 1024,
713
+ },
714
+ );
715
+ if (parsed.error || parsed.status !== 0) return [value, source];
716
+ try {
717
+ const sources = JSON.parse(parsed.stdout);
718
+ return [value, ...(Array.isArray(sources) ? sources : [source])];
719
+ } catch {
720
+ return [value, source];
721
+ }
722
+ }
723
+
724
+ function smokePreflightFailures(experiments) {
725
+ return experiments.filter((experiment) => {
726
+ if (!requiresDurationFeasibilityBenchmark(experiment)) return false;
727
+ const checks = Array.isArray(experiment?.protocol?.preflight?.checks)
728
+ ? experiment.protocol.preflight.checks
729
+ : [];
730
+ const smoke = checks.find((check) => check?.kind === "smoke_test");
731
+ const command = textValue(smoke?.command) ?? "";
732
+ if (!command) return true;
733
+ if (/\bor\s+True\b|\bassert\s+True\b/i.test(command)) return true;
734
+ return !/(?:\b(?:load_model|encode|decode|infer|evaluate|predict|fit|train|run)\s*\(|\bpython(?:3)?\s+\/job\/workspace\/repository\/[A-Za-z0-9._/-]+\.py\b)/i.test(command);
735
+ });
736
+ }
737
+
738
+ function reportedSelectionFailures(experiments, reportedMeasurements) {
739
+ return experiments.flatMap((experiment) => {
740
+ const entrypoint = textValue(experiment?.protocol?.entrypoint) ?? "";
741
+ return (Array.isArray(experiment?.measurements) ? experiment.measurements : []).flatMap((measurement) => {
742
+ const reported = reportedMeasurements.get(measurement?.reportedMeasurementId);
743
+ const selection = textValue(reported?.dimensions?.selection);
744
+ if (!selection) return [];
745
+ const needsRandom = /(?:random|seed)/i.test(selection);
746
+ const needsGrouping = /(?:reader|speaker|balanced|stratif)/i.test(selection);
747
+ if (!needsRandom && !needsGrouping) return [];
748
+ const usesManifest = /(?:selection|sample|reader|speaker)[-_]?(?:manifest|ids?|list)|manifest[-_]?(?:path|file)/i.test(entrypoint);
749
+ const implementsRandom = /(?:\brandom\b|manual_seed|default_rng|RandomState|Generator\s*\(|randperm|shuffle|\bseed\s*[=(])/i.test(entrypoint);
750
+ const implementsGrouping = /(?:reader|speaker|stratif|balance|groupby|group_by|defaultdict)/i.test(entrypoint);
751
+ if ((needsRandom && !implementsRandom && !usesManifest)
752
+ || (needsGrouping && !implementsGrouping && !usesManifest)) {
753
+ return [{
754
+ experimentId: experiment?.id ?? "unnamed-experiment",
755
+ selection,
756
+ }];
757
+ }
758
+ return [];
759
+ });
760
+ });
761
+ }
762
+
763
+ function continuationTargetCoverageFailures(experiments, continuation) {
764
+ const completedTargets = Array.isArray(continuation?.completedTargets)
765
+ ? continuation.completedTargets
766
+ : [];
767
+ const failedTargets = Array.isArray(continuation?.failedTargets)
768
+ ? continuation.failedTargets
769
+ : [];
770
+ return [
771
+ ...completedTargets.map((target) => ({ target, disposition: "completed" })),
772
+ ...failedTargets.map((target) => ({ target, disposition: "failed" })),
773
+ ...(continuation?.requestedTargets ?? []).map((target) => ({ target, disposition: "requested" })),
774
+ ].filter(({ target }) => {
775
+ if (!isRecord(target)) return false;
776
+ const experimentId = textValue(target.experimentId);
777
+ const claimId = textValue(target.claimId);
778
+ if (!experimentId && !claimId) return false;
779
+ if (experimentId) {
780
+ const experiment = experiments.find((candidate) => candidate?.id === experimentId);
781
+ return !experiment
782
+ || (claimId && !experiment.claimIds?.includes(claimId));
783
+ }
784
+ return !experiments.some((experiment) =>
785
+ experiment?.reconstructionFidelity !== "proxy"
786
+ && experiment?.claimIds?.includes(claimId));
787
+ }).map(({ target, disposition }) => ({ ...target, disposition }));
788
+ }
789
+
790
+ function continuationFailedTargetExperiments(experiments, continuation) {
791
+ const failedIds = new Set(
792
+ continuationRepairTargets(continuation)
793
+ .filter(isRecord)
794
+ .map((target) => textValue(target.experimentId))
795
+ .filter(Boolean),
796
+ );
797
+ return failedIds.size
798
+ ? experiments.filter((experiment) => failedIds.has(experiment?.id))
799
+ : experiments;
800
+ }
801
+
802
+ function corpusScopeCollapsedToSingleExplicitInput(experiment, reportedMeasurements) {
803
+ if (experiment?.reconstructionFidelity === "proxy") return false;
804
+ const workloadKeys = /^(?:sample(?:_?count)?|samples|items|clips|utterances|files)$/i;
805
+ const declaredCounts = (Array.isArray(experiment?.measurements) ? experiment.measurements : [])
806
+ .flatMap((measurement) => {
807
+ const reported = reportedMeasurements?.get(measurement?.reportedMeasurementId);
808
+ return Object.entries(isRecord(reported?.dimensions) ? reported.dimensions : {});
809
+ })
810
+ .filter(([key, value]) => workloadKeys.test(key) && Number.isFinite(Number(value)))
811
+ .map(([, value]) => Number(value));
812
+ if (Math.max(0, ...declaredCounts) <= 1) return false;
813
+
814
+ const entrypoint = textValue(experiment?.protocol?.entrypoint) ?? "";
815
+ const enumeratesInputs = /(?:\brglob\s*\(|\bglob\s*\(|glob\.iglob|os\.walk|listdir\s*\(|scandir\s*\(|\bfind\s+[^\n;&|]*-(?:type|name)|manifest|sample[_-]?(?:ids?|list)|reader[_-]?(?:ids?|list)|speaker[_-]?(?:ids?|list))/i
816
+ .test(entrypoint);
817
+ if (enumeratesInputs) return false;
818
+ const explicitInputs = new Set(
819
+ [...entrypoint.matchAll(/(?:^|[\s='"(])([^\s'"()]+\.(?:wav|flac|mp3|ogg|m4a))(?:$|[\s'"),;])/gi)]
820
+ .map((match) => match[1])
821
+ .filter((inputPath) => !inputPath.startsWith("/job/output/")),
822
+ );
823
+ return explicitInputs.size === 1
824
+ && /(?:--input\b|input[_-]?(?:path|file)?\s*=|load_(?:wav|audio)\s*\(|read\s*\([^)]*\.(?:wav|flac)|soundfile\.read|torchaudio\.load)/i
825
+ .test(entrypoint);
826
+ }
827
+
828
+ function requiresDurationFeasibilityBenchmark(experiment) {
829
+ if (experiment?.reconstructionFidelity === "proxy") return false;
830
+ const workloadKeys = /^(?:sample(?:_?count)?|samples|items|clips|cycles?|repetitions?|repeats?|runs?|seeds?|epochs?|steps?|iterations?|trials?)$/i;
831
+ const declaredByDimensions = (Array.isArray(experiment?.measurements) ? experiment.measurements : []).some((measurement) => {
832
+ const dimensions = isRecord(measurement?.dimensions) ? measurement.dimensions : {};
833
+ const factors = Object.entries(dimensions)
834
+ .filter(([key, value]) => workloadKeys.test(key) && Number.isFinite(Number(value)) && Number(value) > 1)
835
+ .map(([, value]) => Number(value));
836
+ if (!factors.length) return false;
837
+ return factors.reduce((product, value) => product * value, 1) >= 1_000;
838
+ });
839
+ if (declaredByDimensions) return true;
840
+
841
+ const protocol = experiment?.protocol ?? {};
842
+ const workloadText = [
843
+ experiment?.title,
844
+ protocol.entrypoint,
845
+ protocol.instructions,
846
+ ...(Array.isArray(experiment?.measurements)
847
+ ? experiment.measurements.flatMap((measurement) => [measurement?.aggregation])
848
+ : []),
849
+ ].map((value) => textValue(value) ?? "").join("\n");
850
+ const corpusCounts = [
851
+ ...[...workloadText.matchAll(/\blen\s*\([^)]*\)\s*==\s*([0-9][0-9,_]*)/gi)],
852
+ ...[...workloadText.matchAll(/\b([0-9][0-9,_]*)\s+(?:distinct\s+|unique\s+|total\s+)?(?:samples?|clips?|items?|utterances?|files?)(?:\b|_)/gi)],
853
+ ].map((match) => Number(match[1].replaceAll(/[,_]/g, ""))).filter(Number.isFinite);
854
+ const loopCounts = [
855
+ ...[...workloadText.matchAll(/\brange\s*\(\s*([0-9][0-9,_]*)\s*\)/gi)],
856
+ ...[...workloadText.matchAll(/\b([0-9][0-9,_]*)\s+(?:total\s+)?(?:cycles?|repetitions?|repeats?|epochs?|steps?|iterations?|trials?)(?:\b|_)/gi)],
857
+ ].map((match) => Number(match[1].replaceAll(/[,_]/g, ""))).filter(Number.isFinite);
858
+ const largestCorpus = Math.max(0, ...corpusCounts);
859
+ const largestLoop = Math.max(1, ...loopCounts);
860
+ return largestCorpus >= 1_000 || largestCorpus * largestLoop >= 1_000;
861
+ }
862
+
863
+ function hasDurationFeasibilityCheck(experiment) {
864
+ const checks = Array.isArray(experiment?.protocol?.preflight?.checks)
865
+ ? experiment.protocol.preflight.checks
866
+ : [];
867
+ const entrypoint = textValue(experiment?.protocol?.entrypoint) ?? "";
868
+ return checks.some((check) => {
869
+ if (check?.kind !== "duration") return false;
870
+ const command = textValue(check.command) ?? "";
871
+ const criterion = textValue(check.successCriterion) ?? "";
872
+ if (experiment?.protocol?.executionMode === "agent_planned") {
873
+ return /(?:representative|throughput|workload|slice|sample|代表性|吞吐|工作量|切片|样本)/i.test(criterion)
874
+ && /(?:extrapolat|complete|full|total|budget|timeout|reserve|完整|总时限|预算|预留)/i.test(criterion);
875
+ }
876
+ const measuresElapsedTime = /(?:perf_counter|monotonic|process_time|date\s+\+%s|\/usr\/bin\/time|\btime\s+-p\b)/i.test(command);
877
+ const executesRepresentativeWork = /(?:\bbenchmark\s*\(|\b(?:encode|decode|infer|evaluate|predict|fit|train|run)\s*\()/i.test(command);
878
+ const enforcesForecast = /(?:\bassert\b|sys\.exit\s*\(|\braise\b|\bexit\s+[1-9]\b|\btest\b[^\n;&|]*(?:-lt|-le)\b)/i.test(command);
879
+ const usesSameModelLoader = !/\bload_model\s*\(/i.test(entrypoint)
880
+ || /\bload_model\s*\(/i.test(command);
881
+ const usesSameInputLoader = !/\bload_wav\s*\(/i.test(entrypoint)
882
+ || /\bload_wav\s*\(/i.test(command);
883
+ const usesRepresentativeCorpusInput = !/(?:\brglob\s*\(|\/job\/assets\/)/i.test(entrypoint)
884
+ || (/(?:\brglob\s*\(|\/job\/assets\/)/i.test(command)
885
+ && !(/\btorch\.(?:randn|rand|zeros|ones)\s*\(/i.test(command)
886
+ && !/\bload_wav\s*\(/i.test(command)));
887
+ return measuresElapsedTime
888
+ && executesRepresentativeWork
889
+ && enforcesForecast
890
+ && usesSameModelLoader
891
+ && usesSameInputLoader
892
+ && usesRepresentativeCorpusInput
893
+ && /(?:extrapolat|complete|full|total|budget|timeout|reserve|完整|总时限|预算|预留)/i.test(criterion);
894
+ });
895
+ }
896
+
897
+ function missingRunnableDurationEstimate(experiment) {
898
+ if (experiment?.reconstructionFidelity === "proxy") return [];
899
+ const experimentId = experiment?.id ?? "unnamed-experiment";
900
+ const compute = experiment?.compute;
901
+ if (!isRecord(compute)) {
902
+ return [{
903
+ experimentId,
904
+ missing: ["declared compute requirement and duration estimate"],
905
+ }];
906
+ }
907
+ const eligible = compute.accelerator === "none"
908
+ ? ["cpu"]
909
+ : compute.accelerator === "required"
910
+ ? ["gpu"]
911
+ : compute.accelerator === "optional"
912
+ ? compute.cpuFallbackAllowed === false
913
+ ? ["gpu"]
914
+ : ["cpu", "gpu"]
915
+ : ["cpu", "gpu"];
916
+ const missing = eligible.filter((accelerator) => {
917
+ const minutes = compute?.estimatedDurationMinutes?.[accelerator];
918
+ return typeof minutes !== "number" || !Number.isFinite(minutes) || minutes <= 0;
919
+ });
920
+ return missing.length
921
+ ? [{
922
+ experimentId,
923
+ missing: missing.map((accelerator) => `compute.estimatedDurationMinutes.${accelerator}`),
924
+ }]
925
+ : [];
926
+ }
927
+
928
+ function compilerExecutionEnvironment(value, {
929
+ compute,
930
+ experimentId,
931
+ reconstructionFidelity,
932
+ warnings,
933
+ }) {
934
+ const environment = isRecord(value) ? { ...value } : {};
935
+ const proposed = Number(environment.timeoutMinutes);
936
+ const capMinutes = executionTimeoutCapMinutes(reconstructionFidelity);
937
+ const estimates = schedulerEligibleDurationEstimates({ compute });
938
+ const longestEstimate = estimates.length ? Math.max(...estimates) : null;
939
+ const reserveMinutes = longestEstimate === null
940
+ ? null
941
+ : Math.max(30, Math.min(360, Math.ceil(longestEstimate * 0.25)));
942
+ const requiredTimeout = longestEstimate === null ? 60 : longestEstimate + reserveMinutes;
943
+ const fallback = Math.min(capMinutes, requiredTimeout);
944
+ const normalized = Number.isFinite(proposed) && proposed > 0
945
+ ? Math.min(capMinutes, Math.max(fallback, proposed))
946
+ : fallback;
947
+ if (normalized !== proposed) {
948
+ const reason = !Number.isFinite(proposed) || proposed <= 0
949
+ ? "was missing or invalid"
950
+ : proposed > capMinutes
951
+ ? `exceeded the ${capMinutes}-minute fidelity cap`
952
+ : `did not include the ${reserveMinutes}-minute anomaly reserve`;
953
+ warnings.push(`Experiment ${experimentId} environment.timeoutMinutes ${reason} and was normalized to ${normalized}.`);
954
+ }
955
+ return { ...environment, timeoutMinutes: normalized };
956
+ }
957
+
958
+ function runnableExecutionTimeoutIssues(experiment) {
959
+ if (experiment?.reconstructionFidelity === "proxy") return [];
960
+ const experimentId = textValue(experiment?.id) ?? "unknown-experiment";
961
+ const timeoutMinutes = Number(experiment?.environment?.timeoutMinutes);
962
+ const fidelity = experiment?.reconstructionFidelity
963
+ ?? (experiment?.implementationOrigin === "citeark_reconstruction" ? "approximate" : "faithful");
964
+ const capMinutes = executionTimeoutCapMinutes(fidelity);
965
+ if (!Number.isFinite(timeoutMinutes) || timeoutMinutes <= 0) {
966
+ return [{ experimentId, reason: "environment.timeoutMinutes is missing or invalid" }];
967
+ }
968
+ if (timeoutMinutes > capMinutes) {
969
+ return [{
970
+ experimentId,
971
+ reason: `environment.timeoutMinutes ${timeoutMinutes} exceeds the ${capMinutes}-minute ${fidelity} circuit-breaker cap`,
972
+ }];
973
+ }
974
+ const estimates = schedulerEligibleDurationEstimates(experiment);
975
+ if (!estimates.length) return [];
976
+ const longestEstimate = Math.max(...estimates);
977
+ const reserveMinutes = Math.max(30, Math.min(360, Math.ceil(longestEstimate * 0.25)));
978
+ const requiredTimeout = longestEstimate + reserveMinutes;
979
+ if (timeoutMinutes < requiredTimeout) {
980
+ return [{
981
+ experimentId,
982
+ reason: `environment.timeoutMinutes ${timeoutMinutes} must be at least ${requiredTimeout} for the ${longestEstimate}-minute estimate and ${reserveMinutes}-minute reserve`,
983
+ }];
984
+ }
985
+ return [];
986
+ }
987
+
988
+ function schedulerEligibleDurationEstimates(experiment) {
989
+ const compute = experiment?.compute;
990
+ if (!isRecord(compute)) return [];
991
+ const eligible = compute.accelerator === "none"
992
+ ? ["cpu"]
993
+ : compute.accelerator === "required"
994
+ ? ["gpu"]
995
+ : compute.accelerator === "optional"
996
+ ? compute.cpuFallbackAllowed === false
997
+ ? ["gpu"]
998
+ : ["cpu", "gpu"]
999
+ : ["cpu", "gpu"];
1000
+ return eligible
1001
+ .map((accelerator) => Number(compute?.estimatedDurationMinutes?.[accelerator]))
1002
+ .filter((minutes) => Number.isFinite(minutes) && minutes > 0);
1003
+ }
1004
+
1005
+ function entrypointForcesCpuWhileDeclaringGpu(experiment) {
1006
+ const compute = experiment?.compute ?? {};
1007
+ const declaresGpu = Number(compute.gpuCount ?? 0) > 0
1008
+ || compute.accelerator === "required"
1009
+ || (Array.isArray(compute.preferredGpuTypes) && compute.preferredGpuTypes.length > 0)
1010
+ || Number.isFinite(Number(compute?.estimatedDurationMinutes?.gpu));
1011
+ if (!declaresGpu) return false;
1012
+ const entrypoint = textValue(experiment?.protocol?.entrypoint) ?? "";
1013
+ return /(?:torch\.device\s*\(\s*["']cpu["']\s*\)|device\s*=\s*["']cpu["']|--device(?:=|\s+)cpu\b|\.to\s*\(\s*["']cpu["']\s*\))/i.test(entrypoint);
1014
+ }
1015
+
1016
+ function largeWorkloadNeedsDeclaredAccelerator(experiment) {
1017
+ if (!requiresDurationFeasibilityBenchmark(experiment)) return false;
1018
+ const compute = experiment?.compute ?? {};
1019
+ if (compute.accelerator === "required" || compute.cpuFallbackAllowed === false) return false;
1020
+ if (Number(compute.gpuCount ?? 0) < 1) return false;
1021
+ const cpuMinutes = Number(compute?.estimatedDurationMinutes?.cpu);
1022
+ const gpuMinutes = Number(compute?.estimatedDurationMinutes?.gpu);
1023
+ return Number.isFinite(cpuMinutes)
1024
+ && Number.isFinite(gpuMinutes)
1025
+ && gpuMinutes > 0
1026
+ && cpuMinutes / gpuMinutes >= 1.5;
1027
+ }
1028
+
1029
+ function repeatedCycleCountFailures(experiments) {
1030
+ return experiments.flatMap((experiment) => {
1031
+ const entrypoint = textValue(experiment?.protocol?.entrypoint) ?? "";
1032
+ if (!/(?:encode\s*\([^;\n]{0,160}decode\s*\(|decode\s*\([^;\n]{0,160}encode\s*\()/i.test(entrypoint)) return [];
1033
+ const scopeText = [
1034
+ experiment?.title,
1035
+ experiment?.protocol?.instructions,
1036
+ ...(Array.isArray(experiment?.measurements)
1037
+ ? experiment.measurements.flatMap((measurement) => [measurement?.aggregation])
1038
+ : []),
1039
+ ].map((value) => textValue(value) ?? "").join("\n");
1040
+ const declaredCycles = [...scopeText.matchAll(/(?:after\s+(?:cycle\s*)?|\b)([0-9][0-9,_]*)\s*(?:total\s+)?cycles?\b/gi)]
1041
+ .map((match) => Number(match[1].replaceAll(/[,_]/g, "")))
1042
+ .filter(Number.isFinite);
1043
+ if (!declaredCycles.length) return [];
1044
+ const expected = Math.max(...declaredCycles);
1045
+ const loopCycles = [...entrypoint.matchAll(/\brange\s*\(\s*([0-9][0-9,_]*)\s*\)/gi)]
1046
+ .map((match) => Number(match[1].replaceAll(/[,_]/g, "")))
1047
+ .filter(Number.isFinite);
1048
+ const offByOne = loopCycles.find((value) => value === expected - 1);
1049
+ return Number.isFinite(offByOne)
1050
+ ? [{
1051
+ experimentId: experiment?.id ?? "unnamed-experiment",
1052
+ declaredCycles: expected,
1053
+ loopCycles: offByOne,
1054
+ }]
1055
+ : [];
1056
+ });
1057
+ }
1058
+
1059
+ function repeatedCycleFinalComparisonFailures(experiments, reportedMeasurements) {
1060
+ return experiments.flatMap((experiment) => {
1061
+ const bindings = Array.isArray(experiment?.measurements) ? experiment.measurements : [];
1062
+ const finalCycleMeasurements = bindings.flatMap((measurement) => {
1063
+ const reported = reportedMeasurements.get(measurement?.reportedMeasurementId);
1064
+ const cycles = Number(reported?.dimensions?.cycles);
1065
+ const comparison = textValue(reported?.dimensions?.comparison) ?? "";
1066
+ return Number.isInteger(cycles) && cycles > 1 && /(?:first encoding|initial|baseline)/i.test(comparison)
1067
+ ? [{ cycles }]
1068
+ : [];
1069
+ });
1070
+ if (!finalCycleMeasurements.length) return [];
1071
+ const sources = inlinePythonExecutableSources(experiment?.protocol?.entrypoint);
1072
+ return finalCycleMeasurements.flatMap(({ cycles }) => {
1073
+ const cycleLoop = new RegExp(`\\bfor\\s+[A-Za-z_][A-Za-z0-9_]*\\s+in\\s+range\\s*\\(\\s*${cycles}\\s*\\)\\s*:[\\s\\S]+`, "i");
1074
+ const invalid = sources.some((source) => {
1075
+ const loop = source.match(cycleLoop)?.[0] ?? "";
1076
+ if (!loop) return false;
1077
+ const baselineInsideLoop = /\bif\s+[A-Za-z_][A-Za-z0-9_]*\s+is\s+None\s*:[^\n;]*\b[A-Za-z_][A-Za-z0-9_]*\s*=/i.test(loop);
1078
+ const aggregatesIntermediateComparisons = /\b(?:matched|matches|agreement|equal_count|total)\s*\+=/i.test(loop)
1079
+ && /\b(?:matched|matches|agreement|equal_count)\s*\/\s*(?:total|cycles|count)\b/i.test(source);
1080
+ return baselineInsideLoop || aggregatesIntermediateComparisons;
1081
+ });
1082
+ return invalid
1083
+ ? [{ experimentId: experiment?.id ?? "unnamed-experiment", cycles }]
1084
+ : [];
1085
+ });
1086
+ });
1087
+ }
1088
+
1089
+ function siSnrFormulaFailures(experiments) {
1090
+ return experiments.filter((experiment) => {
1091
+ const usesSiSnr = (Array.isArray(experiment?.measurements)
1092
+ ? experiment.measurements
1093
+ : []).some((measurement) => /^(?:si[_ -]?snr|scale[_ -]?invariant[_ -]?(?:snr|signal[_ -]?noise[_ -]?ratio))$/i
1094
+ .test(textValue(measurement?.metric) ?? ""));
1095
+ if (!usesSiSnr) return false;
1096
+ const sources = inlinePythonExecutableSources(experiment?.protocol?.entrypoint);
1097
+ if (sources.some((source) =>
1098
+ /\b(?:[A-Za-z_][A-Za-z0-9_]*_)?(?:si_snr|sisnr|scale_invariant_signal_noise_ratio)\s*\(/i.test(source))) return false;
1099
+ const decodedSources = sources.flatMap((source) => {
1100
+ const decoded = source.match(/\b([A-Za-z_][A-Za-z0-9_]*)\s*=\s*[^;\n]{0,200}?\.decode\s*\(\s*[^;\n]*?\.encode\s*\(\s*([A-Za-z_][A-Za-z0-9_]*)\s*\)/i);
1101
+ return decoded ? [{ source, decoded }] : [];
1102
+ });
1103
+ if (!decodedSources.length) return false;
1104
+ return !decodedSources.some(({ source, decoded }) => {
1105
+ const [, estimate, reference] = decoded;
1106
+ const crossProduct = new RegExp(`(?:\\b${escapeRegExp(estimate)}\\s*\\*\\s*${escapeRegExp(reference)}\\b|\\b${escapeRegExp(reference)}\\s*\\*\\s*${escapeRegExp(estimate)}\\b)`);
1107
+ return crossProduct.test(source);
1108
+ });
1109
+ });
1110
+ }
1111
+
1112
+ function siSnrInvariantPreflightFailures(experiments) {
1113
+ return experiments.filter((experiment) => {
1114
+ const usesSiSnr = (Array.isArray(experiment?.measurements)
1115
+ ? experiment.measurements
1116
+ : []).some((measurement) => /^(?:si[_ -]?snr|scale[_ -]?invariant[_ -]?(?:snr|signal[_ -]?noise[_ -]?ratio))$/i
1117
+ .test(textValue(measurement?.metric) ?? ""));
1118
+ if (!usesSiSnr) return false;
1119
+ const checks = Array.isArray(experiment?.protocol?.preflight?.checks)
1120
+ ? experiment.protocol.preflight.checks
1121
+ : [];
1122
+ const smoke = checks.find((check) => check?.kind === "smoke_test");
1123
+ const sources = inlinePythonExecutableSources(smoke?.command);
1124
+ const entrypointSources = inlinePythonExecutableSources(experiment?.protocol?.entrypoint);
1125
+ const hasNonUnitScale = sources.some((source) => {
1126
+ const scaledAssignments = [...source.matchAll(
1127
+ /\b(?:[A-Za-z_][A-Za-z0-9_]*_)?(?:scaled_(?:estimate|est)|(?:estimate|est)_scaled)\s*=\s*([^;\n]+)/gi,
1128
+ )];
1129
+ return scaledAssignments.some(([, expression]) => {
1130
+ const factors = [
1131
+ ...expression.matchAll(/(?:^|[^A-Za-z0-9_.])([+-]?(?:\d+(?:\.\d*)?|\.\d+))\s*\*\s*(?:[A-Za-z_]|\()/g),
1132
+ ...expression.matchAll(/(?:[A-Za-z_][A-Za-z0-9_]*|\))\s*\*\s*([+-]?(?:\d+(?:\.\d*)?|\.\d+))(?![A-Za-z0-9_.])/g),
1133
+ ].map((match) => Number(match[1]));
1134
+ return factors.some((factor) =>
1135
+ Number.isFinite(factor)
1136
+ && Math.abs(factor) > 1e-12
1137
+ && Math.abs(Math.abs(factor) - 1) > 1e-12);
1138
+ });
1139
+ });
1140
+ const hasToleranceAssertion = sources.some((source) =>
1141
+ /\b(?:allclose|isclose)\s*\(|\bassert\b[^\n;]*(?:abs\s*\(|<\s*1e-|<=\s*1e-)/i.test(source));
1142
+ const evaluatorNames = new Set(entrypointSources.flatMap((source) =>
1143
+ [...source.matchAll(/\b((?:[A-Za-z_][A-Za-z0-9_]*_)?(?:si_snr|sisnr|scale_invariant_signal_noise_ratio))\s*\(/gi)]
1144
+ .map((match) => match[1].toLowerCase())));
1145
+ const reusesEvaluatorTwice = [...evaluatorNames].some((name) => {
1146
+ const calls = sources.reduce((total, source) =>
1147
+ total + [...source.matchAll(new RegExp(`\\b${escapeRegExp(name)}\\s*\\(`, "gi"))].length
1148
+ - [...source.matchAll(new RegExp(`\\bdef\\s+${escapeRegExp(name)}\\s*\\(`, "gi"))].length,
1149
+ 0);
1150
+ return calls >= 2;
1151
+ });
1152
+ const namesScaledEstimate = sources.some((source) =>
1153
+ /\b(?:[A-Za-z_][A-Za-z0-9_]*_)?(?:scaled_(?:estimate|est)|(?:estimate|est)_scaled)\s*=/i.test(source));
1154
+ const obviousDegeneratePair = sources.some((source) =>
1155
+ /\b(?:estimate|est)\s*=\s*(?:reference|ref)\s*\*\s*[0-9.]+\b/i.test(source)
1156
+ && !/(?:\bnoise\b|\bperturb|torch\.tensor\s*\()/i.test(source));
1157
+ return !hasNonUnitScale
1158
+ || !hasToleranceAssertion
1159
+ || !reusesEvaluatorTwice
1160
+ || !namesScaledEstimate
1161
+ || obviousDegeneratePair;
1162
+ });
1163
+ }
1164
+
1165
+ function siSnrEntrypointInvariantFixtureFailures(experiments) {
1166
+ return experiments.filter((experiment) => {
1167
+ const usesSiSnr = (Array.isArray(experiment?.measurements)
1168
+ ? experiment.measurements
1169
+ : []).some((measurement) => /^(?:si[_ -]?snr|scale[_ -]?invariant[_ -]?(?:snr|signal[_ -]?noise[_ -]?ratio))$/i
1170
+ .test(textValue(measurement?.metric) ?? ""));
1171
+ if (!usesSiSnr) return false;
1172
+ const entrypoint = textValue(experiment?.protocol?.entrypoint) ?? "";
1173
+ const fragments = Array.isArray(experiment?.protocol?.requiredCommandFragments)
1174
+ ? experiment.protocol.requiredCommandFragments.map((fragment) => textValue(fragment) ?? "")
1175
+ : [];
1176
+ const namesScaledEstimate = /\b(?:[A-Za-z_][A-Za-z0-9_]*_)?(?:scaled_(?:estimate|est)|(?:estimate|est)_scaled)\b/i;
1177
+ const embedsAnalyticalAssertion = namesScaledEstimate.test(entrypoint)
1178
+ && /\b(?:allclose|isclose)\s*\(|\bassert\b[^\n;]*(?:abs\s*\(|<\s*1e-|<=\s*1e-)/i.test(entrypoint);
1179
+ return embedsAnalyticalAssertion || fragments.some((fragment) => namesScaledEstimate.test(fragment));
1180
+ });
1181
+ }
1182
+
1183
+ export function compilerReconstructionIssueFamily(issue) {
1184
+ const value = typeof issue === "string" ? issue.trim() : "";
1185
+ const separator = value.indexOf(":");
1186
+ return separator >= 0 ? value.slice(0, separator) : value;
1187
+ }
1188
+
1189
+ export function novelCompilerReconstructionIssues(issues, seenFamilies = new Set()) {
1190
+ return (Array.isArray(issues) ? issues : []).filter((issue) =>
1191
+ !seenFamilies.has(compilerReconstructionIssueFamily(issue)),
1192
+ );
1193
+ }
1194
+
1195
+ function officialEntrypointIdentityFailures(experiments, context) {
1196
+ if (!Array.isArray(context?.repositoryFiles)) return [];
1197
+ const repositoryFiles = new Set(
1198
+ context.repositoryFiles
1199
+ .map(normalizeRepositoryPath)
1200
+ .filter(Boolean),
1201
+ );
1202
+ if (!repositoryFiles.size) return [];
1203
+ const filesByBasename = new Map();
1204
+ for (const file of repositoryFiles) {
1205
+ const basename = path.posix.basename(file);
1206
+ const matches = filesByBasename.get(basename) ?? [];
1207
+ matches.push(file);
1208
+ filesByBasename.set(basename, matches);
1209
+ }
1210
+ return experiments.flatMap((experiment) => {
1211
+ if (experiment?.implementationOrigin !== "official") return [];
1212
+ const references = officialContractScriptReferences(experiment);
1213
+ if (!references.length) return [];
1214
+ const missing = references.filter((reference) => {
1215
+ const logical = normalizeRepositoryPath(reference);
1216
+ if (!logical) return true;
1217
+ if (repositoryFiles.has(logical)) return false;
1218
+ if (logical.includes("/")) return true;
1219
+ return (filesByBasename.get(logical) ?? []).length !== 1;
1220
+ });
1221
+ return missing.length
1222
+ ? [{ experimentId: experiment.id ?? "unnamed-experiment", references: missing }]
1223
+ : [];
1224
+ });
1225
+ }
1226
+
1227
+ function officialContractScriptReferences(experiment) {
1228
+ const protocol = experiment?.protocol ?? {};
1229
+ const commandSources = [
1230
+ protocol.entrypoint,
1231
+ ...(Array.isArray(protocol.requiredCommandFragments)
1232
+ ? protocol.requiredCommandFragments
1233
+ : []),
1234
+ ...(Array.isArray(protocol.preflight?.checks)
1235
+ ? protocol.preflight.checks.map((check) => check?.command)
1236
+ : []),
1237
+ ];
1238
+ return [...new Set(commandSources.flatMap(entrypointScriptReferences))];
1239
+ }
1240
+
1241
+ function officialCliInterfaceFailures(experiments) {
1242
+ return experiments.flatMap((experiment) => {
1243
+ if (experiment?.implementationOrigin !== "official") return [];
1244
+ const entrypoint = textValue(experiment?.protocol?.entrypoint) ?? "";
1245
+ const invocations = pythonRepositoryScriptInvocations(entrypoint);
1246
+ if (!invocations.length) return [];
1247
+ const entrypointChecks = (Array.isArray(experiment?.protocol?.preflight?.checks)
1248
+ ? experiment.protocol.preflight.checks
1249
+ : []).filter((check) => check?.kind === "entrypoint");
1250
+ return invocations.flatMap(({ script, options }) => {
1251
+ if (!options.length) return [];
1252
+ const verified = entrypointChecks.some((check) => {
1253
+ const command = textValue(check?.command) ?? "";
1254
+ return command.includes(script)
1255
+ && /(?:^|\s)--help(?:\s|$)/.test(command)
1256
+ && options.every((option) => command.includes(option));
1257
+ });
1258
+ return verified
1259
+ ? []
1260
+ : [{ experimentId: experiment.id ?? "unnamed-experiment", script, options }];
1261
+ });
1262
+ });
1263
+ }
1264
+
1265
+ function pythonRepositoryScriptInvocations(command) {
1266
+ const value = textValue(command);
1267
+ if (!value) return [];
1268
+ const pattern = /(?:^|[;&|]\s*)python(?:3(?:\.\d+)?)?\s+(\/job\/workspace\/repository\/[A-Za-z0-9._/-]+\.py)\b([^;&|]*)/gi;
1269
+ return [...value.matchAll(pattern)].map((match) => ({
1270
+ script: match[1],
1271
+ options: [...new Set([...match[2].matchAll(/(?:^|\s)(--[A-Za-z0-9][A-Za-z0-9_-]*)\b/g)]
1272
+ .map((option) => option[1]))],
1273
+ }));
1274
+ }
1275
+
1276
+ function experimentUsesUnavailableDataRoot(experiment) {
1277
+ const protocol = experiment?.protocol ?? {};
1278
+ const commands = [
1279
+ protocol.entrypoint,
1280
+ ...(Array.isArray(protocol.preflight?.checks)
1281
+ ? protocol.preflight.checks.map((check) => check?.command)
1282
+ : []),
1283
+ ];
1284
+ return commands.some((command) => /(?:^|[\s"'=:(])\/data(?:\/|\b)/.test(textValue(command) ?? ""));
1285
+ }
1286
+
1287
+ function parserPreflightFixtureFailures(experiments, reportedMeasurements) {
1288
+ return experiments.flatMap((experiment) => {
1289
+ const checks = (Array.isArray(experiment?.protocol?.preflight?.checks)
1290
+ ? experiment.protocol.preflight.checks
1291
+ : []).filter((check) => check?.kind === "parser" && textValue(check?.command));
1292
+ if (!checks.length) return [];
1293
+ const reasons = new Set();
1294
+ for (const check of checks) {
1295
+ const command = textValue(check.command) ?? "";
1296
+ const mentionsFixture = /(?:fixture|parser[-_]?sample)/i.test(command);
1297
+ const createsFixture = /(?:write_text|write_bytes|json\.dump\s*\(|open\s*\([^)]*,\s*["']w|(?:printf|echo)\b[^;&|]*(?:>|tee\b))/i
1298
+ .test(command);
1299
+ if (mentionsFixture && !createsFixture) reasons.add("reads a fixture it never creates");
1300
+ const bindings = (Array.isArray(experiment?.measurements)
1301
+ ? experiment.measurements
1302
+ : []).map((measurement) => ({
1303
+ measurement,
1304
+ reported: reportedMeasurements.get(measurement?.reportedMeasurementId),
1305
+ }));
1306
+ if (parserPreflightTargetLeaks(command, bindings).length) {
1307
+ reasons.add("embeds the paper-reported target in a blind parser check");
1308
+ }
1309
+ }
1310
+ return reasons.size
1311
+ ? [{ experimentId: experiment.id ?? "unnamed-experiment", reasons: [...reasons] }]
1312
+ : [];
1313
+ });
1314
+ }
1315
+
1316
+ function reconstructionOriginIsolationFailures(experiments, context) {
1317
+ if (!Array.isArray(context?.repositoryFiles)) return [];
1318
+ const repositoryFiles = new Set(
1319
+ context.repositoryFiles
1320
+ .map(normalizeRepositoryPath)
1321
+ .filter(Boolean),
1322
+ );
1323
+ if (!repositoryFiles.size) return [];
1324
+ const filesByBasename = new Map();
1325
+ for (const file of repositoryFiles) {
1326
+ const basename = path.posix.basename(file);
1327
+ const matches = filesByBasename.get(basename) ?? [];
1328
+ matches.push(file);
1329
+ filesByBasename.set(basename, matches);
1330
+ }
1331
+ return experiments.flatMap((experiment) => {
1332
+ if (experiment?.implementationOrigin !== "citeark_reconstruction") return [];
1333
+ const protocol = experiment?.protocol ?? {};
1334
+ const references = [...new Set([
1335
+ protocol.entrypoint,
1336
+ protocol.instructions,
1337
+ ...(Array.isArray(protocol.requiredCommandFragments)
1338
+ ? protocol.requiredCommandFragments
1339
+ : []),
1340
+ ...(Array.isArray(protocol.preflight?.checks)
1341
+ ? protocol.preflight.checks.flatMap((check) => [check?.command, check?.successCriterion])
1342
+ : []),
1343
+ ].flatMap(entrypointScriptReferences))];
1344
+ const leaked = references.filter((reference) => {
1345
+ const logical = normalizeRepositoryPath(reference);
1346
+ if (logical && repositoryFiles.has(logical)) return true;
1347
+ const basename = path.posix.basename(reference.replaceAll("\\", "/"));
1348
+ return (filesByBasename.get(basename) ?? []).length === 1;
1349
+ });
1350
+ return leaked.length
1351
+ ? [{ experimentId: experiment.id ?? "unnamed-experiment", references: leaked }]
1352
+ : [];
1353
+ });
1354
+ }
1355
+
1356
+ function entrypointScriptReferences(value) {
1357
+ const command = textValue(value);
1358
+ if (!command) return [];
1359
+ const scriptPattern = /\/job\/workspace\/repository\/[A-Za-z0-9._/-]+\.(?:py|sh|js|mjs|cjs|rb|pl|jl|r)\b|(?:\.{1,2}\/)?[A-Za-z0-9_.-]+(?:\/[A-Za-z0-9_.-]+)*\.(?:py|sh|js|mjs|cjs|rb|pl|jl|r)\b/gi;
1360
+ return [...new Set([...command.matchAll(scriptPattern)].map((match) => match[0]))];
1361
+ }
1362
+
1363
+ function normalizeRepositoryPath(value) {
1364
+ if (typeof value !== "string" || !value.trim()) return null;
1365
+ let normalized = value.trim().replaceAll("\\", "/");
1366
+ const repositoryPrefix = "/job/workspace/repository/";
1367
+ if (normalized.startsWith(repositoryPrefix)) normalized = normalized.slice(repositoryPrefix.length);
1368
+ else if (normalized.startsWith("/")) return null;
1369
+ normalized = path.posix.normalize(normalized);
1370
+ if (normalized.startsWith("./")) normalized = normalized.slice(2);
1371
+ if (
1372
+ normalized === "."
1373
+ || normalized === ".."
1374
+ || normalized.startsWith("../")
1375
+ || normalized.includes("\0")
1376
+ ) return null;
1377
+ return normalized;
1378
+ }
1379
+
1380
+ function experimentReportedModelVariants(research, experiment) {
1381
+ const boundMeasurementIds = new Set(
1382
+ (Array.isArray(experiment?.measurements) ? experiment.measurements : [])
1383
+ .map((measurement) => measurement?.reportedMeasurementId)
1384
+ .filter(Boolean),
1385
+ );
1386
+ const variants = new Set();
1387
+ for (const claim of (Array.isArray(research?.claims) ? research.claims : [])) {
1388
+ for (const measurement of (Array.isArray(claim?.reportedMeasurements) ? claim.reportedMeasurements : [])) {
1389
+ if (!boundMeasurementIds.has(measurement?.id)) continue;
1390
+ const variant = textValue(measurement?.dimensions?.model_variant);
1391
+ if (variant) variants.add(variant);
1392
+ }
1393
+ }
1394
+ return [...variants];
1395
+ }
1396
+
1397
+ function explicitlyDisclaimsReportedScope(experiment) {
1398
+ const description = [experiment?.title, experiment?.protocol?.instructions]
1399
+ .map(textValue)
1400
+ .filter(Boolean)
1401
+ .join(" ");
1402
+ const statements = description.replace(
1403
+ /\bnot\s+(?:(?:a|the)\s+)?(?:(?:reduced|mere|only)\s+)?(?:smoke|plumbing)(?:[- ]shaped)?\s+(?:evaluation|experiment|input|check|orchestration|command)\b/gi,
1404
+ "",
1405
+ );
1406
+ return /\b(?:smoke|plumbing)(?:[- ]shaped)? (?:evaluation|experiment|input|check|orchestration|command)\b/i.test(statements)
1407
+ || /\bnot equivalent to (?:the )?paper(?:'s)?\b/i.test(description)
1408
+ || /\bnot (?:the )?paper(?:'s)?(?: (?:full|reported|report-defining))? (?:evaluation|experiment|protocol|scope)\b/i.test(description)
1409
+ || /\bdo not promote .{0,120}\b(?:scope|measurement|metric)\b/i.test(description)
1410
+ || /\b(?:entrypoint|command).{0,120}\b(?:example|placeholder)\b/i.test(description)
1411
+ || /\bmust be replaced by (?:the )?execution agent\b/i.test(description);
1412
+ }
1413
+
1414
+ function entrypointEmbedsReportedValue(entrypointValue, measurement, reported) {
1415
+ const entrypoint = textValue(entrypointValue);
1416
+ const metric = textValue(measurement?.metric) ?? textValue(reported?.metric);
1417
+ if (!entrypoint || !metric || typeof reported?.value !== "number" || !Number.isFinite(reported.value)) {
1418
+ return false;
1419
+ }
1420
+ const candidates = [reported.value];
1421
+ const scale = Number(measurement?.parser?.config?.scale ?? 1);
1422
+ const offset = Number(measurement?.parser?.config?.offset ?? 0);
1423
+ if (Number.isFinite(scale) && scale !== 0 && Number.isFinite(offset)) {
1424
+ candidates.push((reported.value - offset) / scale);
1425
+ }
1426
+ const key = escapeRegExp(metric);
1427
+ return [...new Set(candidates.filter(Number.isFinite))].some((value) => {
1428
+ const literal = numericLiteralPattern(value);
1429
+ const match = new RegExp(
1430
+ `["']?${key}["']?\\s*[:=]\\s*${literal}(?![0-9.])`,
1431
+ "i",
1432
+ ).exec(entrypoint);
1433
+ if (!match) return false;
1434
+ const suffix = entrypoint.slice(match.index + match[0].length);
1435
+ if (/^[ \t]*[*/+%-]/.test(suffix)) return false;
1436
+ if (runtimeConditionalLiteral(suffix, value)) return false;
1437
+ return true;
1438
+ });
1439
+ }
1440
+
1441
+ function entrypointIsProsePlaceholder(entrypointValue) {
1442
+ const entrypoint = textValue(entrypointValue);
1443
+ if (!entrypoint) return false;
1444
+ const explicitPlaceholder = /\b(?:todo|placeholder|replace (?:this|me)|execution agent (?:must|should|will) (?:implement|replace|write))\b/i
1445
+ .test(entrypoint);
1446
+ if (explicitPlaceholder) return true;
1447
+
1448
+ const pythonInline = /\bpython(?:3)?\s+-c\s+(["'])([\s\S]*)\1\s*$/i.exec(entrypoint);
1449
+ if (pythonInline) {
1450
+ const code = pythonInline[2];
1451
+ const calls = [...code.matchAll(/(?:^|[^.A-Za-z0-9_])([A-Za-z_][A-Za-z0-9_]*)\s*\(/g)]
1452
+ .map((match) => match[1].toLowerCase());
1453
+ const meaningfulCalls = calls.filter((name) => !new Set(["print"]).has(name));
1454
+ return meaningfulCalls.length === 0
1455
+ && /\b(?:execute|evaluate|run|write|create|generate|compute)\b[\s\S]{0,240}\b(?:driver|model|dataset|corpus|files?|clips?|cycles?|evidence|output|json|metric|aggregate|evaluation)\b/i
1456
+ .test(code);
1457
+ }
1458
+
1459
+ return /^\s*(?:echo|printf)\b[\s\S]*\b(?:execute|evaluate|run|write|create|generate|compute)\b[\s\S]{0,240}\b(?:driver|model|dataset|corpus|files?|clips?|cycles?|evidence|output|json|metric|aggregate|evaluation)\b/i
1460
+ .test(entrypoint);
1461
+ }
1462
+
1463
+ function simpleEntrypointReference(entrypoint) {
1464
+ return isSimpleScriptEntrypoint(entrypoint);
1465
+ }
1466
+
1467
+ function runtimeConditionalLiteral(suffix, reportedValue) {
1468
+ const conditional = /^[ \t]+if[ \t]+(?!true\b|false\b|none\b)([^\n;]{1,160}?)[ \t]+else[ \t]+([+-]?(?:\d+(?:\.\d*)?|\.\d+))/i
1469
+ .exec(suffix);
1470
+ if (!conditional) return false;
1471
+ const condition = conditional[1].trim();
1472
+ const alternative = Number(conditional[2]);
1473
+ return /[A-Za-z_]/.test(condition)
1474
+ && Number.isFinite(alternative)
1475
+ && alternative !== reportedValue;
1476
+ }
1477
+
1478
+ function numericLiteralPattern(value) {
1479
+ if (Number.isInteger(value)) return `${escapeRegExp(String(value))}(?:\\.0+)?`;
1480
+ return escapeRegExp(String(value));
1481
+ }
1482
+
1483
+ function escapeRegExp(value) {
1484
+ return value.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
1485
+ }
1486
+
1487
+ function compilerWork(value, context, warnings) {
1488
+ const work = isRecord(value) ? { ...value } : {};
1489
+ if (!isRecord(value)) warnings.push("Missing work object was reconstructed from compiler inputs.");
1490
+ const fallbackTitle = textValue(context.paperTitle) ?? textValue(context.taskName) ?? "Untitled research work";
1491
+ const title = textValue(work.title) ?? fallbackTitle;
1492
+ if (!textValue(work.title)) warnings.push("Missing work.title was filled from compiler inputs.");
1493
+ const id = textValue(work.id) ?? `work-${sha256Value({ title }).slice(0, 16)}`;
1494
+ if (!textValue(work.id)) warnings.push("Missing work.id was deterministically derived from the title.");
1495
+ const abstract = textValue(work.abstract)
1496
+ ?? "The compiler did not provide an abstract; consult the fixed paper source for the canonical abstract.";
1497
+ if (!textValue(work.abstract)) warnings.push("Missing work.abstract was replaced by an explicit source-reference placeholder.");
1498
+ const significance = textValue(work.significance);
1499
+ if (!significance) {
1500
+ throw new CiteArkError("work.significance 必须解释论文意味着什么、谁可能受益以及价值边界");
1501
+ }
1502
+ const authors = Array.isArray(work.authors)
1503
+ ? work.authors.map(textValue).filter(Boolean).slice(0, 500)
1504
+ : [];
1505
+ if (!Array.isArray(work.authors)) warnings.push("Missing or invalid work.authors was normalized to an empty array.");
1506
+ const subjects = Array.isArray(work.subjects)
1507
+ ? [...new Set(work.subjects.filter((subject) => SUBJECT_TAGS.has(subject)))]
1508
+ : [];
1509
+ if (Array.isArray(work.subjects) && subjects.length !== work.subjects.length) {
1510
+ warnings.push("Unknown subject tags were omitted instead of failing compilation.");
1511
+ }
1512
+ return { ...work, id, title, abstract, significance, authors, subjects };
1513
+ }
1514
+
1515
+ function compilerSources(value, context, warnings) {
1516
+ const sources = [];
1517
+ const usedIds = new Set();
1518
+ for (const [index, candidate] of (Array.isArray(value) ? value : []).entries()) {
1519
+ if (!isRecord(candidate) || !SOURCE_KINDS.has(candidate.kind) || !textValue(candidate.uri)) {
1520
+ warnings.push(`Invalid sources[${index}] was omitted.`);
1521
+ continue;
1522
+ }
1523
+ if (candidate.kind === "repository") {
1524
+ warnings.push(`sources[${index}] repository identity was replaced by the platform trust boundary.`);
1525
+ continue;
1526
+ }
1527
+ const baseId = textValue(candidate.id) ?? `source-${candidate.kind}-${index + 1}`;
1528
+ const id = uniqueCompilerId(baseId, usedIds);
1529
+ if (id !== candidate.id) warnings.push(`sources[${index}].id was missing or duplicated and was normalized to ${id}.`);
1530
+ usedIds.add(id);
1531
+ sources.push({ ...candidate, id, uri: candidate.uri.trim() });
1532
+ }
1533
+ if (!sources.some((source) => source.kind === "paper")) {
1534
+ const uri = textValue(context.paperUri)
1535
+ ?? (textValue(context.paperDigest) ? `citeark://paper/${context.paperDigest}` : "citeark://paper/fixed-input");
1536
+ const id = uniqueCompilerId("paper", usedIds);
1537
+ usedIds.add(id);
1538
+ sources.unshift({ id, kind: "paper", uri, ...(context.paperDigest ? { digest: context.paperDigest } : {}) });
1539
+ warnings.push("Missing paper source was reconstructed from the fixed compiler input.");
1540
+ }
1541
+ const repositoryUrl = textValue(context.repositoryIdentity?.url);
1542
+ const repositoryCommit = textValue(context.repositoryIdentity?.commit);
1543
+ if (repositoryUrl && repositoryCommit) {
1544
+ const id = uniqueCompilerId("repository", usedIds);
1545
+ sources.push({
1546
+ id,
1547
+ kind: "repository",
1548
+ uri: repositoryUrl,
1549
+ version: repositoryCommit,
1550
+ ...(context.repositoryIdentity.snapshotDigest ? { digest: context.repositoryIdentity.snapshotDigest } : {}),
1551
+ });
1552
+ warnings.push("Repository source was reconstructed from the fixed platform-verified identity.");
1553
+ }
1554
+ return sources;
1555
+ }
1556
+
1557
+ function compilerClaims(value, { sourceIds, paperSourceId, warnings }) {
1558
+ const claims = [];
1559
+ const usedClaimIds = new Set();
1560
+ const usedMeasurementIds = new Set();
1561
+ for (const [index, candidate] of (Array.isArray(value) ? value : []).entries()) {
1562
+ if (!isRecord(candidate) || !textValue(candidate.statement)) {
1563
+ warnings.push(`Invalid claims[${index}] without a statement was omitted.`);
1564
+ continue;
1565
+ }
1566
+ const baseId = textValue(candidate.id) ?? `claim-${index + 1}`;
1567
+ const id = uniqueCompilerId(baseId, usedClaimIds);
1568
+ usedClaimIds.add(id);
1569
+ if (id !== candidate.id) warnings.push(`claims[${index}].id was missing or duplicated and was normalized to ${id}.`);
1570
+ const type = CLAIM_TYPES.has(candidate.type) ? candidate.type : "finding";
1571
+ if (type !== candidate.type) warnings.push(`claims[${index}].type was normalized to finding.`);
1572
+ const sourceLocator = compilerLocator(candidate.sourceLocator, {
1573
+ sourceIds,
1574
+ fallbackSourceId: paperSourceId,
1575
+ fallbackLocator: "Location not specified by the compiler",
1576
+ warning: `claims[${index}].sourceLocator was repaired against the fixed paper source.`,
1577
+ warnings,
1578
+ });
1579
+ const measurements = [];
1580
+ for (const [measurementIndex, measurement] of (
1581
+ Array.isArray(candidate.reportedMeasurements) ? candidate.reportedMeasurements : []
1582
+ ).entries()) {
1583
+ if (
1584
+ !isRecord(measurement)
1585
+ || !textValue(measurement.metric)
1586
+ || !textValue(measurement.unit)
1587
+ || typeof measurement.value !== "number"
1588
+ || !Number.isFinite(measurement.value)
1589
+ ) {
1590
+ warnings.push(`Invalid reported measurement at claims[${index}].reportedMeasurements[${measurementIndex}] was omitted.`);
1591
+ continue;
1592
+ }
1593
+ const measurementBaseId = textValue(measurement.id) ?? `${id}-measurement-${measurementIndex + 1}`;
1594
+ const measurementId = uniqueCompilerId(measurementBaseId, usedMeasurementIds);
1595
+ usedMeasurementIds.add(measurementId);
1596
+ if (measurementId !== measurement.id) {
1597
+ warnings.push(`A missing or duplicated reported measurement id was normalized to ${measurementId}.`);
1598
+ }
1599
+ measurements.push({
1600
+ ...measurement,
1601
+ id: measurementId,
1602
+ metric: measurement.metric.trim(),
1603
+ unit: measurement.unit.trim(),
1604
+ sourceLocator: compilerLocator(measurement.sourceLocator, {
1605
+ sourceIds,
1606
+ fallbackSourceId: sourceLocator.sourceId,
1607
+ fallbackLocator: sourceLocator.locator,
1608
+ warning: `Measurement ${measurementId} inherited its claim source locator.`,
1609
+ warnings,
1610
+ }),
1611
+ });
1612
+ }
1613
+ claims.push({
1614
+ ...candidate,
1615
+ id,
1616
+ statement: candidate.statement.trim(),
1617
+ type,
1618
+ sourceLocator,
1619
+ reportedMeasurements: measurements,
1620
+ reproduction: isRecord(candidate.reproduction) ? { ...candidate.reproduction } : {},
1621
+ });
1622
+ }
1623
+ return claims;
1624
+ }
1625
+
1626
+ function compilerExperiments(value, {
1627
+ claims,
1628
+ researchObjects,
1629
+ repositoryIdentity,
1630
+ executionAllowed,
1631
+ reproductionScope,
1632
+ continuation,
1633
+ warnings,
1634
+ }) {
1635
+ const repositoryUrl = textValue(repositoryIdentity?.url);
1636
+ const repositoryCommit = textValue(repositoryIdentity?.commit);
1637
+ const hasOfficialRepository = Boolean(repositoryUrl && repositoryCommit);
1638
+ if (!executionAllowed && Array.isArray(value) && value.length) {
1639
+ warnings.push("Executable experiments were omitted because the recorded license policy does not allow execution.");
1640
+ return [];
1641
+ }
1642
+ const experiments = [];
1643
+ const claimIds = new Set(claims.map((claim) => claim.id));
1644
+ const measurementOwners = new Map();
1645
+ const observationIds = new Set((researchObjects ?? []).filter(object => object.role === "observation").map(object => object.id));
1646
+ for (const claim of claims) {
1647
+ for (const measurement of claim.reportedMeasurements) measurementOwners.set(measurement.id, { claim, measurement });
1648
+ }
1649
+ const usedIds = new Set();
1650
+ const completedExperimentIds = new Set(
1651
+ (Array.isArray(continuation?.completedTargets) ? continuation.completedTargets : [])
1652
+ .map((target) => target?.experimentId)
1653
+ .filter(Boolean),
1654
+ );
1655
+ for (const [index, candidate] of (Array.isArray(value) ? value : []).entries()) {
1656
+ if (!isRecord(candidate)) {
1657
+ warnings.push(`Invalid experiments[${index}] was omitted.`);
1658
+ continue;
1659
+ }
1660
+ if (Array.isArray(candidate.protocol?.measurements)) {
1661
+ if (candidate.measurements !== undefined
1662
+ && sha256Value(candidate.measurements) !== sha256Value(candidate.protocol.measurements)) {
1663
+ throw new CiteArkError(`Conflicting measurement bindings at experiments[${index}]: experiment and protocol disagree`);
1664
+ }
1665
+ candidate.measurements = candidate.protocol.measurements;
1666
+ delete candidate.protocol.measurements;
1667
+ warnings.push(`Experiment ${candidate.id ?? index} measurement bindings were relocated from protocol.measurements without changing their content.`);
1668
+ }
1669
+ const requestedReconstruction = candidate.implementationOrigin === "citeark_reconstruction";
1670
+ const implementationOrigin = requestedReconstruction
1671
+ ? "citeark_reconstruction"
1672
+ : hasOfficialRepository
1673
+ ? "official"
1674
+ : null;
1675
+ if (!implementationOrigin) {
1676
+ warnings.push(`experiments[${index}] was omitted because no official repository exists and it was not declared as a CiteArk independent reconstruction.`);
1677
+ continue;
1678
+ }
1679
+ const executionMode = EXECUTION_MODES.has(candidate.protocol?.executionMode)
1680
+ ? candidate.protocol.executionMode : "agent_planned";
1681
+ // This is a declarative marker, never an invented repository script. The
1682
+ // autonomous workspace chooses and records the actual command at runtime.
1683
+ const entrypoint = textValue(candidate.protocol?.entrypoint)
1684
+ ?? (executionMode === "agent_planned" && textValue(candidate.protocol?.objective) ? "agent_planned" : null);
1685
+ if (!entrypoint) {
1686
+ warnings.push(`experiments[${index}] without a real entrypoint was omitted.`);
1687
+ continue;
1688
+ }
1689
+ const id = uniqueCompilerId(textValue(candidate.id) ?? `experiment-${index + 1}`, usedIds);
1690
+ usedIds.add(id);
1691
+ const boundClaims = new Set(
1692
+ (Array.isArray(candidate.claimIds) ? candidate.claimIds : []).filter((claimId) => claimIds.has(claimId)),
1693
+ );
1694
+ const measurements = [];
1695
+ const measurementBindings = [];
1696
+ for (const [measurementIndex, measurement] of (
1697
+ Array.isArray(candidate.measurements) ? candidate.measurements : []
1698
+ ).entries()) {
1699
+ const owner = isRecord(measurement)
1700
+ ? measurementOwners.get(measurement.reportedMeasurementId)
1701
+ : null;
1702
+ if (!owner && observationIds.has(measurement?.reportedMeasurementId)) {
1703
+ throw new CiteArkError(`Experiment ${id} uses observation ${measurement.reportedMeasurementId} as a scalar reportedMeasurementId. Preserve its method with observationTargets and an unresolved decision_rule limitation; bind measurements only to actual reported numeric IDs.`);
1704
+ }
1705
+ if (!owner || (!isRecord(measurement.parser) && executionMode !== "agent_planned")) {
1706
+ warnings.push(`Invalid measurement binding at experiments[${index}].measurements[${measurementIndex}] was omitted.`);
1707
+ continue;
1708
+ }
1709
+ const provisionalParser = {
1710
+ id: "json.scalar", version: "1",
1711
+ evidencePath: `measurements/${sha256Value(measurement.reportedMeasurementId).slice(0, 16)}.json`,
1712
+ config: { pointer: "/value", metric: owner.measurement.metric, unit: owner.measurement.unit },
1713
+ };
1714
+ const parser = compilerEvidenceParser(measurement.parser ?? provisionalParser, {
1715
+ experimentId: id,
1716
+ measurementId: measurement.reportedMeasurementId,
1717
+ warnings,
1718
+ });
1719
+ const normalized = {
1720
+ ...measurement,
1721
+ metric: textValue(measurement.metric) ?? owner.measurement.metric,
1722
+ unit: textValue(measurement.unit) ?? owner.measurement.unit,
1723
+ parser,
1724
+ };
1725
+ let parserIssues = validateEvidenceParserDescriptor(normalized.parser, normalized);
1726
+ if (parserIssues.length > 0 && executionMode === "agent_planned") {
1727
+ // A partial runtime suggestion must not erase the scientific target.
1728
+ // Use the same provisional destination as an omitted parser; the
1729
+ // execution workspace must establish the real evidence binding.
1730
+ normalized.parser = provisionalParser;
1731
+ warnings.push(`Experiment ${id} measurement ${measurement.reportedMeasurementId} parser suggestion is provisional: ${parserIssues.join("; ")}. The execution workspace must resolve its real evidence source and parser.`);
1732
+ parserIssues = validateEvidenceParserDescriptor(normalized.parser, normalized);
1733
+ }
1734
+ if (
1735
+ normalized.metric !== owner.measurement.metric
1736
+ || normalized.unit !== owner.measurement.unit
1737
+ || parserIssues.length > 0
1738
+ ) {
1739
+ warnings.push(`Non-verifiable measurement binding ${measurement.reportedMeasurementId} in experiment ${id} was omitted.`);
1740
+ continue;
1741
+ }
1742
+ boundClaims.add(owner.claim.id);
1743
+ measurements.push(normalized);
1744
+ measurementBindings.push({
1745
+ claim: owner.claim,
1746
+ measurement: normalized,
1747
+ reported: owner.measurement,
1748
+ });
1749
+ }
1750
+ // Observation comparisons are planning targets, not scalar measurements.
1751
+ // Preserve their methods in the catalog; strict validation below checks
1752
+ // source identity and claim ownership instead of inventing numeric IDs.
1753
+ const observationTargets = Array.isArray(candidate.observationTargets) ? candidate.observationTargets : [];
1754
+ for (const target of observationTargets) {
1755
+ if (claimIds.has(target?.claimId)) boundClaims.add(target.claimId);
1756
+ }
1757
+ if ((!measurements.length && !observationTargets.length) || !boundClaims.size) {
1758
+ warnings.push(`Experiment ${id} had no verifiable measurement binding and was omitted.`);
1759
+ continue;
1760
+ }
1761
+ const hardwareProtocolUnderspecified = (
1762
+ measurementBindings.length > 0 && measurementBindings.every(({ measurement, reported }) =>
1763
+ hardwareSensitiveMetric(measurement?.metric ?? reported?.metric))
1764
+ && explicitlyUnderspecifiedHardwareProtocol(candidate, measurementBindings)
1765
+ );
1766
+ if (hardwareProtocolUnderspecified && executionMode !== "agent_planned") {
1767
+ for (const { claim } of measurementBindings) {
1768
+ claim.reproduction = {
1769
+ status: "blocked",
1770
+ implementationOrigin: "insufficient",
1771
+ blocker: { kind: "scientific_input", evidence: "The fixed-entrypoint protocol explicitly declares the original benchmark conditions unspecified in the fixed sources." },
1772
+ reason: "The original hardware-sensitive benchmark conditions are underspecified by the fixed paper and repository, so a strict comparable verdict is not available.",
1773
+ };
1774
+ }
1775
+ warnings.push(`Hardware-sensitive fixed-entrypoint experiment ${id} was omitted because its own protocol states that the paper does not specify the original benchmark conditions.`);
1776
+ continue;
1777
+ }
1778
+ const repository = implementationOrigin === "official"
1779
+ ? { implementationOrigin, url: repositoryUrl, commit: repositoryCommit }
1780
+ : { ...RECONSTRUCTION_REPOSITORY };
1781
+ const fragments = Array.isArray(candidate.protocol?.requiredCommandFragments)
1782
+ ? candidate.protocol.requiredCommandFragments.map(textValue).filter(Boolean)
1783
+ : [];
1784
+ const objective = textValue(candidate.protocol?.objective)
1785
+ ?? textValue(candidate.protocol?.instructions)
1786
+ ?? textValue(candidate.title)
1787
+ ?? `Reproduce the protocol-bound measurements for ${id}.`;
1788
+ const computeIssues = candidate.compute === undefined
1789
+ ? []
1790
+ : validateDeclaredComputeRequirement(candidate.compute, `experiment ${id}.compute`);
1791
+ if (computeIssues.length) warnings.push(`Invalid compute proposal for experiment ${id} was discarded so the scheduler can infer resources.`);
1792
+ let reconstructionFidelity = RECONSTRUCTION_FIDELITIES.has(candidate.reconstructionFidelity)
1793
+ ? candidate.reconstructionFidelity
1794
+ : implementationOrigin === "official"
1795
+ ? "faithful"
1796
+ : "approximate";
1797
+ if (hardwareProtocolUnderspecified && reconstructionFidelity !== "proxy") {
1798
+ reconstructionFidelity = "approximate";
1799
+ warnings.push(`Hardware-sensitive agent-planned experiment ${id} was retained as an approximate independent benchmark because the paper does not fully specify the original hardware protocol.`);
1800
+ }
1801
+ const preflight = compilerPreflight(candidate.protocol?.preflight, {
1802
+ experimentId: id,
1803
+ entrypoint,
1804
+ executionMode,
1805
+ measurements,
1806
+ measurementBindings,
1807
+ warnings,
1808
+ implementationOrigin,
1809
+ legacyCompleted: completedExperimentIds.has(id),
1810
+ });
1811
+ const environment = compilerExecutionEnvironment(candidate.environment, {
1812
+ compute: computeIssues.length ? undefined : candidate.compute,
1813
+ experimentId: id,
1814
+ reconstructionFidelity,
1815
+ warnings,
1816
+ });
1817
+ experiments.push({
1818
+ ...candidate,
1819
+ id,
1820
+ title: textValue(candidate.title) ?? `Execute ${entrypoint}`,
1821
+ claimIds: [...boundClaims],
1822
+ ...(reproductionScope ? { reproductionScope } : {}),
1823
+ reproductionLevel: REPRODUCTION_LEVELS.has(candidate.reproductionLevel)
1824
+ ? candidate.reproductionLevel
1825
+ : "directional",
1826
+ reconstructionFidelity,
1827
+ implementationOrigin,
1828
+ repository,
1829
+ protocol: {
1830
+ ...candidate.protocol,
1831
+ executionMode,
1832
+ objective,
1833
+ entrypoint,
1834
+ requiredCommandFragments: fragments.length
1835
+ ? fragments
1836
+ : executionMode === "fixed_entrypoint"
1837
+ ? [entrypoint]
1838
+ : [],
1839
+ preflight,
1840
+ },
1841
+ measurements,
1842
+ environment,
1843
+ ...(computeIssues.length ? { compute: undefined } : {}),
1844
+ });
1845
+ if (computeIssues.length) delete experiments.at(-1).compute;
1846
+ }
1847
+ return experiments;
1848
+ }
1849
+
1850
+ function compilerPreflight(value, {
1851
+ experimentId,
1852
+ entrypoint,
1853
+ executionMode,
1854
+ measurements,
1855
+ measurementBindings,
1856
+ warnings,
1857
+ implementationOrigin,
1858
+ legacyCompleted,
1859
+ }) {
1860
+ const provided = isRecord(value);
1861
+ const schemaVersion = legacyCompleted && !provided
1862
+ ? "0.1"
1863
+ : new Set(["0.1", "0.2"]).has(value?.schemaVersion)
1864
+ ? value.schemaVersion
1865
+ : CURRENT_PREFLIGHT_SCHEMA_VERSION;
1866
+ const maxMinutes = Number(value?.maxMinutes);
1867
+ const checks = [];
1868
+ const usedIds = new Set();
1869
+ for (const [index, candidate] of (Array.isArray(value?.checks) ? value.checks : []).entries()) {
1870
+ if (!isRecord(candidate) || !PREFLIGHT_CHECK_KINDS.has(candidate.kind)) continue;
1871
+ const id = uniqueCompilerId(textValue(candidate.id) ?? `preflight-${candidate.kind}-${index + 1}`, usedIds);
1872
+ usedIds.add(id);
1873
+ const proposedCommand = textValue(candidate.command);
1874
+ let command = executionMode === "agent_planned" && candidate.kind === "duration"
1875
+ ? undefined
1876
+ : normalizedPreflightCommand(proposedCommand);
1877
+ if (candidate.kind === "parser" && command) {
1878
+ const sanitized = sanitizeParserPreflightCommand(command, measurementBindings);
1879
+ command = sanitized.command;
1880
+ if (sanitized.replacements.length) {
1881
+ warnings.push(`Experiment ${experimentId} parser preflight ${id} replaced ${sanitized.replacements.length} paper-target literal(s) with a neutral synthetic sentinel.`);
1882
+ }
1883
+ }
1884
+ if (proposedCommand && executionMode === "agent_planned" && candidate.kind === "duration") {
1885
+ warnings.push(`Experiment ${experimentId} duration check kept its success criterion but dropped the compiler-generated timing command; the execution Agent owns the real-device probe and batching strategy.`);
1886
+ }
1887
+ if (
1888
+ proposedCommand
1889
+ && !command
1890
+ && !(executionMode === "agent_planned" && candidate.kind === "duration")
1891
+ ) {
1892
+ warnings.push(`Experiment ${experimentId} preflight check ${id} had a non-executable generated probe; the command was removed while preserving its success criterion for execution-time repair.`);
1893
+ }
1894
+ checks.push({
1895
+ id,
1896
+ kind: candidate.kind,
1897
+ ...(command ? { command } : {}),
1898
+ successCriterion: textValue(candidate.successCriterion)
1899
+ ?? `Resolve the actual ${candidate.kind} requirement against the paper and working implementation before expensive execution.`,
1900
+ });
1901
+ }
1902
+ if (!checks.some((check) => check.kind === "dependencies")) {
1903
+ const dependencyCoverage = checks.find((check) =>
1904
+ new Set(["runtime", "entrypoint", "smoke_test"]).has(check.kind)
1905
+ && commandVerifiesDependencies(check.command),
1906
+ );
1907
+ if (dependencyCoverage) {
1908
+ const id = uniqueCompilerId("preflight-dependencies", usedIds);
1909
+ usedIds.add(id);
1910
+ checks.push({
1911
+ id,
1912
+ kind: "dependencies",
1913
+ command: dependencyCoverage.command,
1914
+ successCriterion: `The dependency imports or package checks already exercised by ${dependencyCoverage.id} succeed in the execution environment.`,
1915
+ });
1916
+ warnings.push(`Experiment ${experimentId} dependency coverage from ${dependencyCoverage.id} was normalized into an explicit dependencies check.`);
1917
+ }
1918
+ }
1919
+ const requiredKinds = ["runtime", "dependencies", "entrypoint", "parser", "smoke_test"];
1920
+ const assets = [];
1921
+ const assetIds = new Set();
1922
+ for (const [index, candidate] of (Array.isArray(value?.assets) ? value.assets : []).entries()) {
1923
+ if (!isRecord(candidate)) continue;
1924
+ if (isCompilerControlAsset(candidate)) {
1925
+ warnings.push(`Experiment ${experimentId} preflight omitted compiler-only control asset ${textValue(candidate.identity) ?? textValue(candidate.id) ?? index + 1}; repository identity is already fixed by the contract and runner audit.`);
1926
+ continue;
1927
+ }
1928
+ const id = uniqueCompilerId(textValue(candidate.id) ?? `asset-${index + 1}`, assetIds);
1929
+ assetIds.add(id);
1930
+ const sourceUrls = [...new Set(
1931
+ (Array.isArray(candidate.sourceUrls) ? candidate.sourceUrls : [])
1932
+ .map(textValue)
1933
+ .filter((sourceUrl) => sourceUrl && validPublicAssetSourceUrl(sourceUrl)),
1934
+ )];
1935
+ const registryDatasetId = textValue(candidate.registryDatasetId);
1936
+ assets.push({
1937
+ id,
1938
+ kind: PREFLIGHT_ASSET_KINDS.has(candidate.kind) ? candidate.kind : "other",
1939
+ identity: textValue(candidate.identity) ?? id,
1940
+ availability: PREFLIGHT_ASSET_AVAILABILITY.has(candidate.availability)
1941
+ ? candidate.availability
1942
+ : "unknown",
1943
+ required: candidate.required !== false,
1944
+ verification: textValue(candidate.verification)
1945
+ ?? "Verify the asset identity and record a digest-bound acquisition manifest before execution.",
1946
+ scientificRole: ASSET_SCIENTIFIC_ROLES.has(candidate.scientificRole)
1947
+ ? candidate.scientificRole
1948
+ : candidate.required === false
1949
+ ? "optional"
1950
+ : "execution_required",
1951
+ acquisition: compilerAssetAcquisition(candidate.acquisition),
1952
+ sourceEvidence: compilerAssetSourceEvidence(candidate.sourceEvidence),
1953
+ ...(sourceUrls.length ? { sourceUrls } : {}),
1954
+ ...(registryDatasetId ? { registryDatasetId } : {}),
1955
+ });
1956
+ }
1957
+ if (assets.length && !checks.some((check) => check.kind === "assets")) {
1958
+ checks.push({
1959
+ id: uniqueCompilerId("preflight-assets", usedIds),
1960
+ kind: "assets",
1961
+ successCriterion: "Every required asset is reachable and its identity can be verified before the full experiment begins.",
1962
+ });
1963
+ }
1964
+ if (assets.some((asset) => asset.required && asset.kind === "dataset")) {
1965
+ requiredKinds.push("dataset_loader");
1966
+ }
1967
+ const requirements = normalizeAssetRequirements(value?.requirements, {
1968
+ fillMissing: schemaVersion === "0.3" && executionMode !== "agent_planned",
1969
+ });
1970
+ const requirementCoverage = evaluateAssetRequirementCoverage({
1971
+ requirements: value?.requirements,
1972
+ assets,
1973
+ checks,
1974
+ });
1975
+ if (schemaVersion === "0.3" && requirementCoverage.status !== "complete") {
1976
+ warnings.push(
1977
+ executionMode === "agent_planned"
1978
+ ? `Experiment ${experimentId} material requirements are ${requirementCoverage.status}; unspecified categories are unknown, not automatically required. Resolve the materials the real implementation needs in the workspace.`
1979
+ : `Experiment ${experimentId} scientific material inventory is ${requirementCoverage.status}; every author-code, model, tokenizer, train/eval data, evaluator, configuration, and dependency row must be explicitly decided and bound before execution.`,
1980
+ );
1981
+ }
1982
+ const declaredKinds = new Set(checks.map((check) => check.kind));
1983
+ const declared = provided
1984
+ && Array.isArray(value.assets)
1985
+ && requiredKinds.every((kind) => declaredKinds.has(kind))
1986
+ && (schemaVersion !== "0.3" || requirementCoverage.status === "complete");
1987
+ const source = declared ? "declared" : "defaulted";
1988
+ if (!declared) {
1989
+ warnings.push(`Experiment ${experimentId} uses default preparation guidance; the execution workspace must verify the actual loader, implementation and scientific scope before expensive work.`);
1990
+ }
1991
+ for (const kind of requiredKinds) {
1992
+ if (checks.some((check) => check.kind === kind)) continue;
1993
+ const id = uniqueCompilerId(`preflight-${kind}`, usedIds);
1994
+ usedIds.add(id);
1995
+ checks.push({
1996
+ id,
1997
+ kind,
1998
+ successCriterion: kind === "entrypoint"
1999
+ ? executionMode === "agent_planned"
2000
+ ? "Inspect the real implementation and choose a working entrypoint for the scientific objective; the initial command is a revisable suggestion."
2001
+ : `The fixed entrypoint ${entrypoint} exists and can be invoked without beginning the full experiment.`
2002
+ : kind === "parser"
2003
+ ? executionMode === "agent_planned"
2004
+ ? "Use representative real output to connect each paper-defined quantity to its parser field, unit conversion and aggregation; a parseable fixture alone does not establish metric meaning."
2005
+ : `Every declared parser can read a tiny format-valid fixture at ${measurements.map((measurement) => measurement.parser.evidencePath).join(", ")}.`
2006
+ : kind === "dataset_loader"
2007
+ ? "The real evaluation loader resolves every required dataset configuration and split, reads representative rows, and verifies the declared schema and identity before any checkpoint download or full evaluation."
2008
+ : `Resolve the actual ${kind} requirement against the paper and working implementation before expensive execution.`,
2009
+ });
2010
+ }
2011
+ const proposedMaxMinutes = Number.isFinite(maxMinutes)
2012
+ ? Math.max(1, Math.min(MAX_PREFLIGHT_MINUTES, maxMinutes))
2013
+ : 10;
2014
+ const requiredPublicAssets = assets.filter((asset) =>
2015
+ asset.required && asset.availability === "public_unverified",
2016
+ );
2017
+ const normalizedMaxMinutes = requiredPublicAssets.length
2018
+ ? Math.max(MIN_PUBLIC_ASSET_PREFLIGHT_MINUTES, proposedMaxMinutes)
2019
+ : proposedMaxMinutes;
2020
+ if (normalizedMaxMinutes !== proposedMaxMinutes) {
2021
+ warnings.push(
2022
+ `Experiment ${experimentId} preflight budget was raised to ${MIN_PUBLIC_ASSET_PREFLIGHT_MINUTES} minutes because required public assets must be acquired and identity-verified before scientific execution.`,
2023
+ );
2024
+ }
2025
+ return {
2026
+ // Keep already-published 0.1/0.2 plans semantically stable. New compiler
2027
+ // output defaults to 0.3 and must close the complete material inventory.
2028
+ schemaVersion,
2029
+ source,
2030
+ maxMinutes: normalizedMaxMinutes,
2031
+ evidencePath: safeRelativePath(value?.evidencePath) ? value.evidencePath : "preflight.json",
2032
+ checks,
2033
+ assets,
2034
+ ...(schemaVersion === "0.3" || value?.requirements !== undefined ? { requirements } : {}),
2035
+ };
2036
+ }
2037
+
2038
+ function validPublicAssetSourceUrl(value) {
2039
+ try {
2040
+ const parsed = new URL(value);
2041
+ return new Set(["http:", "https:"]).has(parsed.protocol)
2042
+ && !parsed.username
2043
+ && !parsed.password;
2044
+ } catch {
2045
+ return false;
2046
+ }
2047
+ }
2048
+
2049
+ function compilerAssetAcquisition(value) {
2050
+ const candidate = isRecord(value) ? value : {};
2051
+ const mode = ASSET_ACQUISITION_MODES.has(candidate.mode) ? candidate.mode : "auto";
2052
+ const normalized = { mode };
2053
+ const revision = textValue(candidate.revision);
2054
+ if (revision) normalized.revision = revision;
2055
+ for (const field of ["requiredPaths", "excludedPaths"]) {
2056
+ const patterns = [...new Set((Array.isArray(candidate[field]) ? candidate[field] : [])
2057
+ .map(textValue)
2058
+ .filter((pattern) => pattern && safeAssetPattern(pattern)))];
2059
+ if (patterns.length) normalized[field] = patterns;
2060
+ }
2061
+ for (const field of [
2062
+ "expectedDownloadBytes",
2063
+ "expectedExpandedBytes",
2064
+ "maximumDownloadBytes",
2065
+ "maximumExpandedBytes",
2066
+ ]) {
2067
+ const number = Number(candidate[field]);
2068
+ if (
2069
+ Number.isSafeInteger(number)
2070
+ && number >= (field.startsWith("maximum") ? 1 : 0)
2071
+ ) {
2072
+ normalized[field] = number;
2073
+ }
2074
+ }
2075
+ if (typeof candidate.supportsStreaming === "boolean") {
2076
+ normalized.supportsStreaming = candidate.supportsStreaming;
2077
+ }
2078
+ return normalized;
2079
+ }
2080
+
2081
+ function compilerAssetSourceEvidence(value) {
2082
+ return (Array.isArray(value) ? value : []).flatMap((candidate) => {
2083
+ if (!isRecord(candidate)) return [];
2084
+ const source = textValue(candidate.source);
2085
+ const locator = textValue(candidate.locator);
2086
+ if (!new Set(["paper", "repository", "dataset_registry"]).has(source) || !locator) return [];
2087
+ const result = { source, locator };
2088
+ const sourcePath = textValue(candidate.path);
2089
+ if (sourcePath && safeRelativePath(sourcePath)) result.path = sourcePath;
2090
+ const commit = textValue(candidate.commit);
2091
+ if (commit) result.commit = commit;
2092
+ const resolver = textValue(candidate.resolver);
2093
+ if (new Set(["exact_url", "append_path", "huggingface_model", "huggingface_dataset"]).has(resolver)) {
2094
+ result.resolver = resolver;
2095
+ }
2096
+ const suffix = textValue(candidate.suffix);
2097
+ if (suffix && safeAssetPattern(suffix)) result.suffix = suffix;
2098
+ return [result];
2099
+ });
2100
+ }
2101
+
2102
+ function safeAssetPattern(value) {
2103
+ const normalized = String(value).replaceAll("\\", "/");
2104
+ return !path.posix.isAbsolute(normalized)
2105
+ && !normalized.split("/").includes("..")
2106
+ && !/[\u0000-\u001f]/.test(normalized);
2107
+ }
2108
+
2109
+ function normalizedPreflightCommand(value) {
2110
+ const command = textValue(value);
2111
+ if (!command || command.includes("\0")) return null;
2112
+ if (
2113
+ /(?:^|\s)python(?:3(?:\.\d+)?)?\s+-c(?:\s|$)/i.test(command)
2114
+ && /\b(?:true|false|null)\b/.test(command)
2115
+ ) return null;
2116
+ return command.trim();
2117
+ }
2118
+
2119
+ function isCompilerControlAsset(candidate) {
2120
+ const description = [candidate.id, candidate.identity, candidate.verification]
2121
+ .map(textValue)
2122
+ .filter(Boolean)
2123
+ .join(" ")
2124
+ .toLowerCase();
2125
+ return description.includes("repository-identity.json")
2126
+ || description.includes("/job/input/repository-identity")
2127
+ || description.includes("compiler-only repository identity");
2128
+ }
2129
+
2130
+ function compilerEvidenceParser(value, { experimentId, measurementId, warnings }) {
2131
+ const parser = { ...value };
2132
+ const original = textValue(parser.evidencePath);
2133
+ if (!original) return parser;
2134
+ const portable = original.replaceAll("\\", "/");
2135
+ const withoutOutputRoot = portable.startsWith("/job/output/")
2136
+ ? portable.slice("/job/output/".length)
2137
+ : portable.replace(/^\/+/, "");
2138
+ if (!safeRelativePath(withoutOutputRoot)) return parser;
2139
+ if (withoutOutputRoot !== original) {
2140
+ warnings.push(`Parser evidence path for ${measurementId} in experiment ${experimentId} was normalized to ${withoutOutputRoot}.`);
2141
+ }
2142
+ parser.evidencePath = withoutOutputRoot;
2143
+ return parser;
2144
+ }
2145
+
2146
+ function reconcileCompilerDispositions(claims, experiments, warnings, computeContext) {
2147
+ for (const claim of claims) {
2148
+ const requiredIds = new Set(claim.reportedMeasurements.map((measurement) => measurement.id));
2149
+ if (requiredIds.size > 0) {
2150
+ for (const experiment of experiments) {
2151
+ if (!experiment.claimIds.includes(claim.id)) continue;
2152
+ const bindsClaimMeasurement = experiment.measurements.some((measurement) =>
2153
+ requiredIds.has(measurement.reportedMeasurementId),
2154
+ );
2155
+ const hasObservationTarget = (experiment.observationTargets ?? []).some(target => target.claimId === claim.id);
2156
+ if (!bindsClaimMeasurement && !hasObservationTarget) {
2157
+ experiment.claimIds = experiment.claimIds.filter((claimId) => claimId !== claim.id);
2158
+ warnings.push(`Experiment ${experiment.id} was detached from claim ${claim.id} because it binds none of that claim's reported measurements.`);
2159
+ }
2160
+ }
2161
+ }
2162
+ const coverage = automaticClaimCoverage(claim, experiments);
2163
+ const requested = REPRODUCTION_DISPOSITIONS.has(claim.reproduction?.status)
2164
+ ? claim.reproduction.status
2165
+ : null;
2166
+ const contextualWithoutMeasurement = claim.type === "method"
2167
+ || claim.type === "limitation"
2168
+ || (Array.isArray(claim.reproduction?.derivedFromClaimIds)
2169
+ && claim.reproduction.derivedFromClaimIds.length > 0);
2170
+ if (requiredIds.size === 0 && contextualWithoutMeasurement) {
2171
+ if (requested === "planned") {
2172
+ warnings.push(`Claim ${claim.id} had no paper-reported measurement and was reclassified as contextual.`);
2173
+ }
2174
+ claim.reproduction = {
2175
+ ...claim.reproduction,
2176
+ status: "not_applicable",
2177
+ reason: textValue(claim.reproduction?.reason)
2178
+ ?? "This contextual claim has no paper-reported measurement and is not scheduled as an independent reproduction target.",
2179
+ };
2180
+ delete claim.reproduction.implementationOrigin;
2181
+ } else if (requiredIds.size === 0) {
2182
+ const comparisons = experiments.flatMap(experiment => experiment.observationTargets ?? [])
2183
+ .filter(target => target.claimId === claim.id);
2184
+ claim.reproduction = {
2185
+ ...claim.reproduction,
2186
+ status: "blocked",
2187
+ implementationOrigin: "insufficient",
2188
+ ...(comparisons.length ? {
2189
+ blocker: { kind: "decision_rule", evidence: comparisons.map(target => target.limitation?.evidence).filter(Boolean).join("; ") },
2190
+ } : {}),
2191
+ reason: (comparisons.length ? "The proposed observation comparison is retained, but a deterministic decision rule is unresolved." : null) ?? textValue(claim.reproduction?.reason)
2192
+ ?? "This empirical claim has no safely extractable paper-reported measurement or direct decision rule for an automatic experiment.",
2193
+ };
2194
+ warnings.push(`Empirical claim ${claim.id} was kept as blocked instead of being reclassified as contextual.`);
2195
+ } else if (coverage.bound.length > 0 && coverage.coveredIds.size > 0) {
2196
+ const partial = coverage.missingIds.length > 0;
2197
+ const coveredByOfficial = new Set(
2198
+ coverage.bound
2199
+ .filter((experiment) => experiment.implementationOrigin === "official")
2200
+ .flatMap((experiment) => experiment.measurements ?? [])
2201
+ .map((measurement) => measurement?.reportedMeasurementId)
2202
+ .filter((measurementId) => coverage.coveredIds.has(measurementId)),
2203
+ );
2204
+ claim.reproduction = {
2205
+ ...claim.reproduction,
2206
+ status: "planned",
2207
+ implementationOrigin: [...coverage.coveredIds].every((measurementId) => coveredByOfficial.has(measurementId))
2208
+ ? "official"
2209
+ : "citeark_reconstruction",
2210
+ ...(partial ? {
2211
+ reason: `Partially planned: runnable experiments cover ${coverage.coveredIds.size} of ${requiredIds.size} paper-reported measurements; these measurements remain unscheduled: ${coverage.missingIds.join(", ")}.`,
2212
+ } : {}),
2213
+ };
2214
+ if (!partial) delete claim.reproduction.reason;
2215
+ if (partial) {
2216
+ warnings.push(`Claim ${claim.id} retained ${coverage.bound.length} runnable partial experiment(s); ${coverage.missingIds.join(", ")} remain explicitly unscheduled.`);
2217
+ }
2218
+ } else {
2219
+ // Absence of a binding is a planning gap, not evidence of infeasibility.
2220
+ const blocker = claim.reproduction?.blocker;
2221
+ const substantiatedBlocker = requested === "blocked" && isRecord(blocker)
2222
+ && ["scientific_input", "access", "resources", "budget", "decision_rule"].includes(blocker.kind)
2223
+ && Boolean(textValue(blocker.evidence))
2224
+ && !(blocker.kind === "budget" && (computeContext?.limits?.maximumCost === null));
2225
+ claim.reproduction = {
2226
+ ...claim.reproduction,
2227
+ status: substantiatedBlocker ? "blocked" : "deferred",
2228
+ reason: blocker?.kind === "budget" && computeContext?.limits?.maximumCost === null
2229
+ ? "The prior monetary blocker does not apply to this uncapped job; this empirical claim still needs a runnable experiment."
2230
+ : textValue(claim.reproduction?.reason)
2231
+ ?? "No experiment has been scheduled for this empirical claim; feasibility remains to be established.",
2232
+ };
2233
+ if (substantiatedBlocker) claim.reproduction.implementationOrigin = "insufficient";
2234
+ else {
2235
+ delete claim.reproduction.implementationOrigin;
2236
+ delete claim.reproduction.blocker;
2237
+ warnings.push(`Claim ${claim.id} remains unscheduled; missing experiment bindings do not establish a blocker.`);
2238
+ }
2239
+ }
2240
+ }
2241
+ }
2242
+
2243
+ function claimCatalogCoverage(claim, experiments) {
2244
+ const requiredIds = new Set(
2245
+ (Array.isArray(claim?.reportedMeasurements) ? claim.reportedMeasurements : [])
2246
+ .map((measurement) => measurement?.id)
2247
+ .filter(Boolean),
2248
+ );
2249
+ const bound = (Array.isArray(experiments) ? experiments : []).filter((experiment) =>
2250
+ Array.isArray(experiment?.claimIds) && experiment.claimIds.includes(claim?.id),
2251
+ );
2252
+ const coveredIds = new Set(bound.flatMap((experiment) =>
2253
+ (experiment.measurements ?? [])
2254
+ .map((measurement) => measurement?.reportedMeasurementId)
2255
+ .filter((measurementId) => requiredIds.has(measurementId)),
2256
+ ));
2257
+ return {
2258
+ bound,
2259
+ coveredIds,
2260
+ missingIds: [...requiredIds].filter((measurementId) => !coveredIds.has(measurementId)),
2261
+ };
2262
+ }
2263
+
2264
+ function compilerCoverage(value, { claims, sourceIds, paperSourceId, warnings }) {
2265
+ const claimIds = new Set(claims.map((claim) => claim.id));
2266
+ const items = [];
2267
+ const usedIds = new Set();
2268
+ for (const [index, candidate] of (Array.isArray(value?.items) ? value.items : []).entries()) {
2269
+ if (!isRecord(candidate)) {
2270
+ warnings.push(`Invalid coverage.items[${index}] was omitted.`);
2271
+ continue;
2272
+ }
2273
+ const id = uniqueCompilerId(textValue(candidate.id) ?? `coverage-${index + 1}`, usedIds);
2274
+ usedIds.add(id);
2275
+ const covered = [...new Set(
2276
+ (Array.isArray(candidate.claimIds) ? candidate.claimIds : []).filter((claimId) => claimIds.has(claimId)),
2277
+ )];
2278
+ let status = COVERAGE_STATUSES.has(candidate.status) ? candidate.status : (covered.length ? "represented" : "not_applicable");
2279
+ if (status === "represented" && !covered.length) status = "not_applicable";
2280
+ if (covered.length) {
2281
+ const dispositions = claims.filter((claim) => covered.includes(claim.id)).map((claim) => claim.reproduction.status);
2282
+ if (dispositions.includes("planned")) status = "represented";
2283
+ else if (dispositions.includes("deferred")) status = "deferred";
2284
+ }
2285
+ items.push({
2286
+ ...candidate,
2287
+ id,
2288
+ description: textValue(candidate.description) ?? `Research coverage item ${index + 1}`,
2289
+ sourceLocator: compilerLocator(candidate.sourceLocator, {
2290
+ sourceIds,
2291
+ fallbackSourceId: paperSourceId,
2292
+ fallbackLocator: "Location not specified by the compiler",
2293
+ warning: `coverage.items[${index}].sourceLocator was repaired against the fixed paper source.`,
2294
+ warnings,
2295
+ }),
2296
+ status,
2297
+ claimIds: covered,
2298
+ ...(status === "represented" ? {} : {
2299
+ reason: textValue(candidate.reason) ?? "This item has no independently verifiable structured claim binding.",
2300
+ }),
2301
+ });
2302
+ }
2303
+ if (!items.length) {
2304
+ warnings.push("Missing coverage inventory was derived from normalized claims.");
2305
+ for (const claim of claims) {
2306
+ const status = claim.reproduction.status === "planned"
2307
+ ? "represented"
2308
+ : claim.reproduction.status === "blocked"
2309
+ ? "blocked"
2310
+ : claim.reproduction.status === "deferred" ? "deferred" : "not_applicable";
2311
+ items.push({
2312
+ id: uniqueCompilerId(`coverage-${claim.id}`, usedIds),
2313
+ description: claim.statement,
2314
+ sourceLocator: claim.sourceLocator,
2315
+ status,
2316
+ claimIds: [claim.id],
2317
+ ...(status === "represented" ? {} : { reason: claim.reproduction.reason }),
2318
+ });
2319
+ }
2320
+ }
2321
+ if (!items.length) {
2322
+ items.push({
2323
+ id: "coverage-unstructured",
2324
+ description: "The compiler did not produce a grounded empirical claim inventory.",
2325
+ sourceLocator: { sourceId: paperSourceId, locator: "Fixed paper input" },
2326
+ status: "not_applicable",
2327
+ claimIds: [],
2328
+ reason: "No grounded structured claim was available for execution scheduling.",
2329
+ });
2330
+ }
2331
+ return { items };
2332
+ }
2333
+
2334
+ function compilerLineage(value, { sources, claims, experiments, warnings }) {
2335
+ const nodeIds = new Set([
2336
+ ...sources.map((source) => source.id),
2337
+ ...claims.map((claim) => claim.id),
2338
+ ...claims.flatMap((claim) => claim.reportedMeasurements.map((measurement) => measurement.id)),
2339
+ ...experiments.map((experiment) => experiment.id),
2340
+ ]);
2341
+ const lineage = [];
2342
+ const keys = new Set();
2343
+ for (const edge of (Array.isArray(value) ? value : [])) {
2344
+ if (!isRecord(edge) || !nodeIds.has(edge.from) || !nodeIds.has(edge.to) || !textValue(edge.relation)) {
2345
+ warnings.push("An invalid lineage edge was omitted.");
2346
+ continue;
2347
+ }
2348
+ const key = `${edge.from}\u0000${edge.to}\u0000${edge.relation}`;
2349
+ if (keys.has(key)) continue;
2350
+ keys.add(key);
2351
+ lineage.push(edge);
2352
+ }
2353
+ const paperId = sources.find((source) => source.kind === "paper")?.id;
2354
+ for (const claim of claims) {
2355
+ addCompilerEdge(lineage, keys, paperId, claim.id, "reports");
2356
+ for (const measurement of claim.reportedMeasurements) {
2357
+ addCompilerEdge(lineage, keys, claim.id, measurement.id, "reports_measurement");
2358
+ }
2359
+ }
2360
+ for (const experiment of experiments) {
2361
+ for (const claimId of experiment.claimIds) addCompilerEdge(lineage, keys, claimId, experiment.id, "tested_by");
2362
+ }
2363
+ return lineage;
2364
+ }
2365
+
2366
+ function compilerLicense(value, policy, warnings) {
2367
+ const license = isRecord(value) ? { ...value } : {};
2368
+ if (!isRecord(value)) warnings.push("Missing license record was reconstructed from platform policy.");
2369
+ return {
2370
+ ...license,
2371
+ paper: textValue(license.paper) ?? textValue(policy?.paperLicense) ?? "unknown",
2372
+ code: textValue(license.code) ?? textValue(policy?.codeLicense) ?? "unknown",
2373
+ executionAllowed: policy?.executable !== false,
2374
+ };
2375
+ }
2376
+
2377
+ function compilerProvenance(value, context, warnings) {
2378
+ const provenance = isRecord(value) ? { ...value } : {};
2379
+ if (!isRecord(value)) warnings.push("Missing provenance was reconstructed by the platform compiler boundary.");
2380
+ return {
2381
+ ...provenance,
2382
+ compiler: textValue(provenance.compiler) ?? textValue(context.compiler) ?? "citeark-research-compiler/0.2",
2383
+ compiledAt: textValue(provenance.compiledAt) ?? new Date().toISOString(),
2384
+ reviewStatus: new Set(["agent-draft", "human-reviewed"]).has(provenance.reviewStatus)
2385
+ ? provenance.reviewStatus
2386
+ : "agent-draft",
2387
+ };
2388
+ }
2389
+
2390
+ function compilerLocator(value, { sourceIds, fallbackSourceId, fallbackLocator, warning, warnings }) {
2391
+ const sourceId = textValue(value?.sourceId);
2392
+ const locator = textValue(value?.locator);
2393
+ if (sourceId && sourceIds.has(sourceId) && locator) return { ...value, sourceId, locator };
2394
+ warnings.push(warning);
2395
+ return {
2396
+ ...(isRecord(value) ? value : {}),
2397
+ sourceId: sourceId && sourceIds.has(sourceId) ? sourceId : fallbackSourceId,
2398
+ locator: locator ?? fallbackLocator,
2399
+ };
2400
+ }
2401
+
2402
+ function addCompilerEdge(lineage, keys, from, to, relation) {
2403
+ if (!from || !to) return;
2404
+ const key = `${from}\u0000${to}\u0000${relation}`;
2405
+ if (keys.has(key)) return;
2406
+ keys.add(key);
2407
+ lineage.push({ from, to, relation });
2408
+ }
2409
+
2410
+ function uniqueCompilerId(base, used) {
2411
+ const normalized = base.trim().replaceAll(/[^a-zA-Z0-9._:-]+/g, "-") || "item";
2412
+ if (!used.has(normalized)) return normalized;
2413
+ let suffix = 2;
2414
+ while (used.has(`${normalized}-${suffix}`)) suffix += 1;
2415
+ return `${normalized}-${suffix}`;
2416
+ }
2417
+
2418
+ function textValue(value) {
2419
+ return typeof value === "string" && value.trim() ? value.trim() : null;
2420
+ }
2421
+
2422
+ export function normalizeResearchStructure(input, options = {}) {
2423
+ const issues = [];
2424
+ if (!isRecord(input)) throw new CiteArkError("research.json 的根节点必须是对象");
2425
+ const inventoryOnly = input.compilationStage === "inventory";
2426
+ if (input.compilationStage !== undefined && !inventoryOnly) issues.push("Unknown compilationStage");
2427
+ if (inventoryOnly && (input.experiments?.length || input.claims?.some((claim) => claim.reproduction?.status !== "deferred"))) issues.push("Inventory must contain pending claims and no executable experiments");
2428
+ if (input.schemaVersion !== RESEARCH_SCHEMA_VERSION) {
2429
+ issues.push(`schemaVersion 必须是 ${RESEARCH_SCHEMA_VERSION}`);
2430
+ }
2431
+ if (!isRecord(input.work)) issues.push("work 必须是对象");
2432
+ const rawWork = isRecord(input.work) ? input.work : {};
2433
+ const work = options.requireAuthors === true && rawWork.authors === undefined
2434
+ ? { ...rawWork, authors: [] }
2435
+ : rawWork;
2436
+ required(work.id, "work.id", issues);
2437
+ required(work.title, "work.title", issues);
2438
+ required(work.abstract, "work.abstract", issues);
2439
+ if (textValue(work.abstract) && textValue(work.abstract).length > 1200) {
2440
+ issues.push("work.abstract 最多 1200 字符:编译器必须用自己的话重写为 80–120 词的客观陈述,不得照抄论文原摘要");
2441
+ }
2442
+ if (work.authors !== undefined) {
2443
+ const authors = array(work.authors, "work.authors", issues);
2444
+ if (authors.length > 500) issues.push("work.authors 最多包含 500 位作者");
2445
+ authors.forEach((author, index) => required(author, `work.authors[${index}]`, issues));
2446
+ }
2447
+ if (work.subjects !== undefined) {
2448
+ array(work.subjects, "work.subjects", issues).forEach((tag, index) => {
2449
+ if (!SUBJECT_TAGS.has(tag)) issues.push(`work.subjects[${index}] 无效:${tag}`);
2450
+ });
2451
+ }
2452
+
2453
+ const sources = array(input.sources, "sources", issues);
2454
+ const sourceIds = new Set();
2455
+ for (const [index, source] of sources.entries()) {
2456
+ if (!isRecord(source)) {
2457
+ issues.push(`sources[${index}] 必须是对象`);
2458
+ continue;
2459
+ }
2460
+ required(source.id, `sources[${index}].id`, issues);
2461
+ if (sourceIds.has(source.id)) issues.push(`source id 重复:${source.id}`);
2462
+ sourceIds.add(source.id);
2463
+ if (!SOURCE_KINDS.has(source.kind)) issues.push(`sources[${index}].kind 无效`);
2464
+ required(source.uri, `sources[${index}].uri`, issues);
2465
+ }
2466
+
2467
+ const claims = array(input.claims, "claims", issues);
2468
+ const claimIds = new Set();
2469
+ const reportedMeasurementIds = new Set();
2470
+ const normalizedClaims = claims.map((claim, index) => {
2471
+ if (!isRecord(claim)) {
2472
+ issues.push(`claims[${index}] 必须是对象`);
2473
+ return claim;
2474
+ }
2475
+ required(claim.id, `claims[${index}].id`, issues);
2476
+ if (claimIds.has(claim.id)) issues.push(`claim id 重复:${claim.id}`);
2477
+ claimIds.add(claim.id);
2478
+ required(claim.statement, `claims[${index}].statement`, issues);
2479
+ if (!CLAIM_TYPES.has(claim.type)) issues.push(`claims[${index}].type 无效`);
2480
+ validateLocator(claim.sourceLocator, `claims[${index}].sourceLocator`, sourceIds, issues);
2481
+ const measurements = array(claim.reportedMeasurements, `claims[${index}].reportedMeasurements`, issues);
2482
+ measurements.forEach((measurement, measurementIndex) => {
2483
+ validateMeasurement(measurement, `claims[${index}].reportedMeasurements[${measurementIndex}]`, issues);
2484
+ if (isRecord(measurement)) {
2485
+ if (measurement.value === undefined) issues.push(`claims[${index}].reportedMeasurements[${measurementIndex}].value 缺少报告数值;无可靠数值的经验结论应保留为空测量数组及明确判据阻塞`);
2486
+ required(measurement.id, `claims[${index}].reportedMeasurements[${measurementIndex}].id`, issues);
2487
+ if (reportedMeasurementIds.has(measurement.id)) issues.push(`reported measurement id 重复:${measurement.id}`);
2488
+ reportedMeasurementIds.add(measurement.id);
2489
+ validateLocator(measurement.sourceLocator, `claims[${index}].reportedMeasurements[${measurementIndex}].sourceLocator`, sourceIds, issues);
2490
+ }
2491
+ });
2492
+ const reproduction = normalizeClaimReproduction(claim, index, measurements, issues, inventoryOnly);
2493
+ const body = { ...claim };
2494
+ body.reproduction = reproduction;
2495
+ delete body.versionId;
2496
+ return { ...body, versionId: `sha256:${sha256Value(body)}` };
2497
+ });
2498
+
2499
+ const experiments = array(input.experiments, "experiments", issues);
2500
+ const experimentIds = new Set();
2501
+ const normalizedExperiments = experiments.map((experiment, index) => {
2502
+ if (!isRecord(experiment)) {
2503
+ issues.push(`experiments[${index}] 必须是对象`);
2504
+ return experiment;
2505
+ }
2506
+ required(experiment.id, `experiments[${index}].id`, issues);
2507
+ if (experimentIds.has(experiment.id)) issues.push(`experiment id 重复:${experiment.id}`);
2508
+ experimentIds.add(experiment.id);
2509
+ required(experiment.title, `experiments[${index}].title`, issues);
2510
+ if (!REPRODUCTION_LEVELS.has(experiment.reproductionLevel)) {
2511
+ issues.push(`experiments[${index}].reproductionLevel 无效`);
2512
+ }
2513
+ if (
2514
+ experiment.reconstructionFidelity !== undefined
2515
+ && !RECONSTRUCTION_FIDELITIES.has(experiment.reconstructionFidelity)
2516
+ ) {
2517
+ issues.push(`experiments[${index}].reconstructionFidelity 无效`);
2518
+ }
2519
+ if (experiment.implementationOrigin !== undefined && !IMPLEMENTATION_ORIGINS.has(experiment.implementationOrigin)) {
2520
+ issues.push(`experiments[${index}].implementationOrigin 无效`);
2521
+ }
2522
+ const boundClaims = array(experiment.claimIds, `experiments[${index}].claimIds`, issues);
2523
+ boundClaims.forEach((claimId) => {
2524
+ if (!claimIds.has(claimId)) issues.push(`experiment 引用了不存在的 claim:${claimId}`);
2525
+ });
2526
+ if (!isRecord(experiment.repository)) issues.push(`experiments[${index}].repository 必须是对象`);
2527
+ else {
2528
+ required(experiment.repository.url, `experiments[${index}].repository.url`, issues);
2529
+ required(experiment.repository.commit, `experiments[${index}].repository.commit`, issues);
2530
+ if (experiment.repository.implementationOrigin !== undefined && !IMPLEMENTATION_ORIGINS.has(experiment.repository.implementationOrigin)) {
2531
+ issues.push(`experiments[${index}].repository.implementationOrigin 无效`);
2532
+ }
2533
+ }
2534
+ if (!isRecord(experiment.protocol)) issues.push(`experiments[${index}].protocol 必须是对象`);
2535
+ else {
2536
+ if (
2537
+ experiment.protocol.executionMode !== undefined
2538
+ && !EXECUTION_MODES.has(experiment.protocol.executionMode)
2539
+ ) {
2540
+ issues.push(`experiments[${index}].protocol.executionMode 无效`);
2541
+ }
2542
+ if (experiment.protocol.executionMode === "agent_planned") {
2543
+ required(experiment.protocol.objective, `experiments[${index}].protocol.objective`, issues);
2544
+ }
2545
+ required(experiment.protocol.entrypoint, `experiments[${index}].protocol.entrypoint`, issues);
2546
+ const fragments = array(experiment.protocol.requiredCommandFragments, `experiments[${index}].protocol.requiredCommandFragments`, issues);
2547
+ fragments.forEach((fragment, fragmentIndex) => required(fragment, `experiments[${index}].protocol.requiredCommandFragments[${fragmentIndex}]`, issues));
2548
+ validateProtocolPreflight(experiment.protocol.preflight, `experiments[${index}].protocol.preflight`, issues, experiment.protocol.executionMode === "agent_planned");
2549
+ }
2550
+ const measurements = array(experiment.measurements, `experiments[${index}].measurements`, issues);
2551
+ measurements.forEach((measurement, measurementIndex) => {
2552
+ const label = `experiments[${index}].measurements[${measurementIndex}]`;
2553
+ validateMeasurement(measurement, label, issues);
2554
+ if (isRecord(measurement)) {
2555
+ required(measurement.reportedMeasurementId, `${label}.reportedMeasurementId`, issues);
2556
+ if (measurement.reportedMeasurementId && !reportedMeasurementIds.has(measurement.reportedMeasurementId)) {
2557
+ issues.push(`${label}.reportedMeasurementId 不存在:${measurement.reportedMeasurementId}`);
2558
+ }
2559
+ const owners = claims.filter((claim) =>
2560
+ Array.isArray(claim?.reportedMeasurements)
2561
+ && claim.reportedMeasurements.some((reported) => reported?.id === measurement.reportedMeasurementId),
2562
+ );
2563
+ if (owners.length && !owners.some((claim) => boundClaims.includes(claim.id))) {
2564
+ issues.push(`${label} 引用的 reported measurement 不属于 experiment.claimIds`);
2565
+ }
2566
+ const reported = owners.flatMap((claim) => claim.reportedMeasurements).find(
2567
+ (candidate) => candidate?.id === measurement.reportedMeasurementId,
2568
+ );
2569
+ if (reported && (reported.metric !== measurement.metric || reported.unit !== measurement.unit)) {
2570
+ issues.push(`${label} 与 ${measurement.reportedMeasurementId} 的 metric/unit 不一致`);
2571
+ }
2572
+ }
2573
+ if (!isRecord(measurement?.parser)) issues.push(`${label}.parser 必须是对象`);
2574
+ else {
2575
+ required(measurement.parser.id, `${label}.parser.id`, issues);
2576
+ required(measurement.parser.version, `${label}.parser.version`, issues);
2577
+ required(measurement.parser.evidencePath, `${label}.parser.evidencePath`, issues);
2578
+ for (const parserIssue of validateEvidenceParserDescriptor(
2579
+ measurement.parser,
2580
+ measurement,
2581
+ )) {
2582
+ issues.push(`${label}.parser:${parserIssue}`);
2583
+ }
2584
+ }
2585
+ });
2586
+ if (experiment.protocol?.benchmark || measurements.some(measurementRequiresHardwareProtocol)) {
2587
+ validateHardwareBenchmarkProtocol(
2588
+ experiment.protocol?.benchmark,
2589
+ `experiments[${index}].protocol.benchmark`,
2590
+ issues,
2591
+ {
2592
+ requiresComparator: measurements.some((measurement) =>
2593
+ typeof measurement?.metric === "string"
2594
+ && /(?:speedup|relative[_ -]?speed|ratio)/i.test(measurement.metric)),
2595
+ },
2596
+ );
2597
+ }
2598
+ issues.push(...measurementHardwareIssues(experiment).map((issue) => `${experiment.id}: ${issue}`));
2599
+ issues.push(...executionWorkloadIssues(experiment.protocol?.workload).map((issue) => `${experiment.id}: ${issue}`));
2600
+ if (experiment.compute !== undefined) {
2601
+ issues.push(...validateDeclaredComputeRequirement(experiment.compute, `experiments[${index}].compute`));
2602
+ }
2603
+ issues.push(...publicContractPolicyIssues(researchPublicContract(experiment))
2604
+ .map((issue) => `experiments[${index}].publicContract: ${issue}`));
2605
+ const body = { ...experiment };
2606
+ delete body.versionId;
2607
+ return { ...body, versionId: `sha256:${sha256Value(body)}` };
2608
+ });
2609
+
2610
+ for (const [index, claim] of normalizedClaims.entries()) {
2611
+ if (!isRecord(claim) || claim.type === "limitation") continue;
2612
+ const disposition = claim.reproduction?.status;
2613
+ const measurements = Array.isArray(claim.reportedMeasurements) ? claim.reportedMeasurements : [];
2614
+ const claimExperiments = experiments.filter((experiment) =>
2615
+ Array.isArray(experiment?.claimIds) && experiment.claimIds.includes(claim.id),
2616
+ );
2617
+ if (disposition === "planned") {
2618
+ if (!claimExperiments.length) issues.push(`claims[${index}] 标记为 planned,但没有绑定 experiment`);
2619
+ const requiredMeasurementIds = new Set(measurements.map((measurement) => measurement?.id).filter(Boolean));
2620
+ const collectivelyCovered = new Set(
2621
+ claimExperiments.flatMap((experiment) =>
2622
+ (experiment.measurements ?? [])
2623
+ .map((measurement) => measurement?.reportedMeasurementId)
2624
+ .filter((measurementId) => requiredMeasurementIds.has(measurementId)),
2625
+ ),
2626
+ );
2627
+ for (const experiment of claimExperiments) {
2628
+ const bindsClaimMeasurement = (experiment.measurements ?? []).some((measurement) =>
2629
+ requiredMeasurementIds.has(measurement?.reportedMeasurementId),
2630
+ );
2631
+ const hasObservationTarget = (experiment.observationTargets ?? []).some(target => target.claimId === claim.id);
2632
+ if (!bindsClaimMeasurement && !hasObservationTarget) {
2633
+ issues.push(`experiment ${experiment.id} 没有绑定 claim ${claim.id} 的任何 reported measurement`);
2634
+ }
2635
+ }
2636
+ if (requiredMeasurementIds.size > 0 && collectivelyCovered.size === 0) {
2637
+ issues.push(`claims[${index}] 标记为 planned,但没有任何 reported measurement 被实验测量绑定`);
2638
+ }
2639
+ }
2640
+ }
2641
+
2642
+ const lineage = array(input.lineage, "lineage", issues);
2643
+ const nodeIds = new Set([...sourceIds, ...claimIds, ...reportedMeasurementIds, ...experimentIds]);
2644
+ lineage.forEach((edge, index) => {
2645
+ if (!isRecord(edge)) {
2646
+ issues.push(`lineage[${index}] 必须是对象`);
2647
+ return;
2648
+ }
2649
+ if (!nodeIds.has(edge.from)) issues.push(`lineage[${index}].from 不存在:${edge.from}`);
2650
+ if (!nodeIds.has(edge.to)) issues.push(`lineage[${index}].to 不存在:${edge.to}`);
2651
+ required(edge.relation, `lineage[${index}].relation`, issues);
2652
+ });
2653
+
2654
+ if (!isRecord(input.license)) issues.push("license 必须是对象");
2655
+ else {
2656
+ required(input.license.paper, "license.paper", issues);
2657
+ required(input.license.code, "license.code", issues);
2658
+ if (typeof input.license.executionAllowed !== "boolean") issues.push("license.executionAllowed 必须是布尔值");
2659
+ }
2660
+ if (!isRecord(input.provenance)) issues.push("provenance 必须是对象");
2661
+ else {
2662
+ required(input.provenance.compiler, "provenance.compiler", issues);
2663
+ required(input.provenance.compiledAt, "provenance.compiledAt", issues);
2664
+ if (!new Set(["agent-draft", "human-reviewed"]).has(input.provenance.reviewStatus)) {
2665
+ issues.push("provenance.reviewStatus 必须是 agent-draft 或 human-reviewed");
2666
+ }
2667
+ }
2668
+ if (input.sourceInventory !== undefined) {
2669
+ const source = input.sourceInventory;
2670
+ if (inventoryOnly || source?.research?.compilationStage !== "inventory" || source.research.sourceInventory !== undefined) {
2671
+ issues.push("sourceInventory must reference a standalone inventory");
2672
+ } else {
2673
+ try {
2674
+ const inventory = normalizeResearchStructure(source.research);
2675
+ if (source.digest !== `sha256:${sha256Value(inventory)}`) issues.push("sourceInventory digest mismatch");
2676
+ const revisedClaims = normalizeResearchStructure({ ...inventory,
2677
+ claims: revisedInventoryClaims(inventory, source.revisions).map((claim) => ({ ...claim, reproduction: { status: "deferred", reason: "Provisional source interpretation awaiting preparation." } })),
2678
+ }).claims.map(sourceClaim);
2679
+ if (sha256Value(normalizedClaims.map(sourceClaim)) !== sha256Value(revisedClaims)) issues.push("Preparation cannot rewrite source inventory claims or reported measurements without a recorded source-grounded revision");
2680
+ if (sha256Value(input.hypotheses ?? []) !== sha256Value(inventory.hypotheses ?? [])) issues.push("Preparation cannot rewrite source hypotheses");
2681
+ for (const field of ["reading", "researchRelations"]) if (sha256Value(input[field] ?? null) !== sha256Value(inventory[field] ?? null)) issues.push(`Preparation cannot rewrite source ${field}`);
2682
+ const preparedObjects = new Map((input.researchObjects ?? []).map(object => [object.id, object]));
2683
+ for (const object of inventory.researchObjects ?? []) if (!preparedObjects.has(object.id)
2684
+ || sha256Value(preparedObjects.get(object.id)) !== sha256Value(object)) issues.push(`Preparation cannot rewrite source research object ${object.id}`);
2685
+ } catch (error) { issues.push(`Invalid sourceInventory: ${error.message}`); }
2686
+ }
2687
+ }
2688
+ validateResearchMap(input, issues);
2689
+ const researchObjects = validateResearchObjects({ ...input, claims: normalizedClaims, experiments: normalizedExperiments }, issues);
2690
+ const coverage = validateCoverageInventory(input.coverage, claimIds, sourceIds, issues);
2691
+ issues.push(...reproductionScopeIssues(input));
2692
+ const experimentSelection = normalizeExperimentSelection(
2693
+ input.experimentSelection,
2694
+ experimentIds,
2695
+ );
2696
+
2697
+ if (issues.length) throw new CiteArkError(`研究结构校验失败:\n- ${issues.join("\n- ")}`);
2698
+ return {
2699
+ schemaVersion: RESEARCH_SCHEMA_VERSION,
2700
+ ...(inventoryOnly ? { compilationStage: "inventory" } : {}),
2701
+ ...(input.sourceInventory ? { sourceInventory: structuredClone(input.sourceInventory) } : {}),
2702
+ ...(input.reproductionScope !== undefined ? { reproductionScope: input.reproductionScope } : {}),
2703
+ ...(experimentSelection ? { experimentSelection } : {}),
2704
+ work,
2705
+ sources,
2706
+ claims: normalizedClaims,
2707
+ ...(input.reading !== undefined ? { reading: structuredClone(input.reading) } : {}),
2708
+ ...(input.researchRelations !== undefined ? { researchRelations: structuredClone(input.researchRelations) } : {}),
2709
+ ...(input.hypotheses !== undefined ? { hypotheses: structuredClone(input.hypotheses) } : {}),
2710
+ experiments: normalizedExperiments,
2711
+ ...(input.researchObjects !== undefined ? { researchObjects } : {}),
2712
+ lineage,
2713
+ coverage,
2714
+ license: input.license,
2715
+ provenance: input.provenance,
2716
+ };
2717
+ }
2718
+
2719
+ function normalizeExperimentSelection(value, experimentIds) {
2720
+ if (!isRecord(value)) return null;
2721
+ const groups = [];
2722
+ const assigned = new Set();
2723
+ for (const candidate of Array.isArray(value.groups) ? value.groups.slice(0, 5) : []) {
2724
+ if (!isRecord(candidate)) continue;
2725
+ const id = textValue(candidate.id);
2726
+ const title = textValue(candidate.title);
2727
+ const description = textValue(candidate.description);
2728
+ const ids = [...new Set(
2729
+ (Array.isArray(candidate.experimentIds) ? candidate.experimentIds : [])
2730
+ .filter((experimentId) => typeof experimentId === "string"
2731
+ && experimentIds.has(experimentId)
2732
+ && !assigned.has(experimentId)),
2733
+ )];
2734
+ if (!id || !title || !description || !ids.length) continue;
2735
+ ids.forEach((experimentId) => assigned.add(experimentId));
2736
+ groups.push({ id, title, description, experimentIds: ids });
2737
+ }
2738
+ const presets = {};
2739
+ for (const preset of ["low", "medium", "high"]) {
2740
+ const candidate = value.presets?.[preset];
2741
+ if (!isRecord(candidate)) continue;
2742
+ const ids = [...new Set(
2743
+ (Array.isArray(candidate.experimentIds) ? candidate.experimentIds : [])
2744
+ .filter((id) => typeof id === "string" && experimentIds.has(id)),
2745
+ )];
2746
+ const rationale = textValue(candidate.rationale);
2747
+ if (rationale) presets[preset] = { experimentIds: ids, rationale };
2748
+ }
2749
+ const importance = normalizeExperimentImportance(value.importance, experimentIds);
2750
+ return Object.keys(presets).length || groups.length || importance.length ? {
2751
+ ...(importance.length ? { importance } : {}),
2752
+ ...(groups.length ? { groups } : {}),
2753
+ presets,
2754
+ } : null;
2755
+ }
2756
+
2757
+ function validateProtocolPreflight(value, label, issues, agentPlanned = false) {
2758
+ // Research Plan 1.0 records created before structured preflight remain valid.
2759
+ if (value === undefined) return;
2760
+ if (!isRecord(value)) {
2761
+ issues.push(`${label} 必须是对象`);
2762
+ return;
2763
+ }
2764
+ if (!supportsPreflightSchemaVersion(value.schemaVersion)) {
2765
+ issues.push(`${label}.schemaVersion 必须是 ${describeSupportedPreflightSchemaVersions()}`);
2766
+ }
2767
+ if (!new Set(["declared", "defaulted"]).has(value.source)) issues.push(`${label}.source 无效`);
2768
+ if (!Number.isFinite(value.maxMinutes) || value.maxMinutes < 1 || value.maxMinutes > MAX_PREFLIGHT_MINUTES) {
2769
+ issues.push(`${label}.maxMinutes 必须在 1 到 ${MAX_PREFLIGHT_MINUTES} 之间`);
2770
+ }
2771
+ if (!safeRelativePath(value.evidencePath)) issues.push(`${label}.evidencePath 必须是安全相对路径`);
2772
+ const checks = array(value.checks, `${label}.checks`, issues);
2773
+ const checkIds = new Set();
2774
+ for (const [index, check] of checks.entries()) {
2775
+ if (!isRecord(check)) {
2776
+ issues.push(`${label}.checks[${index}] 必须是对象`);
2777
+ continue;
2778
+ }
2779
+ required(check.id, `${label}.checks[${index}].id`, issues);
2780
+ if (checkIds.has(check.id)) issues.push(`${label}.checks[${index}].id 重复`);
2781
+ checkIds.add(check.id);
2782
+ if (!PREFLIGHT_CHECK_KINDS.has(check.kind)) issues.push(`${label}.checks[${index}].kind 无效`);
2783
+ if (check.command !== undefined) required(check.command, `${label}.checks[${index}].command`, issues);
2784
+ required(check.successCriterion, `${label}.checks[${index}].successCriterion`, issues);
2785
+ }
2786
+ const assets = array(value.assets, `${label}.assets`, issues);
2787
+ const assetIds = new Set();
2788
+ for (const [index, asset] of assets.entries()) {
2789
+ if (!isRecord(asset)) {
2790
+ issues.push(`${label}.assets[${index}] 必须是对象`);
2791
+ continue;
2792
+ }
2793
+ required(asset.id, `${label}.assets[${index}].id`, issues);
2794
+ if (assetIds.has(asset.id)) issues.push(`${label}.assets[${index}].id 重复`);
2795
+ assetIds.add(asset.id);
2796
+ if (!PREFLIGHT_ASSET_KINDS.has(asset.kind)) issues.push(`${label}.assets[${index}].kind 无效`);
2797
+ if (!PREFLIGHT_ASSET_AVAILABILITY.has(asset.availability)) issues.push(`${label}.assets[${index}].availability 无效`);
2798
+ if (typeof asset.required !== "boolean") issues.push(`${label}.assets[${index}].required 必须是布尔值`);
2799
+ required(asset.identity, `${label}.assets[${index}].identity`, issues);
2800
+ required(asset.verification, `${label}.assets[${index}].verification`, issues);
2801
+ if (asset.scientificRole !== undefined && !ASSET_SCIENTIFIC_ROLES.has(asset.scientificRole)) {
2802
+ issues.push(`${label}.assets[${index}].scientificRole 无效`);
2803
+ }
2804
+ validateAssetAcquisition(asset.acquisition, `${label}.assets[${index}].acquisition`, issues);
2805
+ validateAssetSourceEvidence(asset.sourceEvidence, `${label}.assets[${index}].sourceEvidence`, issues);
2806
+ const sourceUrls = asset.sourceUrls === undefined
2807
+ ? []
2808
+ : array(asset.sourceUrls, `${label}.assets[${index}].sourceUrls`, issues);
2809
+ for (const [sourceIndex, sourceUrl] of sourceUrls.entries()) {
2810
+ if (!validPublicAssetSourceUrl(sourceUrl)) {
2811
+ issues.push(`${label}.assets[${index}].sourceUrls[${sourceIndex}] 必须是无凭据的 HTTP(S) URL`);
2812
+ }
2813
+ }
2814
+ if (asset.registryDatasetId !== undefined) {
2815
+ required(asset.registryDatasetId, `${label}.assets[${index}].registryDatasetId`, issues);
2816
+ if (asset.kind !== "dataset") {
2817
+ issues.push(`${label}.assets[${index}].registryDatasetId 只适用于 dataset`);
2818
+ }
2819
+ }
2820
+ if (
2821
+ asset.required === true
2822
+ && asset.availability === "public_unverified"
2823
+ && sourceUrls.length === 0
2824
+ && !textValue(asset.registryDatasetId)
2825
+ ) {
2826
+ issues.push(`${label}.assets[${index}] 是必需的 public_unverified 资产,必须声明 sourceUrls 或 registryDatasetId`);
2827
+ }
2828
+ }
2829
+ if (
2830
+ assets.some((asset) => isRecord(asset) && asset.required === true && asset.kind === "dataset")
2831
+ && !checks.some((check) => isRecord(check) && check.kind === "dataset_loader")
2832
+ ) {
2833
+ issues.push(`${label}.checks 对每个必需 dataset 资产必须包含 dataset_loader,并实际加载配置、split 和代表性行`);
2834
+ }
2835
+ if (
2836
+ assets.some((asset) =>
2837
+ isRecord(asset) && asset.required === true && asset.availability === "public_unverified",
2838
+ )
2839
+ && value.maxMinutes < MIN_PUBLIC_ASSET_PREFLIGHT_MINUTES
2840
+ ) {
2841
+ issues.push(`${label}.maxMinutes 对必需的 public_unverified 资产不得低于 ${MIN_PUBLIC_ASSET_PREFLIGHT_MINUTES}`);
2842
+ }
2843
+ if (value.schemaVersion === "0.3") {
2844
+ validateAssetRequirements(value.requirements, assets, checks, label, issues, agentPlanned && value.source === "defaulted");
2845
+ }
2846
+ }
2847
+
2848
+ function validateAssetRequirements(value, assets, checks, label, issues, deferredBindings = false) {
2849
+ const requirements = array(value, `${label}.requirements`, issues);
2850
+ const kinds = new Set();
2851
+ for (const [index, requirement] of requirements.entries()) {
2852
+ const itemLabel = `${label}.requirements[${index}]`;
2853
+ if (!isRecord(requirement)) {
2854
+ issues.push(`${itemLabel} 必须是对象`);
2855
+ continue;
2856
+ }
2857
+ if (!ASSET_REQUIREMENT_KINDS.includes(requirement.kind)) {
2858
+ issues.push(`${itemLabel}.kind 无效`);
2859
+ }
2860
+ if (kinds.has(requirement.kind)) issues.push(`${itemLabel}.kind 重复`);
2861
+ kinds.add(requirement.kind);
2862
+ if (typeof requirement.required !== "boolean") issues.push(`${itemLabel}.required 必须是布尔值`);
2863
+ required(requirement.reason, `${itemLabel}.reason`, issues);
2864
+ array(requirement.assetIds, `${itemLabel}.assetIds`, issues);
2865
+ array(requirement.checkIds, `${itemLabel}.checkIds`, issues);
2866
+ if (typeof requirement.repository !== "boolean") issues.push(`${itemLabel}.repository 必须是布尔值`);
2867
+ }
2868
+ for (const kind of deferredBindings ? [] : ASSET_REQUIREMENT_KINDS) {
2869
+ if (!kinds.has(kind)) issues.push(`${label}.requirements 缺少 ${kind}`);
2870
+ }
2871
+ const coverage = evaluateAssetRequirementCoverage({ requirements, assets, checks });
2872
+ for (const kind of deferredBindings ? [] : coverage.missingRequirements) {
2873
+ issues.push(`${label}.requirements 的必需项 ${kind} 没有绑定仓库、资产或检查`);
2874
+ }
2875
+ for (const binding of coverage.invalidBindings) {
2876
+ issues.push(`${label}.requirements 引用了不存在或非必需的绑定 ${binding}`);
2877
+ }
2878
+ }
2879
+
2880
+ function validateAssetAcquisition(value, label, issues) {
2881
+ if (value === undefined) return;
2882
+ if (!isRecord(value)) {
2883
+ issues.push(`${label} 必须是对象`);
2884
+ return;
2885
+ }
2886
+ if (!ASSET_ACQUISITION_MODES.has(value.mode)) issues.push(`${label}.mode 无效`);
2887
+ if (value.revision !== undefined) required(value.revision, `${label}.revision`, issues);
2888
+ for (const field of ["requiredPaths", "excludedPaths"]) {
2889
+ if (value[field] === undefined) continue;
2890
+ const patterns = array(value[field], `${label}.${field}`, issues);
2891
+ for (const [index, pattern] of patterns.entries()) {
2892
+ if (typeof pattern !== "string" || !safeAssetPattern(pattern)) {
2893
+ issues.push(`${label}.${field}[${index}] 必须是安全相对路径或 glob`);
2894
+ }
2895
+ }
2896
+ }
2897
+ for (const field of ["expectedDownloadBytes", "expectedExpandedBytes"]) {
2898
+ if (value[field] !== undefined && (!Number.isSafeInteger(value[field]) || value[field] < 0)) {
2899
+ issues.push(`${label}.${field} 必须是非负安全整数`);
2900
+ }
2901
+ }
2902
+ if (
2903
+ value.maximumDownloadBytes !== undefined
2904
+ && (!Number.isSafeInteger(value.maximumDownloadBytes) || value.maximumDownloadBytes < 1)
2905
+ ) {
2906
+ issues.push(`${label}.maximumDownloadBytes 必须是正安全整数`);
2907
+ }
2908
+ if (
2909
+ value.maximumExpandedBytes !== undefined
2910
+ && (!Number.isSafeInteger(value.maximumExpandedBytes) || value.maximumExpandedBytes < 1)
2911
+ ) {
2912
+ issues.push(`${label}.maximumExpandedBytes 必须是正安全整数`);
2913
+ }
2914
+ if (value.supportsStreaming !== undefined && typeof value.supportsStreaming !== "boolean") {
2915
+ issues.push(`${label}.supportsStreaming 必须是布尔值`);
2916
+ }
2917
+ }
2918
+
2919
+ function validateAssetSourceEvidence(value, label, issues) {
2920
+ if (value === undefined) return;
2921
+ const entries = array(value, label, issues);
2922
+ for (const [index, entry] of entries.entries()) {
2923
+ if (!isRecord(entry)) {
2924
+ issues.push(`${label}[${index}] 必须是对象`);
2925
+ continue;
2926
+ }
2927
+ if (!new Set(["paper", "repository", "dataset_registry"]).has(entry.source)) {
2928
+ issues.push(`${label}[${index}].source 无效`);
2929
+ }
2930
+ required(entry.locator, `${label}[${index}].locator`, issues);
2931
+ if (entry.path !== undefined && !safeRelativePath(entry.path)) {
2932
+ issues.push(`${label}[${index}].path 必须是安全相对路径`);
2933
+ }
2934
+ if (entry.commit !== undefined) required(entry.commit, `${label}[${index}].commit`, issues);
2935
+ if (
2936
+ entry.resolver !== undefined
2937
+ && !new Set(["exact_url", "append_path", "huggingface_model", "huggingface_dataset"]).has(entry.resolver)
2938
+ ) {
2939
+ issues.push(`${label}[${index}].resolver 无效`);
2940
+ }
2941
+ if (entry.suffix !== undefined && !safeAssetPattern(entry.suffix)) {
2942
+ issues.push(`${label}[${index}].suffix 必须是安全相对路径`);
2943
+ }
2944
+ if (entry.resolver === "append_path" && !textValue(entry.suffix)) {
2945
+ issues.push(`${label}[${index}].resolver=append_path 时必须声明 suffix`);
2946
+ }
2947
+ if (entry.suffix !== undefined && entry.resolver !== "append_path") {
2948
+ issues.push(`${label}[${index}].suffix 只适用于 resolver=append_path`);
2949
+ }
2950
+ }
2951
+ }
2952
+
2953
+ function validateHardwareBenchmarkProtocol(value, label, issues, { requiresComparator }) {
2954
+ if (!isRecord(value)) {
2955
+ issues.push(`${label} 对硬件敏感指标必须完整声明批量、精度、预热、计时重复、同步、输入构造和确切实现入口`);
2956
+ return;
2957
+ }
2958
+ const batchSizes = Array.isArray(value.batchSize) ? value.batchSize : [value.batchSize];
2959
+ if (!batchSizes.length || batchSizes.some((batchSize) =>
2960
+ !Number.isInteger(batchSize) || batchSize < 1)) {
2961
+ issues.push(`${label}.batchSize 必须是正整数或非空正整数数组`);
2962
+ }
2963
+ required(value.dtype, `${label}.dtype`, issues);
2964
+ if (!Number.isInteger(value.warmupIterations) || value.warmupIterations < 1) {
2965
+ issues.push(`${label}.warmupIterations 必须是正整数`);
2966
+ }
2967
+ if (!Number.isInteger(value.measurementIterations) || value.measurementIterations < 1) {
2968
+ issues.push(`${label}.measurementIterations 必须是正整数`);
2969
+ }
2970
+ required(value.timer, `${label}.timer`, issues);
2971
+ required(value.synchronization, `${label}.synchronization`, issues);
2972
+ required(value.inputConstruction, `${label}.inputConstruction`, issues);
2973
+ const implementations = array(value.implementations, `${label}.implementations`, issues);
2974
+ if (requiresComparator && implementations.length < 2) {
2975
+ issues.push(`${label}.implementations 对相对速度指标必须至少声明被测实现和比较实现`);
2976
+ }
2977
+ const implementationIds = new Set();
2978
+ for (const [index, implementation] of implementations.entries()) {
2979
+ if (!isRecord(implementation)) {
2980
+ issues.push(`${label}.implementations[${index}] 必须是对象`);
2981
+ continue;
2982
+ }
2983
+ required(implementation.id, `${label}.implementations[${index}].id`, issues);
2984
+ required(implementation.callable, `${label}.implementations[${index}].callable`, issues);
2985
+ required(implementation.provenance, `${label}.implementations[${index}].provenance`, issues);
2986
+ if (implementationIds.has(implementation.id)) {
2987
+ issues.push(`${label}.implementations[${index}].id 重复`);
2988
+ }
2989
+ implementationIds.add(implementation.id);
2990
+ }
2991
+ }
2992
+
2993
+ function normalizeClaimReproduction(claim, index, measurements, issues, inventoryOnly = false) {
2994
+ if (inventoryOnly && claim.reproduction?.status === "deferred") {
2995
+ required(claim.reproduction.reason, `claims[${index}].reproduction.reason`, issues);
2996
+ return claim.reproduction;
2997
+ }
2998
+ if (!isRecord(claim.reproduction)) {
2999
+ issues.push(`claims[${index}].reproduction 必须是对象`);
3000
+ return claim.reproduction;
3001
+ }
3002
+ const status = claim.reproduction.status;
3003
+ if (!REPRODUCTION_DISPOSITIONS.has(status)) {
3004
+ issues.push(`claims[${index}].reproduction.status 无效`);
3005
+ return claim.reproduction;
3006
+ }
3007
+ const implementationOrigin = claim.reproduction.implementationOrigin;
3008
+ if (
3009
+ implementationOrigin !== undefined
3010
+ && !new Set([...IMPLEMENTATION_ORIGINS, "insufficient"]).has(implementationOrigin)
3011
+ ) {
3012
+ issues.push(`claims[${index}].reproduction.implementationOrigin 无效`);
3013
+ }
3014
+ if (status === "planned" && implementationOrigin === "insufficient") {
3015
+ issues.push(`claims[${index}] 已计划执行,implementationOrigin 不能是 insufficient`);
3016
+ }
3017
+ if (status === "blocked" && implementationOrigin && implementationOrigin !== "insufficient") {
3018
+ issues.push(`claims[${index}] 已阻塞,implementationOrigin 必须是 insufficient`);
3019
+ }
3020
+ const contextualWithoutMeasurement = claim.type === "method"
3021
+ || claim.type === "limitation"
3022
+ || (Array.isArray(claim.reproduction.derivedFromClaimIds)
3023
+ && claim.reproduction.derivedFromClaimIds.length > 0);
3024
+ if (status === "planned" && measurements.length === 0 && contextualWithoutMeasurement) {
3025
+ return {
3026
+ ...claim.reproduction,
3027
+ status: "not_applicable",
3028
+ reason: typeof claim.reproduction.reason === "string" && claim.reproduction.reason.trim()
3029
+ ? claim.reproduction.reason
3030
+ : "No paper-reported measurement is attached to this claim, so CiteArk keeps it as contextual research structure instead of scheduling an independent measurement-based reproduction.",
3031
+ };
3032
+ }
3033
+ if (
3034
+ (claim.type === "finding" || claim.type === "measurement")
3035
+ && measurements.length === 0
3036
+ && !contextualWithoutMeasurement
3037
+ ) {
3038
+ return {
3039
+ ...claim.reproduction,
3040
+ status: "blocked",
3041
+ implementationOrigin: "insufficient",
3042
+ reason: typeof claim.reproduction.reason === "string" && claim.reproduction.reason.trim()
3043
+ ? claim.reproduction.reason
3044
+ : "This empirical claim has no safely extractable paper-reported measurement or direct decision rule for an automatic experiment.",
3045
+ };
3046
+ }
3047
+ if (status === "not_applicable" && !contextualWithoutMeasurement) {
3048
+ return {
3049
+ ...claim.reproduction,
3050
+ status: "blocked",
3051
+ implementationOrigin: "insufficient",
3052
+ reason: typeof claim.reproduction.reason === "string" && claim.reproduction.reason.trim()
3053
+ ? claim.reproduction.reason
3054
+ : "This empirical claim does not yet have a complete executable measurement binding.",
3055
+ };
3056
+ }
3057
+ if (status !== "planned") {
3058
+ required(claim.reproduction.reason, `claims[${index}].reproduction.reason`, issues);
3059
+ }
3060
+ if (status === "not_applicable") {
3061
+ const derived = claim.reproduction.derivedFromClaimIds;
3062
+ if (derived !== undefined && (!Array.isArray(derived) || derived.some((value) => typeof value !== "string" || !value))) {
3063
+ issues.push(`claims[${index}].reproduction.derivedFromClaimIds 必须是非空字符串数组`);
3064
+ }
3065
+ }
3066
+ return claim.reproduction;
3067
+ }
3068
+
3069
+ function validateCoverageInventory(value, claimIds, sourceIds, issues) {
3070
+ if (!isRecord(value)) {
3071
+ issues.push("coverage 必须是对象");
3072
+ return value;
3073
+ }
3074
+ const items = array(value.items, "coverage.items", issues);
3075
+ if (items.length === 0) issues.push("coverage.items 必须至少包含一项论文经验结果盘点");
3076
+ const itemIds = new Set();
3077
+ items.forEach((item, index) => {
3078
+ const label = `coverage.items[${index}]`;
3079
+ if (!isRecord(item)) return issues.push(`${label} 必须是对象`);
3080
+ required(item.id, `${label}.id`, issues);
3081
+ if (itemIds.has(item.id)) issues.push(`coverage item id 重复:${item.id}`);
3082
+ itemIds.add(item.id);
3083
+ required(item.description, `${label}.description`, issues);
3084
+ validateLocator(item.sourceLocator, `${label}.sourceLocator`, sourceIds, issues);
3085
+ if (!COVERAGE_STATUSES.has(item.status)) issues.push(`${label}.status 无效`);
3086
+ const coveredClaims = array(item.claimIds, `${label}.claimIds`, issues);
3087
+ for (const claimId of coveredClaims) {
3088
+ if (!claimIds.has(claimId)) issues.push(`${label} 引用了不存在的 claim:${claimId}`);
3089
+ }
3090
+ if (item.status === "represented" && coveredClaims.length === 0) {
3091
+ issues.push(`${label}.status=represented 时必须引用至少一个 claim`);
3092
+ }
3093
+ if (item.status !== "represented") required(item.reason, `${label}.reason`, issues);
3094
+ });
3095
+ return value;
3096
+ }
3097
+
3098
+ function validateMeasurement(value, label, issues) {
3099
+ if (!isRecord(value)) {
3100
+ issues.push(`${label} 必须是对象`);
3101
+ return;
3102
+ }
3103
+ required(value.metric, `${label}.metric`, issues);
3104
+ required(value.unit, `${label}.unit`, issues);
3105
+ if (value.value !== undefined && (typeof value.value !== "number" || !Number.isFinite(value.value))) {
3106
+ issues.push(`${label}.value 必须是有限数值`);
3107
+ }
3108
+ if (
3109
+ value.verificationTolerance !== undefined
3110
+ && (
3111
+ typeof value.verificationTolerance !== "number"
3112
+ || !Number.isFinite(value.verificationTolerance)
3113
+ || value.verificationTolerance < 0
3114
+ )
3115
+ ) {
3116
+ issues.push(`${label}.verificationTolerance 必须是非负有限数值`);
3117
+ }
3118
+ if (
3119
+ value.evidenceBasis !== undefined
3120
+ && !new Set(["reported_numeric", "digitized_figure", "derived_reported_relation"])
3121
+ .has(value.evidenceBasis)
3122
+ ) {
3123
+ issues.push(`${label}.evidenceBasis 无效`);
3124
+ }
3125
+ if (
3126
+ value.extractionUncertainty !== undefined
3127
+ && (
3128
+ typeof value.extractionUncertainty !== "number"
3129
+ || !Number.isFinite(value.extractionUncertainty)
3130
+ || value.extractionUncertainty < 0
3131
+ )
3132
+ ) {
3133
+ issues.push(`${label}.extractionUncertainty 必须是非负有限数值`);
3134
+ }
3135
+ if (
3136
+ value.evidenceBasis === "digitized_figure"
3137
+ && value.extractionUncertainty === undefined
3138
+ ) {
3139
+ issues.push(`${label}.evidenceBasis=digitized_figure 时必须声明 extractionUncertainty`);
3140
+ }
3141
+ }
3142
+
3143
+ function validateLocator(value, label, sourceIds, issues) {
3144
+ if (!isRecord(value)) {
3145
+ issues.push(`${label} 必须是对象`);
3146
+ return;
3147
+ }
3148
+ required(value.sourceId, `${label}.sourceId`, issues);
3149
+ required(value.locator, `${label}.locator`, issues);
3150
+ if (value.sourceId && !sourceIds.has(value.sourceId)) issues.push(`${label}.sourceId 不存在:${value.sourceId}`);
3151
+ }
3152
+
3153
+ function array(value, label, issues) {
3154
+ if (!Array.isArray(value)) {
3155
+ issues.push(`${label} 必须是数组`);
3156
+ return [];
3157
+ }
3158
+ return value;
3159
+ }
3160
+
3161
+ function required(value, label, issues) {
3162
+ if (typeof value !== "string" || !value.trim()) issues.push(`${label} 必须是非空字符串`);
3163
+ }