@citeark/agent 0.3.20

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (347) hide show
  1. package/LICENSE +202 -0
  2. package/README.md +128 -0
  3. package/data/dataset-source-registry.v1.json +300 -0
  4. package/dist/arkgraph/boot.js +6 -0
  5. package/dist/arkgraph/index.html +1 -0
  6. package/dist/arkgraph/viewer.css +1 -0
  7. package/dist/arkgraph/viewer.en.css +1 -0
  8. package/dist/arkgraph/viewer.en.js +49 -0
  9. package/dist/arkgraph/viewer.en.js.LEGAL.txt +56 -0
  10. package/dist/arkgraph/viewer.js +49 -0
  11. package/dist/arkgraph/viewer.js.LEGAL.txt +56 -0
  12. package/docker/claude-code/Dockerfile +97 -0
  13. package/docker/claude-code/codex-pro-relay.mjs +466 -0
  14. package/docker/claude-code/runtime-contract-check.mjs +79 -0
  15. package/docs/arkgraph-reading.md +79 -0
  16. package/docs/configuration.md +100 -0
  17. package/docs/integration.md +92 -0
  18. package/docs/maturity-plan.md +27 -0
  19. package/docs/npm-release.md +44 -0
  20. package/docs/paper-reading.md +40 -0
  21. package/docs/research-plan-granularity.md +27 -0
  22. package/docs/terminal.md +49 -0
  23. package/examples/toy-evaluation/compile-task.json +27 -0
  24. package/examples/toy-evaluation/paper.md +5 -0
  25. package/examples/toy-evaluation/repository/README.md +9 -0
  26. package/examples/toy-evaluation/repository/checkpoint.json +4 -0
  27. package/examples/toy-evaluation/repository/evaluate.py +17 -0
  28. package/examples/toy-evaluation/task.json +81 -0
  29. package/package.json +59 -0
  30. package/prompts/compile-research.md +58 -0
  31. package/prompts/execute-contract.md +72 -0
  32. package/prompts/execute-workspace-simple.md +51 -0
  33. package/prompts/execute-workspace.md +34 -0
  34. package/prompts/prepare-reproduction.md +82 -0
  35. package/prompts/repair-research.md +45 -0
  36. package/protocol/CAP.md +129 -0
  37. package/protocol/LICENSE +12 -0
  38. package/protocol/MAPPINGS.md +72 -0
  39. package/protocol/README.md +38 -0
  40. package/protocol/conformance-v2.0-alpha.1.json +36 -0
  41. package/protocol/examples/arkgraph/checkpoint-evaluation.json +309 -0
  42. package/protocol/examples/arkgraph/fixtures.mjs +49 -0
  43. package/protocol/examples/arkgraph/paper-free.json +291 -0
  44. package/protocol/examples/arkgraph/partial-failure.json +344 -0
  45. package/protocol/examples/arkgraph/training-evaluation.json +443 -0
  46. package/protocol/profiles/agent-trace.md +16 -0
  47. package/protocol/profiles/computational-run.md +16 -0
  48. package/protocol/profiles/core.md +15 -0
  49. package/protocol/profiles/public-bundle.md +18 -0
  50. package/protocol/profiles/reproduction.md +29 -0
  51. package/protocol/profiles/research-compilation.md +44 -0
  52. package/protocol/profiles/research-plan.md +39 -0
  53. package/protocol/profiles/restricted-evidence.md +15 -0
  54. package/runtime/bootstrap-autodl-runtime.sh +314 -0
  55. package/runtime/create-runtime-venv.sh +41 -0
  56. package/runtime/install-local-cpu-runtime.sh +23 -0
  57. package/runtime/install-scientific-runtime.sh +153 -0
  58. package/runtime/mineru/parse.py +62 -0
  59. package/runtime/mineru/requirements.txt +4 -0
  60. package/runtime/requirements-baseline.txt +38 -0
  61. package/schemas/cap/v2/activity.schema.json +47 -0
  62. package/schemas/cap/v2/agent.schema.json +32 -0
  63. package/schemas/cap/v2/assertion.schema.json +110 -0
  64. package/schemas/cap/v2/descriptor.schema.json +243 -0
  65. package/schemas/cap/v2/entity.schema.json +64 -0
  66. package/schemas/cap/v2/manifest.schema.json +67 -0
  67. package/schemas/cap/v2/relation.schema.json +82 -0
  68. package/schemas/compute-catalog.schema.json +63 -0
  69. package/schemas/compute-decision.schema.json +27 -0
  70. package/schemas/execution-contract.schema.json +1024 -0
  71. package/schemas/research-card.schema.json +30 -0
  72. package/schemas/research-inventory-draft.schema.json +366 -0
  73. package/schemas/research.schema.json +1044 -0
  74. package/schemas/result.schema.json +173 -0
  75. package/schemas/verification-policy.schema.json +47 -0
  76. package/schemas/verified-conclusion.schema.json +58 -0
  77. package/schemas/workspace-summary.schema.json +24 -0
  78. package/scripts/build-arkgraph-view.mjs +12 -0
  79. package/scripts/check-execution-feasibility.mjs +24 -0
  80. package/scripts/check-syntax.mjs +15 -0
  81. package/scripts/deterministic-asset-preparation.py +438 -0
  82. package/scripts/package-cap.mjs +23 -0
  83. package/scripts/package-local-agent.mjs +23 -0
  84. package/scripts/preview-arkgraph.mjs +25 -0
  85. package/scripts/replay-research-compiler-candidate.mjs +134 -0
  86. package/scripts/review-compiler-sources.mjs +44 -0
  87. package/scripts/run-asset-preparation.sh +17 -0
  88. package/scripts/run-research-plan.mjs +98 -0
  89. package/scripts/validate-asset-preparation.py +290 -0
  90. package/scripts/verify-local-runtime.mjs +57 -0
  91. package/scripts/verify-npm-package.mjs +57 -0
  92. package/src/adapters/paper2agent.mjs +107 -0
  93. package/src/assets/cache.mjs +159 -0
  94. package/src/assets/compute.mjs +98 -0
  95. package/src/assets/executor.mjs +145 -0
  96. package/src/assets/lifecycle.mjs +213 -0
  97. package/src/assets/manifest.mjs +242 -0
  98. package/src/assets/opportunistic-preparation.mjs +81 -0
  99. package/src/assets/plan.mjs +411 -0
  100. package/src/assets/prompts.mjs +29 -0
  101. package/src/assets/public-asset-probe.mjs +525 -0
  102. package/src/assets/qualification.mjs +119 -0
  103. package/src/assets/readiness.mjs +130 -0
  104. package/src/assets/reproduction-admission.mjs +355 -0
  105. package/src/assets/requirements.mjs +152 -0
  106. package/src/assets/source-grounding.mjs +341 -0
  107. package/src/assets/source-policy.mjs +118 -0
  108. package/src/autodl/client.mjs +260 -0
  109. package/src/autodl/ssh.mjs +380 -0
  110. package/src/autodl/tools.mjs +129 -0
  111. package/src/cap/redaction.mjs +38 -0
  112. package/src/cap/v2/archive.mjs +152 -0
  113. package/src/cap/v2/attestation.mjs +204 -0
  114. package/src/cap/v2/canonical-json.mjs +114 -0
  115. package/src/cap/v2/compilation-artifact.mjs +240 -0
  116. package/src/cap/v2/core.mjs +282 -0
  117. package/src/cap/v2/measurement-assessment-records.mjs +23 -0
  118. package/src/cap/v2/pipeline-artifact.mjs +922 -0
  119. package/src/cap/v2/read.mjs +41 -0
  120. package/src/cap/v2/reassessment-artifact.mjs +383 -0
  121. package/src/cap/v2/research-artifact.mjs +231 -0
  122. package/src/cap/v2/research-map-records.mjs +46 -0
  123. package/src/cap/v2/research-object-records.mjs +163 -0
  124. package/src/cap/v2/research-records.mjs +187 -0
  125. package/src/cap/v2/verify.mjs +642 -0
  126. package/src/cli.mjs +1146 -0
  127. package/src/compute/autodl-pro-compiler.mjs +347 -0
  128. package/src/compute/autodl-pro-executor.mjs +459 -0
  129. package/src/compute/autodl-pro-job.mjs +843 -0
  130. package/src/compute/autodl-pro-network.mjs +295 -0
  131. package/src/compute/autodl-pro-remote.mjs +810 -0
  132. package/src/compute/autodl-pro-staging.mjs +117 -0
  133. package/src/compute/campaign.mjs +110 -0
  134. package/src/compute/catalog.mjs +123 -0
  135. package/src/compute/checkpoint-protocol.mjs +154 -0
  136. package/src/compute/codex-account-lock.mjs +111 -0
  137. package/src/compute/codex-account-session.mjs +107 -0
  138. package/src/compute/compiler-profile.mjs +38 -0
  139. package/src/compute/compiler-router.mjs +23 -0
  140. package/src/compute/coordinator-recovery.mjs +210 -0
  141. package/src/compute/executor-router.mjs +29 -0
  142. package/src/compute/gcp-batch-compiler.mjs +685 -0
  143. package/src/compute/gcp-batch-executor.mjs +1215 -0
  144. package/src/compute/gcp-batch-failure.mjs +92 -0
  145. package/src/compute/gcp-batch-job.mjs +527 -0
  146. package/src/compute/gcp-batch-lifecycle.mjs +81 -0
  147. package/src/compute/gcp-checkpoint-worker.mjs +1633 -0
  148. package/src/compute/local-codex-compiler.mjs +52 -0
  149. package/src/compute/measurement-hardware.mjs +128 -0
  150. package/src/compute/remote-attempt.mjs +226 -0
  151. package/src/compute/requirements.mjs +124 -0
  152. package/src/compute/research-phases.mjs +48 -0
  153. package/src/compute/scheduler.mjs +452 -0
  154. package/src/compute/shared-workloads.mjs +26 -0
  155. package/src/compute/stage-archive.mjs +79 -0
  156. package/src/contracts/campaign-contract.mjs +52 -0
  157. package/src/contracts/execution-contract.mjs +819 -0
  158. package/src/contracts/execution-mode.mjs +19 -0
  159. package/src/contracts/execution-timeouts.mjs +45 -0
  160. package/src/contracts/execution-workload.mjs +68 -0
  161. package/src/contracts/preflight-schema.mjs +25 -0
  162. package/src/contracts/public-contract.mjs +63 -0
  163. package/src/contracts/subject-tags.mjs +31 -0
  164. package/src/dashboard/data.mjs +898 -0
  165. package/src/dashboard/server.mjs +79 -0
  166. package/src/dashboard/static/dashboard.css +366 -0
  167. package/src/dashboard/static/dashboard.js +560 -0
  168. package/src/dashboard/static/index.html +85 -0
  169. package/src/deployment/community-policy.mjs +9 -0
  170. package/src/deployment/environment.mjs +112 -0
  171. package/src/deployment/guided.mjs +98 -0
  172. package/src/deployment/handoff.mjs +102 -0
  173. package/src/deployment/local-contract.mjs +31 -0
  174. package/src/deployment/local.mjs +100 -0
  175. package/src/deployment/prepare.mjs +46 -0
  176. package/src/deployment/recipe.mjs +108 -0
  177. package/src/deployment/supplement.mjs +51 -0
  178. package/src/deployment/terminal.mjs +43 -0
  179. package/src/diagnosis/renderer.mjs +75 -0
  180. package/src/diagnosis/target-failure.mjs +46 -0
  181. package/src/evidence/parser-registry.mjs +54 -0
  182. package/src/evidence/parsers/fasttext-classification.mjs +82 -0
  183. package/src/evidence/parsers/json-scalar.mjs +96 -0
  184. package/src/evidence/parsers/simcse-senteval.mjs +104 -0
  185. package/src/evidence/parsers/starspace-classification.mjs +78 -0
  186. package/src/evidence/registry.mjs +147 -0
  187. package/src/execution/runner-audit.mjs +473 -0
  188. package/src/gcp/auth.mjs +106 -0
  189. package/src/gcp/batch-client.mjs +120 -0
  190. package/src/gcp/resource-discovery.mjs +177 -0
  191. package/src/gcp/rest.mjs +82 -0
  192. package/src/gcp/secret-manager.mjs +34 -0
  193. package/src/gcp/signed-url.mjs +133 -0
  194. package/src/gcp/storage.mjs +220 -0
  195. package/src/graph/command.mjs +41 -0
  196. package/src/graph/execution.mjs +97 -0
  197. package/src/graph/model.mjs +37 -0
  198. package/src/graph/presentation.mjs +110 -0
  199. package/src/graph/query.mjs +159 -0
  200. package/src/graph/research-relations.mjs +69 -0
  201. package/src/graph/source-page.mjs +12 -0
  202. package/src/graph/source-preview.mjs +34 -0
  203. package/src/graph/validate.mjs +76 -0
  204. package/src/job.mjs +496 -0
  205. package/src/network/autodl-routing-proxy.mjs +462 -0
  206. package/src/network/egress-proxy.mjs +158 -0
  207. package/src/observability/event-contract.mjs +230 -0
  208. package/src/observability/pipeline-monitor.mjs +166 -0
  209. package/src/pipeline/orchestrator.mjs +1281 -0
  210. package/src/pipeline/recovery-error.mjs +11 -0
  211. package/src/pipeline/replay.mjs +304 -0
  212. package/src/pipeline/shared-execution.mjs +115 -0
  213. package/src/pipeline/stage-checkpoint.mjs +86 -0
  214. package/src/pipeline/stage-recovery.mjs +101 -0
  215. package/src/pipeline/targets.mjs +110 -0
  216. package/src/process.mjs +143 -0
  217. package/src/protocol.mjs +312 -0
  218. package/src/provider/codex-account.mjs +44 -0
  219. package/src/provider/codex-completion.mjs +49 -0
  220. package/src/provider/completion.mjs +292 -0
  221. package/src/provider/model-client.mjs +44 -0
  222. package/src/provider/model-route.mjs +29 -0
  223. package/src/provider/openrouter-readiness.mjs +189 -0
  224. package/src/provider/reader-bridge.mjs +35 -0
  225. package/src/provider/relay.mjs +263 -0
  226. package/src/provider/runtime-auth.mjs +40 -0
  227. package/src/public/cap.d.mts +90 -0
  228. package/src/public/cap.mjs +12 -0
  229. package/src/public/contracts.d.mts +2 -0
  230. package/src/public/host.mjs +171 -0
  231. package/src/public/operations.d.mts +11 -0
  232. package/src/public/presentation.d.mts +4 -0
  233. package/src/records/views.mjs +26 -0
  234. package/src/remote/command.mjs +178 -0
  235. package/src/remote/ssh.mjs +59 -0
  236. package/src/repository-origin.mjs +81 -0
  237. package/src/reproduction/evidence-feedback.mjs +96 -0
  238. package/src/reproduction/incomplete-initialization.mjs +25 -0
  239. package/src/reproduction/lifecycle.mjs +253 -0
  240. package/src/reproduction/plan.mjs +132 -0
  241. package/src/reproduction/prompts.mjs +70 -0
  242. package/src/reproduction/runner.mjs +188 -0
  243. package/src/reproduction/summary.mjs +130 -0
  244. package/src/reproduction/workspace-mode.mjs +7 -0
  245. package/src/research/automatic-admission.mjs +156 -0
  246. package/src/research/compiler-coverage.mjs +85 -0
  247. package/src/research/compiler-failure.mjs +24 -0
  248. package/src/research/compiler-normalization-guards.mjs +112 -0
  249. package/src/research/compiler-repair.mjs +3 -0
  250. package/src/research/compiler.mjs +853 -0
  251. package/src/research/continuation-selection.mjs +26 -0
  252. package/src/research/execution-graph-context.mjs +43 -0
  253. package/src/research/experiment-importance.mjs +15 -0
  254. package/src/research/inventory-handoff.mjs +104 -0
  255. package/src/research/inventory-revisions.mjs +32 -0
  256. package/src/research/mineru-local.mjs +73 -0
  257. package/src/research/paper-command.mjs +19 -0
  258. package/src/research/paper-markdown.mjs +180 -0
  259. package/src/research/paper-source-map.mjs +69 -0
  260. package/src/research/planning-policy.mjs +88 -0
  261. package/src/research/reference-materials.mjs +11 -0
  262. package/src/research/reproduction-scope.mjs +30 -0
  263. package/src/research/research-map.mjs +94 -0
  264. package/src/research/research-objects.mjs +88 -0
  265. package/src/research/source-discovery.mjs +646 -0
  266. package/src/research/source-observations.mjs +75 -0
  267. package/src/research/source-review-cli-mcp.mjs +26 -0
  268. package/src/research/source-review-input.mjs +209 -0
  269. package/src/research/source-review-local-codex.mjs +36 -0
  270. package/src/research/source-review-model.mjs +70 -0
  271. package/src/research/source-review.mjs +173 -0
  272. package/src/research/structure.mjs +3163 -0
  273. package/src/research-card/renderer.mjs +277 -0
  274. package/src/research-card/verified-conclusion.mjs +143 -0
  275. package/src/results/output-registry.mjs +183 -0
  276. package/src/runtime/claude-code.mjs +52 -0
  277. package/src/runtime/codex-capacity-retry.mjs +87 -0
  278. package/src/runtime/codex.mjs +64 -0
  279. package/src/runtime/config.mjs +157 -0
  280. package/src/runtime/final-output.mjs +40 -0
  281. package/src/runtime/index.mjs +21 -0
  282. package/src/runtime/local-codex.mjs +74 -0
  283. package/src/runtime/opencode.mjs +95 -0
  284. package/src/runtime/prompt.mjs +13 -0
  285. package/src/sandbox/docker.mjs +363 -0
  286. package/src/settings/command.mjs +297 -0
  287. package/src/settings/store.mjs +119 -0
  288. package/src/telemetry/pricing.mjs +68 -0
  289. package/src/telemetry/usage.mjs +265 -0
  290. package/src/terminal/events.mjs +97 -0
  291. package/src/terminal/input.mjs +40 -0
  292. package/src/terminal/plain.mjs +40 -0
  293. package/src/terminal/remote-stream.mjs +22 -0
  294. package/src/terminal/screen.mjs +214 -0
  295. package/src/terminal/transcript.mjs +69 -0
  296. package/src/util.mjs +107 -0
  297. package/src/verification/ai-assessor.mjs +534 -0
  298. package/src/verification/claim-evaluator.mjs +242 -0
  299. package/src/verification/evidence-context.mjs +165 -0
  300. package/src/verification/evidence-reader.mjs +95 -0
  301. package/src/verification/integrity.mjs +570 -0
  302. package/src/verification/tolerance.mjs +32 -0
  303. package/src/workloads/cpu-research-preparation.mjs +56 -0
  304. package/src/workloads/definition.mjs +74 -0
  305. package/src/workloads/phase-aware-reproduction.mjs +46 -0
  306. package/src/workloads/reproduction.mjs +85 -0
  307. package/src/workspace/command.mjs +242 -0
  308. package/src/workspace/control.mjs +49 -0
  309. package/src/workspace/entry.mjs +28 -0
  310. package/src/workspace/input.mjs +93 -0
  311. package/src/workspace/interactive.mjs +94 -0
  312. package/src/workspace/jobs.mjs +418 -0
  313. package/src/workspace/session.mjs +97 -0
  314. package/src/workspace/worker.mjs +137 -0
  315. package/ui/arkgraph/ambient-motion.mjs +10 -0
  316. package/ui/arkgraph/app.jsx +153 -0
  317. package/ui/arkgraph/boot.js +6 -0
  318. package/ui/arkgraph/camera-motion.mjs +20 -0
  319. package/ui/arkgraph/context-reveal.mjs +39 -0
  320. package/ui/arkgraph/details.css +3 -0
  321. package/ui/arkgraph/entry.jsx +28 -0
  322. package/ui/arkgraph/experiment-curves.mjs +17 -0
  323. package/ui/arkgraph/experiment-selection.mjs +15 -0
  324. package/ui/arkgraph/experiment-style.css +26 -0
  325. package/ui/arkgraph/experiment-ui.jsx +32 -0
  326. package/ui/arkgraph/frame.html +1 -0
  327. package/ui/arkgraph/graph-gestures.mjs +62 -0
  328. package/ui/arkgraph/label-layout.mjs +57 -0
  329. package/ui/arkgraph/locales/en.json +229 -0
  330. package/ui/arkgraph/locales/source-types.json +15 -0
  331. package/ui/arkgraph/localization-build.mjs +27 -0
  332. package/ui/arkgraph/material-build.mjs +23 -0
  333. package/ui/arkgraph/material-colors.mjs +39 -0
  334. package/ui/arkgraph/material-style.css +15 -0
  335. package/ui/arkgraph/open-graph.jsx +326 -0
  336. package/ui/arkgraph/outline.jsx +49 -0
  337. package/ui/arkgraph/package-lock.json +888 -0
  338. package/ui/arkgraph/package.json +17 -0
  339. package/ui/arkgraph/reading-layout.mjs +130 -0
  340. package/ui/arkgraph/reading-presentation.mjs +73 -0
  341. package/ui/arkgraph/record-detail.css +51 -0
  342. package/ui/arkgraph/record-details.jsx +29 -0
  343. package/ui/arkgraph/research-types.mjs +31 -0
  344. package/ui/arkgraph/selection-mark.jsx +6 -0
  345. package/ui/arkgraph/soft-spine.mjs +26 -0
  346. package/ui/arkgraph/steering-style.css +187 -0
  347. package/ui/arkgraph/style.css +272 -0
@@ -0,0 +1,253 @@
1
+ import { stageReferenceMaterials } from "../research/reference-materials.mjs";
2
+ import { hasResearchWorkspace } from "./workspace-mode.mjs";
3
+ import path from "node:path";
4
+ import { lstat, rename } from "node:fs/promises";
5
+ import { restoreRunRecovery } from "../compute/coordinator-recovery.mjs";
6
+ import { preserveIncompleteInitialization } from './incomplete-initialization.mjs';
7
+ import { initialResearchPlan, resolveResearchPlan, stageResearchPlanTools, PLAN_FILE } from "./plan.mjs";
8
+
9
+ import { redactSensitive } from "../cap/redaction.mjs";
10
+ import { validateComputeDecision } from "../compute/scheduler.mjs";
11
+ import { buildRunnerAudit } from "../execution/runner-audit.mjs";
12
+ import { ensureFinalWorkspaceCapture, prepareRunBundle, updateRunState } from "../job.mjs";
13
+ import { loadAndNormalizeTask, validateResultFile } from "../protocol.mjs";
14
+ import { publicAgentRuntime, resolveAgentRuntime } from "../runtime/config.mjs";
15
+ import { readAgentUsage } from "../telemetry/usage.mjs";
16
+ import { CiteArkError, writeJson } from "../util.mjs";
17
+
18
+ export async function prepareReproduction({
19
+ taskPath,
20
+ referenceMaterialsPath,
21
+ runsDirectory,
22
+ imageOverride,
23
+ repositoryOverride,
24
+ paperOverride,
25
+ computeDecision,
26
+ runId,
27
+ agentRuntime,
28
+ requireApi = true,
29
+ sandboxProvider = "docker",
30
+ validateScientificComputeDecision = true,
31
+ }) {
32
+ if (computeDecision && imageOverride) {
33
+ throw new CiteArkError("使用 compute decision 时不能再用 imageOverride 改写已记录的执行环境");
34
+ }
35
+ const loaded = await loadAndNormalizeTask(taskPath);
36
+ if (computeDecision && validateScientificComputeDecision) {
37
+ validateComputeDecision(computeDecision, loaded.task);
38
+ }
39
+ const runtimeAgent = resolveAgentRuntime({
40
+ taskAgent: loaded.task.agent,
41
+ ...agentRuntime,
42
+ requireApi,
43
+ });
44
+ const restored = await restoreRunRecovery({ runsDirectory, runId, task: loaded.task });
45
+ if (!restored && runId) await preserveIncompleteInitialization({ runsDirectory, runId, task: loaded.task });
46
+ const bundle = restored ?? await prepareRunBundle({
47
+ ...loaded,
48
+ runsDirectory,
49
+ runId,
50
+ imageOverride,
51
+ repositoryOverride,
52
+ paperOverride,
53
+ });
54
+ if (restored && hasResearchWorkspace(bundle.runtimeTask)) {
55
+ await stageResearchPlanTools(bundle.directories.input);
56
+ }
57
+ await stageReferenceMaterials(referenceMaterialsPath, bundle.directories.input);
58
+ bundle.runtimeAgent = runtimeAgent;
59
+ bundle.computeDecision = computeDecision ?? null;
60
+ bundle.executionEnvironment = structuredClone(computeDecision?.environment ?? bundle.runtimeTask.environment);
61
+ bundle.sandboxProvider = sandboxProvider;
62
+ if (computeDecision) {
63
+ await writeJson(path.join(bundle.directories.input, "compute-decision.json"), computeDecision);
64
+ }
65
+ await updateRunState(bundle, {
66
+ runtime: runtimeAgent.runtime,
67
+ agent: publicAgentRuntime(runtimeAgent),
68
+ sandbox: sandboxProvider,
69
+ image: bundle.executionEnvironment.image,
70
+ executionEnvironment: bundle.executionEnvironment,
71
+ computeDecision: computeDecision ?? null,
72
+ });
73
+ return bundle;
74
+ }
75
+
76
+ export async function finalizeReproduction({
77
+ bundle,
78
+ validation,
79
+ attempts,
80
+ timedOut = false,
81
+ sandboxImageIdentity = null,
82
+ cloudExecution = null,
83
+ }) {
84
+ bundle.sandboxImageIdentity = sandboxImageIdentity;
85
+ await ensureFinalWorkspaceCapture(bundle);
86
+ const usage = await readAgentUsage(path.join(bundle.directories.execution, "trace.jsonl"), {
87
+ runtimeHome: bundle.directories.runtimeHome,
88
+ model: bundle.runtimeAgent.model,
89
+ });
90
+ const runStatus = timedOut ? "timed_out" : validation.issues.length === 0 ? "completed" : "invalid_result";
91
+ await updateRunState(bundle, {
92
+ status: runStatus,
93
+ finishedAt: new Date().toISOString(),
94
+ attempts,
95
+ validationIssues: validation.issues,
96
+ resultRecovery: validation.recovery ?? null,
97
+ verificationStatus: validation.protocolVerification?.status ?? null,
98
+ ...(validation.executionPlanDigest ? { executionPlanDigest: validation.executionPlanDigest } : {}),
99
+ sandboxImageIdentity: bundle.sandboxImageIdentity,
100
+ cloudExecution,
101
+ usage,
102
+ });
103
+ const runnerAudit = await buildRunnerAudit({
104
+ runDirectory: bundle.directories.run,
105
+ runtime: bundle.runtimeAgent.runtime,
106
+ sandbox: bundle.sandboxProvider ?? "docker",
107
+ });
108
+ await updateRunState(bundle, {
109
+ runnerAudit: {
110
+ status: runnerAudit.audit.status,
111
+ auditDigest: runnerAudit.audit.auditDigest,
112
+ captureBoundary: runnerAudit.audit.captureBoundary,
113
+ commandRecordCount: runnerAudit.audit.commandRecordCount,
114
+ },
115
+ });
116
+ if (validation.issues.length > 0) {
117
+ throw new CiteArkError(`最终结果未通过协议校验:\n- ${validation.issues.join("\n- ")}`, { exitCode: 2 });
118
+ }
119
+ return { dryRun: false, bundle, runStatus, validation, attempts, runnerAudit };
120
+ }
121
+
122
+ /**
123
+ * Preserve an honest execution record when semantic retries are exhausted but
124
+ * the remaining problem is the Agent's result envelope. This never creates a
125
+ * metric or success verdict; the normal verifier will classify it as
126
+ * inconclusive from a platform-authored failure record.
127
+ */
128
+ export async function recoverInvalidExecutionResult({
129
+ bundle,
130
+ validation,
131
+ attempts,
132
+ timedOut = false,
133
+ }) {
134
+ if (
135
+ validation.issues.length === 0
136
+ || bundle.runtimeTask?.kind !== "citeark.execution-contract"
137
+ ) {
138
+ return validation;
139
+ }
140
+ const evidencePath = "citeark-execution-failure.json";
141
+ const originalIssues = validation.issues.map((issue) => String(issue));
142
+ let rejectedPlan = null;
143
+ if (hasResearchWorkspace(bundle.runtimeTask)) {
144
+ try { await resolveResearchPlan(bundle.runtimeTask, bundle.directories.output); }
145
+ catch {
146
+ const filename = path.join(bundle.directories.output, PLAN_FILE);
147
+ const stat = await lstat(filename).catch(() => null);
148
+ if (stat?.isFile()) {
149
+ rejectedPlan = "rejected-research-plan.json";
150
+ await rename(filename, path.join(bundle.directories.output, rejectedPlan));
151
+ }
152
+ if (!stat || stat.isFile()) {
153
+ const fallback = initialResearchPlan(bundle.runtimeTask);
154
+ fallback.reason = "Platform failure-envelope recovery after invalid plan; no scientific success is asserted. See citeark-execution-failure.json.";
155
+ await writeJson(filename, fallback);
156
+ }
157
+ }
158
+ }
159
+ await writeJson(
160
+ path.join(bundle.directories.output, evidencePath),
161
+ redactSensitive({
162
+ schemaVersion: "1.0",
163
+ kind: "citeark.execution-failure",
164
+ runId: bundle.runId,
165
+ contractDigest: bundle.runtimeTask.contractDigest,
166
+ timedOut,
167
+ attempts,
168
+ resultValidationIssues: originalIssues,
169
+ recordedAt: new Date().toISOString(),
170
+ }),
171
+ );
172
+
173
+ const submitted = validation.result && typeof validation.result === "object"
174
+ ? validation.result
175
+ : {};
176
+ const submittedExperiment = submitted.experiment && typeof submitted.experiment === "object"
177
+ ? submitted.experiment
178
+ : {};
179
+ const commands = Array.isArray(submittedExperiment.commands)
180
+ ? submittedExperiment.commands.filter(
181
+ (command) => typeof command === "string" && command.trim(),
182
+ )
183
+ : [];
184
+ const reason = timedOut
185
+ ? "The execution exhausted its time budget before producing a valid evidence envelope."
186
+ : "The execution did not produce a valid evidence envelope after the available semantic repair attempts.";
187
+ const diagnosis = [
188
+ typeof submitted.diagnosis === "string" ? submitted.diagnosis.trim() : "",
189
+ "CiteArk preserved this run as an honest failed execution instead of discarding the paper-level workflow because of result-envelope errors.",
190
+ `Authoritative validation issues: ${originalIssues.join("; ")}`,
191
+ ].filter(Boolean).join("\n\n");
192
+ await writeJson(
193
+ path.join(bundle.directories.output, "result.json"),
194
+ redactSensitive({
195
+ schemaVersion: "1.0",
196
+ execution: {
197
+ status: "failed",
198
+ summary: reason,
199
+ failure: {
200
+ category: timedOut ? "timeout" : "execution_error",
201
+ stage: "unknown",
202
+ retryable: false,
203
+ reason: `${reason} Fix the listed submission errors using preserved evidence; do not start another experiment attempt for this envelope.`,
204
+ },
205
+ },
206
+ ...(bundle.runtimeTask.protocol?.preflight ? {
207
+ preflight: {
208
+ source: "platform_recovery",
209
+ status: "failed",
210
+ evidencePath,
211
+ checks: (bundle.runtimeTask.protocol.preflight.checks ?? []).map((check) => ({
212
+ id: check.id,
213
+ status: "failed",
214
+ detail: "The Agent did not submit a protocol-valid preflight record; see the platform recovery evidence.",
215
+ })),
216
+ },
217
+ } : {}),
218
+ experiment: {
219
+ commands,
220
+ entrypoint: typeof bundle.runtimeTask.protocol?.entrypoint === "string"
221
+ ? bundle.runtimeTask.protocol.entrypoint
222
+ : null,
223
+ checkpoint: typeof submittedExperiment.checkpoint === "string"
224
+ ? submittedExperiment.checkpoint
225
+ : null,
226
+ },
227
+ codeChanges: [],
228
+ evidence: [{
229
+ path: evidencePath,
230
+ description: "CiteArk platform record of execution attempts and result-envelope validation failures",
231
+ }, ...(rejectedPlan ? [{ path: rejectedPlan, description: "Rejected Agent plan retained without promoting its interpretation" }] : [])],
232
+ outputs: [],
233
+ limitations: [
234
+ "No trustworthy claim metric is asserted by this conservative recovery record.",
235
+ "Agent-declared evidence and outputs were not promoted because their final envelope did not pass authoritative validation.",
236
+ ],
237
+ diagnosis,
238
+ }),
239
+ );
240
+ const recovered = await validateResultFile(
241
+ path.join(bundle.directories.output, "result.json"),
242
+ bundle.runtimeTask,
243
+ bundle.directories.output,
244
+ );
245
+ return {
246
+ ...recovered,
247
+ recovery: {
248
+ kind: "conservative-failure-envelope",
249
+ evidencePath,
250
+ originalValidationIssues: redactSensitive(originalIssues),
251
+ },
252
+ };
253
+ }
@@ -0,0 +1,132 @@
1
+ import path from "node:path";
2
+ import { cp, lstat, mkdir, readFile, readdir, writeFile } from "node:fs/promises";
3
+ import { validateEvidenceParserDescriptor } from "../evidence/parser-registry.mjs";
4
+ import { CiteArkError, pathExists, sha256Value, writeJson } from "../util.mjs";
5
+ import { hasResearchWorkspace } from "./workspace-mode.mjs";
6
+
7
+ export const PLAN_FILE = "research-plan.json";
8
+
9
+ // Stage the same implementation used by the coordinator, not a second schema
10
+ // validator that can disagree with it. All dependencies here use Node builtins.
11
+ export async function stageResearchPlanTools(inputDirectory) {
12
+ for (const relative of ["scripts/run-research-plan.mjs", "src/reproduction/plan.mjs", "src/reproduction/workspace-mode.mjs", "src/reproduction/evidence-feedback.mjs",
13
+ "src/util.mjs", "src/evidence/parser-registry.mjs", "src/evidence/parsers"]) {
14
+ const destination = path.join(inputDirectory, "runtime-tools", relative);
15
+ await mkdir(path.dirname(destination), { recursive: true });
16
+ await cp(new URL(`../../${relative}`, import.meta.url), destination, { recursive: true });
17
+ }
18
+ await writeFile(path.join(inputDirectory, "run-research-plan.mjs"),
19
+ 'import "./runtime-tools/scripts/run-research-plan.mjs";\n');
20
+ }
21
+
22
+ export function initialResearchPlan(contract) {
23
+ return {
24
+ schemaVersion: "1.0",
25
+ baseContractDigest: contract.contractDigest,
26
+ protocol: structuredClone(contract.protocol),
27
+ measurements: structuredClone(contract.measurements),
28
+ reason: "Initial plan; verify its interpretation against the source materials before execution.",
29
+ sources: [],
30
+ };
31
+ }
32
+
33
+ export async function initializeResearchPlan(contract, outputDirectory) {
34
+ if (!hasResearchWorkspace(contract)) return;
35
+ const filename = path.join(outputDirectory, PLAN_FILE);
36
+ // Recovery must retain the Agent's plan, including an unfinished edit.
37
+ if (!(await pathExists(filename))) await writeJson(filename, initialResearchPlan(contract));
38
+ }
39
+
40
+ /** Resolve an execution view without rewriting the signed input contract. */
41
+ export async function resolveResearchPlan(contract, outputDirectory) {
42
+ if (!hasResearchWorkspace(contract)) return contract;
43
+ const filename = path.join(outputDirectory, PLAN_FILE);
44
+ const stat = await lstat(filename);
45
+ if (!stat.isFile()) throw new CiteArkError("research-plan.json must be a regular file");
46
+ const plan = JSON.parse(await readFile(filename, "utf8"));
47
+ if (!plan || typeof plan !== "object" || Array.isArray(plan)) {
48
+ throw new CiteArkError("Research plan invalid: research-plan.json must be an object");
49
+ }
50
+ const issues = [];
51
+ if (plan.schemaVersion !== "1.0" || plan.baseContractDigest !== contract.contractDigest) issues.push("plan must reference the original contract");
52
+ if (!plan.protocol || typeof plan.protocol.objective !== "string" || !plan.protocol.objective.trim()) issues.push("plan.protocol.objective is required");
53
+ if (typeof plan.reason !== "string" || !plan.reason.trim() || !Array.isArray(plan.sources)) issues.push("plan reason and sources are required");
54
+ const measurements = Array.isArray(plan.measurements) ? plan.measurements : [];
55
+ if (!Array.isArray(plan.measurements)) issues.push("plan.measurements must be an array");
56
+ const goals = contract.workspacePlan.measurements;
57
+ const ids = measurements.map((item) => item?.measurementId);
58
+ if (new Set(ids).size !== ids.length) issues.push("plan measurement IDs must be unique");
59
+ for (const original of contract.measurements) {
60
+ if (!ids.includes(original.measurementId)) issues.push(`retain assigned measurement ${original.measurementId}; report missing evidence rather than deleting the goal`);
61
+ }
62
+ for (const measurement of measurements) {
63
+ if (!measurement || typeof measurement !== "object") {
64
+ issues.push("each plan measurement must be an object");
65
+ continue;
66
+ }
67
+ const goal = goals.find((item) => item.measurementId === measurement.measurementId);
68
+ if (!goal) issues.push(`measurement is outside the assigned claim: ${measurement.measurementId}`);
69
+ if (typeof measurement.metric !== "string" || typeof measurement.unit !== "string") issues.push("measurement metric and unit are required");
70
+ issues.push(...validateEvidenceParserDescriptor(measurement.parser, measurement)
71
+ .map((issue) => `measurements[${measurement.measurementId}].parser: ${issue}`));
72
+ }
73
+ const changed = sha256Value(plan.protocol) !== sha256Value(contract.protocol)
74
+ || sha256Value(measurements) !== sha256Value(contract.measurements);
75
+ if (changed && Array.isArray(plan.sources) && !plan.sources.some((source) =>
76
+ typeof source === "string" ? source.trim()
77
+ : source && typeof source === "object" && !Array.isArray(source)
78
+ && typeof source.sourceId === "string" && source.sourceId.trim()
79
+ && typeof source.locator === "string" && source.locator.trim())) {
80
+ issues.push('plan.sources needs a nonempty locator string or {"sourceId":"paper","locator":"Section / table / file path"} for a revised interpretation');
81
+ }
82
+ if (issues.length) throw new CiteArkError(`Research plan invalid:\n- ${issues.join("\n- ")}`);
83
+ // Resource limits belong to the runner, not the editable interpretation.
84
+ const protocol = { ...plan.protocol, executionMode: contract.protocol.executionMode };
85
+ if (contract.protocol.preflight && protocol.preflight) {
86
+ protocol.preflight = { ...protocol.preflight,
87
+ maxMinutes: contract.protocol.preflight.maxMinutes };
88
+ }
89
+ const uncoveredMeasurementIds = contract.research.claimMeasurementIds.filter((id) => !ids.includes(id));
90
+ return {
91
+ ...contract,
92
+ protocol,
93
+ measurements,
94
+ research: { ...contract.research, measurementIds: ids, uncoveredMeasurementIds,
95
+ claimCoverage: uncoveredMeasurementIds.length ? "partial" : "complete" },
96
+ executionPlan: {
97
+ digest: `sha256:${sha256Value(plan)}`,
98
+ changed,
99
+ plan,
100
+ originalProtocol: contract.protocol,
101
+ originalMeasurements: contract.measurements,
102
+ },
103
+ };
104
+ }
105
+
106
+ export async function planEvidencePaths(outputDirectory) {
107
+ const files = [PLAN_FILE];
108
+ if (await pathExists(path.join(outputDirectory, "cpu-preparation.json"))) files.push("cpu-preparation.json");
109
+ const phases = path.join(outputDirectory, "phase-history");
110
+ if (await pathExists(phases)) {
111
+ if (!(await lstat(phases)).isDirectory()) throw new CiteArkError("phase-history must be a directory");
112
+ for (const name of await readdir(phases)) {
113
+ if (!/^[a-f0-9]{64}\.json$/.test(name)) continue;
114
+ const receipt = path.join(phases, name);
115
+ if (!(await lstat(receipt)).isFile()) throw new CiteArkError("phase receipt must be a regular file");
116
+ files.push(`phase-history/${name}`);
117
+ }
118
+ }
119
+ const directory = path.join(outputDirectory, "plan-history");
120
+ if (await pathExists(directory)) {
121
+ if (!(await lstat(directory)).isDirectory()) throw new CiteArkError("plan-history must be a directory");
122
+ for (const name of await readdir(directory)) {
123
+ if (!/^[a-f0-9]{64}\.json$/.test(name)) continue;
124
+ const filename = path.join(directory, name);
125
+ if (!(await lstat(filename)).isFile()) throw new CiteArkError("plan snapshot must be a regular file");
126
+ const snapshot = JSON.parse(await readFile(filename, "utf8"));
127
+ if (`${sha256Value(snapshot)}.json` !== name) throw new CiteArkError("plan snapshot digest mismatch");
128
+ files.push(`plan-history/${name}`);
129
+ }
130
+ }
131
+ return files;
132
+ }
@@ -0,0 +1,70 @@
1
+ import { RESUME_HANDOFF_PROMPT } from "../deployment/handoff.mjs";
2
+ import { hasResearchWorkspace } from "./workspace-mode.mjs";
3
+ import { agentPlansExecution } from "../contracts/execution-mode.mjs";
4
+
5
+ export function initialReproductionPrompt(bundle) {
6
+ if (bundle.deploymentHandoff) return RESUME_HANDOFF_PROMPT;
7
+ if (hasResearchWorkspace(bundle.runtimeTask)) return [
8
+ `Run ID: ${bundle.runId}`,
9
+ "Read /job/input/agent-instructions.md and /job/output/research-plan.json. Complete the assigned claim using the persistent research workspace. The initial scientific interpretation is revisable, with source-grounded reasons and retained plan snapshots.",
10
+ "When workspacePlan.sourceInventory is present, read its source context, including qualitative findings, negative results, relations and ambiguities. Reopen the referenced original paper and fixed code where needed. The preparation plan is an initial interpretation: investigate inputs and define or revise comparisons and decision rules with source-grounded reasons before expensive execution. Preserve every original source claim and reported value; do not silently drop hard findings or turn a source statement into observed evidence. Context outside the assigned measurement scope does not authorize new experiments; retain relevant unresolved findings in progress.md and result limitations for assessment.",
11
+ "Task measurements are assigned to this run; workspacePlan.measurements lists the allowed scope; on graded tasks it is restricted to the assigned experiment. Finish assigned work first, then fill feasible gaps using existing evidence or justified work within the same budget. Use the current plan for execution and evidence bindings. Correct known source-resolvable defects before delivery, preserve valid raw outputs and negative results, and never reset the cumulative budget or silently replace the original goal.",
12
+ ].join("\n");
13
+ const agentPlanned = agentPlansExecution(bundle.runtimeTask?.protocol);
14
+ const executionStrategy = agentPlanned
15
+ ? [
16
+ "This contract uses protocol.executionMode=agent_planned. The scientific objective, repository identity, measurement dimensions, aggregation, parser bindings, and resource ceiling are immutable; the concrete shell command is not.",
17
+ "Treat protocol.entrypoint as a grounded starting point. Inspect the fixed implementation, run small probes, and choose or revise the concrete command, batching, helper scripts, and evidence adapters needed to satisfy protocol.objective efficiently. Record the decisive commands and every helper as execution evidence. Do not change the scientific scope to make the run easier.",
18
+ ]
19
+ : [
20
+ "This legacy contract uses fixed_entrypoint semantics. Treat protocol.entrypoint as immutable. If it is a placeholder or embeds a reported result, fail closed as protocol_ambiguous instead of inventing a replacement scientific command.",
21
+ "Capture the complete stdout and stderr of the fixed entrypoint on its first successful full run. Group compound commands before applying tee or redirection, and do not rerun an expensive successful entrypoint merely to repair a logging wrapper.",
22
+ ];
23
+ return [
24
+ `Run ID: ${bundle.runId}`,
25
+ "Read /job/input/task.json and complete the scientific reproduction autonomously.",
26
+ "If /job/input/compute-decision.json exists, treat its environment as the actual allocated compute and task.json as the immutable scientific contract.",
27
+ "Inspect /job/input/dataset-source-registry.json for platform-observed candidate transports, but verify every selected archive at runtime and preserve an acquisition manifest; registry confidence is not a scientific verdict.",
28
+ "Inspect /job/assets, /job/runtime-cache, and /job/runtime-venv for optional bytes prepared by an earlier low-cost cache phase. Reuse them only when useful; missing, partial, or incompatible cache content is not a blocker. You retain full authority to ignore it, download from the original declared URLs, choose another paper-grounded transport, and replace task-local dependencies.",
29
+ "In legacy contracts, asset.acquisition.requiredPaths is a selective-download hint and may list alternative formats together. Do not require every pattern to exist. Determine which concrete files the real implementation needs, record any unmatched hints, and verify the selected files during preflight.",
30
+ "Treat transient public-download failures as resumable execution work. Create the acquisition manifest before a large archive or checkpoint download, write to an atomic temporary file under a checkpointed root, use redirects plus bounded all-error retries and resume support (for curl: --fail --location --retry 8 --retry-all-errors --retry-delay 2 --continue-at -), and verify size and digest before promotion. One 5xx or timeout must not end the task; after repeated failures, try a materially distinct paper-grounded public transport and record both attempts without claiming unverified equivalence. For a failing public Hugging Face resolve URL, the identical owner/repository/revision/path under https://hf-mirror.com is a candidate transport only and must pass content-identity and checkpoint-format verification against the declared asset.",
31
+ "The repository is ready at /job/workspace/repository.",
32
+ "Before execution, make a checklist from every measurement's dimensions, aggregation, repetition or seed schedule, hardware scope, and parser evidence path; preserve that exact scientific scope.",
33
+ "For every deterministic evidence adapter, write down the declared metric equation and unit before coding, then trace each operand to raw outputs from the required scientific commands. Never hardcode observed arrays or checkpoint/log values into adapter source, and never substitute a conveniently available statistic such as a minimum for a declared ratio, mean, percentage, or cross-command comparison.",
34
+ "A smoke test, synthetic carrier, one-sample round trip, structural invariant, or configuration calculation cannot be promoted into a paper-reported full-dataset metric. If the contract binds a reduced run to a broader dataset, sample-count, split, repetition, aggregation, or evaluator scope, fail closed as protocol_ambiguous instead of emitting a convenient scalar.",
35
+ ...executionStrategy,
36
+ "Do not infer protocol equivalence from a similar model name or architecture. Audit the exact checkpoint/model, preprocessing and feature construction, dataset version and split, label mapping, training objective, optimizer, evaluation implementation, repetitions, aggregation, and hardware-sensitive metrics against the immutable contract.",
37
+ "For checkpoint evaluation, training history is provenance rather than required computation: run fresh full evaluation of the fixed model with its controls. Missing weights do not authorize retraining a replacement. Do not convert an explicitly assigned training-process question into final-weight evaluation; report a contract mismatch for replanning.",
38
+ "Named model variants are protocol inputs, not presentation labels. Trace each dimensions.model_variant to the exact function, branch, checkpoint, and command it selects; an unused selector, two names mapped to the same code path, or duplicated raw output under different filenames must fail preflight instead of being reported as two measurements.",
39
+ "Treat task.json.protocol.preflight as a hard phase boundary and bounded repair phase: write its declared evidence file, create an isolated locked environment when needed, repair invalid operational probes, and pass every runtime, dependency, required-asset, entrypoint, parser, and smoke-path check before starting full training or evaluation. Required public_unverified assets must be acquired from their sourceUrls or exact registryDatasetId and identity-verified; an initially empty /job/assets directory is work to perform, not proof that the asset is unavailable. Let resumable acquisition and a still-progressing locked dependency install use the declared preflight budget instead of imposing a smaller ad hoc timeout.",
40
+ "For every required dataset, complete the dataset_loader preflight with the real evaluator loader: resolve the declared configuration and split, read representative rows, and verify schema and identity before downloading a large checkpoint or starting full evaluation. A package import, HTTP reachability probe, or dataset-card request is not enough.",
41
+ "For a hardware-sensitive measurement, protocol.benchmark is immutable scientific scope. Verify batch size, dtype, warmup count, timed iteration count, timer, device synchronization, input construction, and every exact implementation callable before timing. Do not guess a missing comparator or benchmark condition during execution; report protocol_ambiguous so the Research Plan can be repaired.",
42
+ "For dependency repair, inspect the preinstalled baseline and /opt/citeark/wheels before downloading or building packages. The baseline saves downloads; it is not a version policy. Inside /job/runtime-venv you may upgrade, downgrade, or replace any Python package when repository metadata, package metadata, a known compatibility boundary, or a direct preflight gives a concrete reason. Record the change and test the resulting stack. The fixed Rust/Cargo toolchain is already present, so do not invoke rustup.",
43
+ "No global PIP_CONSTRAINT or UV_CONSTRAINT should restrict task packages. If a repair changes Torch/CUDA or another major numerical stack, verify imports, device availability, ABI-sensitive operators, and the repository's real entrypoint before scientific execution; this is a validation obligation, not a prohibition.",
44
+ "Before corpus-scale execution, verify that the chosen portable enumerator resolves exactly the contract-declared sample count and unique identities. POSIX `sh` does not implement recursive `**` globbing; repair an agent-planned command that depends on it, and fail closed only when a fixed_entrypoint contract cannot preserve scope.",
45
+ "Run a representative duration probe after assets and runtime are ready. Preserve the timing, allocated device, workload extrapolation, contract timeout, and evidence-writing reserve. For agent_planned execution, improve orchestration or batching while preserving scope when the first strategy does not fit; for fixed_entrypoint execution, fail preflight rather than changing the command. A checkpoint-load or one-item smoke test is not throughput evidence.",
46
+ "If the duration probe predicts that a decisive command will outlive the shell tool's default per-command timeout, set an explicit tool timeout that covers the extrapolated run within the remaining ledger, or launch one checkpoint-safe background process with its PID and complete log under /job/workspace or /job/output and poll it. Before retrying, inspect the prior PID, process state, log, and output; never launch a duplicate while the earlier command may still be alive. A shell tool timeout is not a scientific timeout and must not kill an otherwise viable full evaluation.",
47
+ "Repository identity is fixed in task.json.repository and runner-owned audit; /job/input/repository-identity.json is intentionally unavailable during execution and must never be required. An agent-planned helper may call fixed official functions or independently implement a declared CiteArk reconstruction, but it must preserve provenance and scientific scope.",
48
+ "Use the available uv and managed Python 3.12 baseline for repositories that declare them. Stop with an honest failed or partial result only after materially distinct bounded setup repairs are exhausted or a genuine private/unverifiable asset blocks progress; a preflight result is never the scientific measurement.",
49
+ "If the observed result is surprising, audit implementation and protocol comparability once using the existing evidence before interpreting it. Never change the claim, reported target, tolerance, split, metric, or conclusion to make the run look successful; record the concrete mismatch or failure when the contract cannot be satisfied.",
50
+ "Write all deliverables under /job/output and finish by writing /job/output/result.json.",
51
+ "As soon as the decisive evidence and a schema-valid result.json are stable, stop the task instead of continuing optional exploration.",
52
+ ].join("\n");
53
+ }
54
+
55
+ export function retryReproductionPrompt(validationIssues, task) {
56
+ if (hasResearchWorkspace(task)) return [
57
+ "The submitted research plan or result needs correction:",
58
+ ...validationIssues.map((issue) => `- ${issue}`),
59
+ "Continue from research-plan.json, progress.md and the existing evidence. Repair the listed defects within the remaining cumulative budget. You may correct the scientific interpretation with sources; preserve the original goals and prior results. Reuse successful computation, record the final plan using run-research-plan.mjs, then write a valid result.json. Do not seek a favorable number or restart completed experiments merely to repair an envelope.",
60
+ ].join("\n");
61
+ return [
62
+ "CiteArk validated the submitted result and found protocol errors:",
63
+ ...validationIssues.map((issue) => `- ${issue}`),
64
+ "Inspect the existing result and evidence first. Reuse already-successful experiment outputs and parser inputs; repair only the listed envelope defects with the minimum necessary change.",
65
+ "Do not rerun an expensive experiment, overwrite valid parser evidence, or reinterpret a completed scientific run unless a listed validation issue proves that evidence is unusable.",
66
+ "After rewriting /job/output/result.json, verify every referenced path and stop immediately once the listed defects are resolved.",
67
+ "Do not change scientific conclusions merely to satisfy validation.",
68
+ "Do not replace a failed or mismatching condition with an easier model, split, preprocessing path, seed count, or metric. If the immutable contract itself is infeasible, preserve the evidence and report the blocker instead of silently changing the question.",
69
+ ].join("\n");
70
+ }
@@ -0,0 +1,188 @@
1
+ import { saveRunRecovery } from '../compute/coordinator-recovery.mjs';
2
+ import { stageExecutionHandoff } from "../deployment/handoff.mjs";
3
+ import { mkdir, writeFile } from "node:fs/promises";
4
+ import path from "node:path";
5
+
6
+ import { updateRunState } from "../job.mjs";
7
+ import { previewEgressProxy, startEgressProxy } from "../network/egress-proxy.mjs";
8
+ import {
9
+ previewProviderSession,
10
+ startProviderSession,
11
+ } from "../provider/relay.mjs";
12
+ import { buildAgentInvocation } from "../runtime/index.mjs";
13
+ import {
14
+ assertLocalRunStopped, buildDockerRun,
15
+ inspectDockerImage,
16
+ requireRuntimeCredentials,
17
+ runDockerInvocation,
18
+ writeInvocation,
19
+ } from "../sandbox/docker.mjs";
20
+ import {
21
+ prepareReproduction,
22
+ } from "./lifecycle.mjs";
23
+ import { agentWorkloadMetadata } from "../workloads/definition.mjs";
24
+ import { createAutonomousReproductionWorkload } from "../workloads/reproduction.mjs";
25
+
26
+ export async function executeReproduction({
27
+ taskPath,
28
+ referenceMaterialsPath,
29
+ handoff,
30
+ localEnvironment,
31
+ localPreflight,
32
+ runsDirectory,
33
+ imageOverride,
34
+ repositoryOverride,
35
+ paperOverride,
36
+ computeDecision,
37
+ runId,
38
+ agentRuntime,
39
+ dryRun = false,
40
+ progress = () => {},
41
+ }) {
42
+ const bundle = await prepareReproduction({
43
+ taskPath,
44
+ referenceMaterialsPath,
45
+ runsDirectory,
46
+ runId,
47
+ imageOverride,
48
+ repositoryOverride,
49
+ paperOverride,
50
+ computeDecision,
51
+ agentRuntime,
52
+ requireApi: !dryRun,
53
+ sandboxProvider: "docker",
54
+ });
55
+ if (handoff) await stageExecutionHandoff(bundle, handoff);
56
+ if (handoff && localEnvironment) {
57
+ bundle.executionEnvironment = structuredClone(localEnvironment);
58
+ bundle.runtimeTask.environment = structuredClone(localEnvironment);
59
+ const { writeJson } = await import('../util.mjs');
60
+ await writeJson(path.join(bundle.directories.input, 'task.json'), bundle.runtimeTask);
61
+ await writeJson(path.join(bundle.directories.input, 'local-environment.json'), { environment: localEnvironment, preflight: localPreflight });
62
+ await updateRunState(bundle, { executionEnvironment: localEnvironment, localPreflight });
63
+ }
64
+ const workload = createAutonomousReproductionWorkload();
65
+ await updateRunState(bundle, agentWorkloadMetadata(workload));
66
+ const runtimeAgent = bundle.runtimeAgent;
67
+ const attempts = bundle.state.attempts ?? [];
68
+ const firstAttempt = bundle.restored && bundle.state.status !== "prepared" ? Math.max(1, ...attempts.map(a => a.attempt)) + 1 : 1;
69
+ await saveRunRecovery(bundle, async () => {});
70
+ let validation = { result: null, issues: ["尚未运行 Agent"], protocolVerification: null };
71
+ if (!dryRun) { await assertLocalRunStopped(bundle); requireRuntimeCredentials(runtimeAgent); }
72
+ const providerSession = dryRun
73
+ ? previewProviderSession(runtimeAgent)
74
+ : await startProviderSession(runtimeAgent, {
75
+ correlation: { runId: bundle.runId, workload: "reproduction" },
76
+ });
77
+ let egressSession = previewEgressProxy();
78
+
79
+ try {
80
+ if (!dryRun) egressSession = await startEgressProxy();
81
+ const instructions = await workload.instructions(bundle);
82
+ if (bundle.restored && !dryRun && bundle.state.status === 'completed') {
83
+ const retained = await workload.validate({ bundle });
84
+ if (retained.issues.length === 0) return await workload.finalize({ bundle, validation: retained, attempts, timedOut: false, sandboxImageIdentity: bundle.sandboxImageIdentity });
85
+ throw Error('Saved completed results failed validation. The experiment was not restarted.');
86
+ }
87
+ const firstRuntimeInvocation = await buildAgentInvocation({
88
+ bundle,
89
+ attempt: firstAttempt,
90
+ prompt: workload.prompt({ bundle, attempt: firstAttempt, validation }),
91
+ instructions,
92
+ providerSession,
93
+ });
94
+ const firstDockerInvocation = {
95
+ ...buildDockerRun({
96
+ bundle,
97
+ runtimeCommand: firstRuntimeInvocation.command,
98
+ runtimeArgs: firstRuntimeInvocation.args,
99
+ attempt: firstAttempt,
100
+ providerSession,
101
+ egressSession,
102
+ }),
103
+ attempt: firstAttempt,
104
+ };
105
+ const recordedInvocations = [firstDockerInvocation];
106
+ await writeInvocation(bundle, recordedInvocations);
107
+ await mkdir(path.join(bundle.directories.execution, "prompts"), { recursive: true });
108
+ await writeFile(path.join(bundle.directories.execution, "prompts", `attempt-${firstAttempt}.txt`), `${firstRuntimeInvocation.prompt}\n`);
109
+
110
+ if (dryRun) {
111
+ await updateRunState(bundle, { status: "prepared" });
112
+ return { dryRun: true, bundle, invocation: firstDockerInvocation };
113
+ }
114
+
115
+ await updateRunState(bundle, { status: "running", startedAt: new Date().toISOString() });
116
+ const maxAttempts = 1 + runtimeAgent.validationRetries;
117
+ let timedOut = false;
118
+ for (let attempt = firstAttempt; attempt < firstAttempt + maxAttempts; attempt += 1) {
119
+ const runtimeInvocation = attempt === firstAttempt
120
+ ? firstRuntimeInvocation
121
+ : await buildAgentInvocation({
122
+ bundle,
123
+ attempt,
124
+ prompt: workload.prompt({ bundle, attempt, validation }),
125
+ instructions,
126
+ providerSession,
127
+ });
128
+ const dockerInvocation = attempt === firstAttempt
129
+ ? firstDockerInvocation
130
+ : {
131
+ ...buildDockerRun({
132
+ bundle,
133
+ runtimeCommand: runtimeInvocation.command,
134
+ runtimeArgs: runtimeInvocation.args,
135
+ attempt,
136
+ providerSession,
137
+ egressSession,
138
+ }),
139
+ attempt,
140
+ };
141
+ if (attempt > firstAttempt) {
142
+ recordedInvocations.push(dockerInvocation);
143
+ await writeInvocation(bundle, recordedInvocations);
144
+ await writeFile(path.join(bundle.directories.execution, "prompts", `attempt-${attempt}.txt`), `${runtimeInvocation.prompt}\n`);
145
+ }
146
+ progress(`CiteArk reproduction ${bundle.runId}: attempt ${attempt - firstAttempt + 1}/${maxAttempts}`);
147
+ const startedAt = new Date().toISOString();
148
+ const processResult = await runDockerInvocation(dockerInvocation, {
149
+ bundle,
150
+ attempt,
151
+ timeoutMs: bundle.executionEnvironment.timeoutMinutes * 60_000,
152
+ });
153
+ timedOut ||= processResult.timedOut;
154
+ attempts.push({
155
+ attempt,
156
+ startedAt,
157
+ finishedAt: new Date().toISOString(),
158
+ exitCode: processResult.code,
159
+ signal: processResult.signal,
160
+ timedOut: processResult.timedOut,
161
+ });
162
+ validation = await workload.validate({
163
+ bundle,
164
+ completionPath: path.join(bundle.directories.output, "result.json"),
165
+ });
166
+ if (validation.issues.length === 0) break;
167
+ if (attempt < firstAttempt + maxAttempts - 1) progress(`结果协议校验失败,继续同一 session:${validation.issues.join(";")}`);
168
+ }
169
+
170
+ validation = await workload.recover({
171
+ bundle,
172
+ validation,
173
+ attempts,
174
+ timedOut,
175
+ });
176
+ await writeInvocation(bundle, recordedInvocations);
177
+ return await workload.finalize({
178
+ bundle,
179
+ validation,
180
+ attempts,
181
+ timedOut,
182
+ sandboxImageIdentity: await inspectDockerImage(bundle.executionEnvironment.image),
183
+ });
184
+ } finally {
185
+ await saveRunRecovery(bundle, async () => {});
186
+ await Promise.all([providerSession.close(), egressSession.close()]);
187
+ }
188
+ }