@citeark/agent 0.3.20

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (347) hide show
  1. package/LICENSE +202 -0
  2. package/README.md +128 -0
  3. package/data/dataset-source-registry.v1.json +300 -0
  4. package/dist/arkgraph/boot.js +6 -0
  5. package/dist/arkgraph/index.html +1 -0
  6. package/dist/arkgraph/viewer.css +1 -0
  7. package/dist/arkgraph/viewer.en.css +1 -0
  8. package/dist/arkgraph/viewer.en.js +49 -0
  9. package/dist/arkgraph/viewer.en.js.LEGAL.txt +56 -0
  10. package/dist/arkgraph/viewer.js +49 -0
  11. package/dist/arkgraph/viewer.js.LEGAL.txt +56 -0
  12. package/docker/claude-code/Dockerfile +97 -0
  13. package/docker/claude-code/codex-pro-relay.mjs +466 -0
  14. package/docker/claude-code/runtime-contract-check.mjs +79 -0
  15. package/docs/arkgraph-reading.md +79 -0
  16. package/docs/configuration.md +100 -0
  17. package/docs/integration.md +92 -0
  18. package/docs/maturity-plan.md +27 -0
  19. package/docs/npm-release.md +44 -0
  20. package/docs/paper-reading.md +40 -0
  21. package/docs/research-plan-granularity.md +27 -0
  22. package/docs/terminal.md +49 -0
  23. package/examples/toy-evaluation/compile-task.json +27 -0
  24. package/examples/toy-evaluation/paper.md +5 -0
  25. package/examples/toy-evaluation/repository/README.md +9 -0
  26. package/examples/toy-evaluation/repository/checkpoint.json +4 -0
  27. package/examples/toy-evaluation/repository/evaluate.py +17 -0
  28. package/examples/toy-evaluation/task.json +81 -0
  29. package/package.json +59 -0
  30. package/prompts/compile-research.md +58 -0
  31. package/prompts/execute-contract.md +72 -0
  32. package/prompts/execute-workspace-simple.md +51 -0
  33. package/prompts/execute-workspace.md +34 -0
  34. package/prompts/prepare-reproduction.md +82 -0
  35. package/prompts/repair-research.md +45 -0
  36. package/protocol/CAP.md +129 -0
  37. package/protocol/LICENSE +12 -0
  38. package/protocol/MAPPINGS.md +72 -0
  39. package/protocol/README.md +38 -0
  40. package/protocol/conformance-v2.0-alpha.1.json +36 -0
  41. package/protocol/examples/arkgraph/checkpoint-evaluation.json +309 -0
  42. package/protocol/examples/arkgraph/fixtures.mjs +49 -0
  43. package/protocol/examples/arkgraph/paper-free.json +291 -0
  44. package/protocol/examples/arkgraph/partial-failure.json +344 -0
  45. package/protocol/examples/arkgraph/training-evaluation.json +443 -0
  46. package/protocol/profiles/agent-trace.md +16 -0
  47. package/protocol/profiles/computational-run.md +16 -0
  48. package/protocol/profiles/core.md +15 -0
  49. package/protocol/profiles/public-bundle.md +18 -0
  50. package/protocol/profiles/reproduction.md +29 -0
  51. package/protocol/profiles/research-compilation.md +44 -0
  52. package/protocol/profiles/research-plan.md +39 -0
  53. package/protocol/profiles/restricted-evidence.md +15 -0
  54. package/runtime/bootstrap-autodl-runtime.sh +314 -0
  55. package/runtime/create-runtime-venv.sh +41 -0
  56. package/runtime/install-local-cpu-runtime.sh +23 -0
  57. package/runtime/install-scientific-runtime.sh +153 -0
  58. package/runtime/mineru/parse.py +62 -0
  59. package/runtime/mineru/requirements.txt +4 -0
  60. package/runtime/requirements-baseline.txt +38 -0
  61. package/schemas/cap/v2/activity.schema.json +47 -0
  62. package/schemas/cap/v2/agent.schema.json +32 -0
  63. package/schemas/cap/v2/assertion.schema.json +110 -0
  64. package/schemas/cap/v2/descriptor.schema.json +243 -0
  65. package/schemas/cap/v2/entity.schema.json +64 -0
  66. package/schemas/cap/v2/manifest.schema.json +67 -0
  67. package/schemas/cap/v2/relation.schema.json +82 -0
  68. package/schemas/compute-catalog.schema.json +63 -0
  69. package/schemas/compute-decision.schema.json +27 -0
  70. package/schemas/execution-contract.schema.json +1024 -0
  71. package/schemas/research-card.schema.json +30 -0
  72. package/schemas/research-inventory-draft.schema.json +366 -0
  73. package/schemas/research.schema.json +1044 -0
  74. package/schemas/result.schema.json +173 -0
  75. package/schemas/verification-policy.schema.json +47 -0
  76. package/schemas/verified-conclusion.schema.json +58 -0
  77. package/schemas/workspace-summary.schema.json +24 -0
  78. package/scripts/build-arkgraph-view.mjs +12 -0
  79. package/scripts/check-execution-feasibility.mjs +24 -0
  80. package/scripts/check-syntax.mjs +15 -0
  81. package/scripts/deterministic-asset-preparation.py +438 -0
  82. package/scripts/package-cap.mjs +23 -0
  83. package/scripts/package-local-agent.mjs +23 -0
  84. package/scripts/preview-arkgraph.mjs +25 -0
  85. package/scripts/replay-research-compiler-candidate.mjs +134 -0
  86. package/scripts/review-compiler-sources.mjs +44 -0
  87. package/scripts/run-asset-preparation.sh +17 -0
  88. package/scripts/run-research-plan.mjs +98 -0
  89. package/scripts/validate-asset-preparation.py +290 -0
  90. package/scripts/verify-local-runtime.mjs +57 -0
  91. package/scripts/verify-npm-package.mjs +57 -0
  92. package/src/adapters/paper2agent.mjs +107 -0
  93. package/src/assets/cache.mjs +159 -0
  94. package/src/assets/compute.mjs +98 -0
  95. package/src/assets/executor.mjs +145 -0
  96. package/src/assets/lifecycle.mjs +213 -0
  97. package/src/assets/manifest.mjs +242 -0
  98. package/src/assets/opportunistic-preparation.mjs +81 -0
  99. package/src/assets/plan.mjs +411 -0
  100. package/src/assets/prompts.mjs +29 -0
  101. package/src/assets/public-asset-probe.mjs +525 -0
  102. package/src/assets/qualification.mjs +119 -0
  103. package/src/assets/readiness.mjs +130 -0
  104. package/src/assets/reproduction-admission.mjs +355 -0
  105. package/src/assets/requirements.mjs +152 -0
  106. package/src/assets/source-grounding.mjs +341 -0
  107. package/src/assets/source-policy.mjs +118 -0
  108. package/src/autodl/client.mjs +260 -0
  109. package/src/autodl/ssh.mjs +380 -0
  110. package/src/autodl/tools.mjs +129 -0
  111. package/src/cap/redaction.mjs +38 -0
  112. package/src/cap/v2/archive.mjs +152 -0
  113. package/src/cap/v2/attestation.mjs +204 -0
  114. package/src/cap/v2/canonical-json.mjs +114 -0
  115. package/src/cap/v2/compilation-artifact.mjs +240 -0
  116. package/src/cap/v2/core.mjs +282 -0
  117. package/src/cap/v2/measurement-assessment-records.mjs +23 -0
  118. package/src/cap/v2/pipeline-artifact.mjs +922 -0
  119. package/src/cap/v2/read.mjs +41 -0
  120. package/src/cap/v2/reassessment-artifact.mjs +383 -0
  121. package/src/cap/v2/research-artifact.mjs +231 -0
  122. package/src/cap/v2/research-map-records.mjs +46 -0
  123. package/src/cap/v2/research-object-records.mjs +163 -0
  124. package/src/cap/v2/research-records.mjs +187 -0
  125. package/src/cap/v2/verify.mjs +642 -0
  126. package/src/cli.mjs +1146 -0
  127. package/src/compute/autodl-pro-compiler.mjs +347 -0
  128. package/src/compute/autodl-pro-executor.mjs +459 -0
  129. package/src/compute/autodl-pro-job.mjs +843 -0
  130. package/src/compute/autodl-pro-network.mjs +295 -0
  131. package/src/compute/autodl-pro-remote.mjs +810 -0
  132. package/src/compute/autodl-pro-staging.mjs +117 -0
  133. package/src/compute/campaign.mjs +110 -0
  134. package/src/compute/catalog.mjs +123 -0
  135. package/src/compute/checkpoint-protocol.mjs +154 -0
  136. package/src/compute/codex-account-lock.mjs +111 -0
  137. package/src/compute/codex-account-session.mjs +107 -0
  138. package/src/compute/compiler-profile.mjs +38 -0
  139. package/src/compute/compiler-router.mjs +23 -0
  140. package/src/compute/coordinator-recovery.mjs +210 -0
  141. package/src/compute/executor-router.mjs +29 -0
  142. package/src/compute/gcp-batch-compiler.mjs +685 -0
  143. package/src/compute/gcp-batch-executor.mjs +1215 -0
  144. package/src/compute/gcp-batch-failure.mjs +92 -0
  145. package/src/compute/gcp-batch-job.mjs +527 -0
  146. package/src/compute/gcp-batch-lifecycle.mjs +81 -0
  147. package/src/compute/gcp-checkpoint-worker.mjs +1633 -0
  148. package/src/compute/local-codex-compiler.mjs +52 -0
  149. package/src/compute/measurement-hardware.mjs +128 -0
  150. package/src/compute/remote-attempt.mjs +226 -0
  151. package/src/compute/requirements.mjs +124 -0
  152. package/src/compute/research-phases.mjs +48 -0
  153. package/src/compute/scheduler.mjs +452 -0
  154. package/src/compute/shared-workloads.mjs +26 -0
  155. package/src/compute/stage-archive.mjs +79 -0
  156. package/src/contracts/campaign-contract.mjs +52 -0
  157. package/src/contracts/execution-contract.mjs +819 -0
  158. package/src/contracts/execution-mode.mjs +19 -0
  159. package/src/contracts/execution-timeouts.mjs +45 -0
  160. package/src/contracts/execution-workload.mjs +68 -0
  161. package/src/contracts/preflight-schema.mjs +25 -0
  162. package/src/contracts/public-contract.mjs +63 -0
  163. package/src/contracts/subject-tags.mjs +31 -0
  164. package/src/dashboard/data.mjs +898 -0
  165. package/src/dashboard/server.mjs +79 -0
  166. package/src/dashboard/static/dashboard.css +366 -0
  167. package/src/dashboard/static/dashboard.js +560 -0
  168. package/src/dashboard/static/index.html +85 -0
  169. package/src/deployment/community-policy.mjs +9 -0
  170. package/src/deployment/environment.mjs +112 -0
  171. package/src/deployment/guided.mjs +98 -0
  172. package/src/deployment/handoff.mjs +102 -0
  173. package/src/deployment/local-contract.mjs +31 -0
  174. package/src/deployment/local.mjs +100 -0
  175. package/src/deployment/prepare.mjs +46 -0
  176. package/src/deployment/recipe.mjs +108 -0
  177. package/src/deployment/supplement.mjs +51 -0
  178. package/src/deployment/terminal.mjs +43 -0
  179. package/src/diagnosis/renderer.mjs +75 -0
  180. package/src/diagnosis/target-failure.mjs +46 -0
  181. package/src/evidence/parser-registry.mjs +54 -0
  182. package/src/evidence/parsers/fasttext-classification.mjs +82 -0
  183. package/src/evidence/parsers/json-scalar.mjs +96 -0
  184. package/src/evidence/parsers/simcse-senteval.mjs +104 -0
  185. package/src/evidence/parsers/starspace-classification.mjs +78 -0
  186. package/src/evidence/registry.mjs +147 -0
  187. package/src/execution/runner-audit.mjs +473 -0
  188. package/src/gcp/auth.mjs +106 -0
  189. package/src/gcp/batch-client.mjs +120 -0
  190. package/src/gcp/resource-discovery.mjs +177 -0
  191. package/src/gcp/rest.mjs +82 -0
  192. package/src/gcp/secret-manager.mjs +34 -0
  193. package/src/gcp/signed-url.mjs +133 -0
  194. package/src/gcp/storage.mjs +220 -0
  195. package/src/graph/command.mjs +41 -0
  196. package/src/graph/execution.mjs +97 -0
  197. package/src/graph/model.mjs +37 -0
  198. package/src/graph/presentation.mjs +110 -0
  199. package/src/graph/query.mjs +159 -0
  200. package/src/graph/research-relations.mjs +69 -0
  201. package/src/graph/source-page.mjs +12 -0
  202. package/src/graph/source-preview.mjs +34 -0
  203. package/src/graph/validate.mjs +76 -0
  204. package/src/job.mjs +496 -0
  205. package/src/network/autodl-routing-proxy.mjs +462 -0
  206. package/src/network/egress-proxy.mjs +158 -0
  207. package/src/observability/event-contract.mjs +230 -0
  208. package/src/observability/pipeline-monitor.mjs +166 -0
  209. package/src/pipeline/orchestrator.mjs +1281 -0
  210. package/src/pipeline/recovery-error.mjs +11 -0
  211. package/src/pipeline/replay.mjs +304 -0
  212. package/src/pipeline/shared-execution.mjs +115 -0
  213. package/src/pipeline/stage-checkpoint.mjs +86 -0
  214. package/src/pipeline/stage-recovery.mjs +101 -0
  215. package/src/pipeline/targets.mjs +110 -0
  216. package/src/process.mjs +143 -0
  217. package/src/protocol.mjs +312 -0
  218. package/src/provider/codex-account.mjs +44 -0
  219. package/src/provider/codex-completion.mjs +49 -0
  220. package/src/provider/completion.mjs +292 -0
  221. package/src/provider/model-client.mjs +44 -0
  222. package/src/provider/model-route.mjs +29 -0
  223. package/src/provider/openrouter-readiness.mjs +189 -0
  224. package/src/provider/reader-bridge.mjs +35 -0
  225. package/src/provider/relay.mjs +263 -0
  226. package/src/provider/runtime-auth.mjs +40 -0
  227. package/src/public/cap.d.mts +90 -0
  228. package/src/public/cap.mjs +12 -0
  229. package/src/public/contracts.d.mts +2 -0
  230. package/src/public/host.mjs +171 -0
  231. package/src/public/operations.d.mts +11 -0
  232. package/src/public/presentation.d.mts +4 -0
  233. package/src/records/views.mjs +26 -0
  234. package/src/remote/command.mjs +178 -0
  235. package/src/remote/ssh.mjs +59 -0
  236. package/src/repository-origin.mjs +81 -0
  237. package/src/reproduction/evidence-feedback.mjs +96 -0
  238. package/src/reproduction/incomplete-initialization.mjs +25 -0
  239. package/src/reproduction/lifecycle.mjs +253 -0
  240. package/src/reproduction/plan.mjs +132 -0
  241. package/src/reproduction/prompts.mjs +70 -0
  242. package/src/reproduction/runner.mjs +188 -0
  243. package/src/reproduction/summary.mjs +130 -0
  244. package/src/reproduction/workspace-mode.mjs +7 -0
  245. package/src/research/automatic-admission.mjs +156 -0
  246. package/src/research/compiler-coverage.mjs +85 -0
  247. package/src/research/compiler-failure.mjs +24 -0
  248. package/src/research/compiler-normalization-guards.mjs +112 -0
  249. package/src/research/compiler-repair.mjs +3 -0
  250. package/src/research/compiler.mjs +853 -0
  251. package/src/research/continuation-selection.mjs +26 -0
  252. package/src/research/execution-graph-context.mjs +43 -0
  253. package/src/research/experiment-importance.mjs +15 -0
  254. package/src/research/inventory-handoff.mjs +104 -0
  255. package/src/research/inventory-revisions.mjs +32 -0
  256. package/src/research/mineru-local.mjs +73 -0
  257. package/src/research/paper-command.mjs +19 -0
  258. package/src/research/paper-markdown.mjs +180 -0
  259. package/src/research/paper-source-map.mjs +69 -0
  260. package/src/research/planning-policy.mjs +88 -0
  261. package/src/research/reference-materials.mjs +11 -0
  262. package/src/research/reproduction-scope.mjs +30 -0
  263. package/src/research/research-map.mjs +94 -0
  264. package/src/research/research-objects.mjs +88 -0
  265. package/src/research/source-discovery.mjs +646 -0
  266. package/src/research/source-observations.mjs +75 -0
  267. package/src/research/source-review-cli-mcp.mjs +26 -0
  268. package/src/research/source-review-input.mjs +209 -0
  269. package/src/research/source-review-local-codex.mjs +36 -0
  270. package/src/research/source-review-model.mjs +70 -0
  271. package/src/research/source-review.mjs +173 -0
  272. package/src/research/structure.mjs +3163 -0
  273. package/src/research-card/renderer.mjs +277 -0
  274. package/src/research-card/verified-conclusion.mjs +143 -0
  275. package/src/results/output-registry.mjs +183 -0
  276. package/src/runtime/claude-code.mjs +52 -0
  277. package/src/runtime/codex-capacity-retry.mjs +87 -0
  278. package/src/runtime/codex.mjs +64 -0
  279. package/src/runtime/config.mjs +157 -0
  280. package/src/runtime/final-output.mjs +40 -0
  281. package/src/runtime/index.mjs +21 -0
  282. package/src/runtime/local-codex.mjs +74 -0
  283. package/src/runtime/opencode.mjs +95 -0
  284. package/src/runtime/prompt.mjs +13 -0
  285. package/src/sandbox/docker.mjs +363 -0
  286. package/src/settings/command.mjs +297 -0
  287. package/src/settings/store.mjs +119 -0
  288. package/src/telemetry/pricing.mjs +68 -0
  289. package/src/telemetry/usage.mjs +265 -0
  290. package/src/terminal/events.mjs +97 -0
  291. package/src/terminal/input.mjs +40 -0
  292. package/src/terminal/plain.mjs +40 -0
  293. package/src/terminal/remote-stream.mjs +22 -0
  294. package/src/terminal/screen.mjs +214 -0
  295. package/src/terminal/transcript.mjs +69 -0
  296. package/src/util.mjs +107 -0
  297. package/src/verification/ai-assessor.mjs +534 -0
  298. package/src/verification/claim-evaluator.mjs +242 -0
  299. package/src/verification/evidence-context.mjs +165 -0
  300. package/src/verification/evidence-reader.mjs +95 -0
  301. package/src/verification/integrity.mjs +570 -0
  302. package/src/verification/tolerance.mjs +32 -0
  303. package/src/workloads/cpu-research-preparation.mjs +56 -0
  304. package/src/workloads/definition.mjs +74 -0
  305. package/src/workloads/phase-aware-reproduction.mjs +46 -0
  306. package/src/workloads/reproduction.mjs +85 -0
  307. package/src/workspace/command.mjs +242 -0
  308. package/src/workspace/control.mjs +49 -0
  309. package/src/workspace/entry.mjs +28 -0
  310. package/src/workspace/input.mjs +93 -0
  311. package/src/workspace/interactive.mjs +94 -0
  312. package/src/workspace/jobs.mjs +418 -0
  313. package/src/workspace/session.mjs +97 -0
  314. package/src/workspace/worker.mjs +137 -0
  315. package/ui/arkgraph/ambient-motion.mjs +10 -0
  316. package/ui/arkgraph/app.jsx +153 -0
  317. package/ui/arkgraph/boot.js +6 -0
  318. package/ui/arkgraph/camera-motion.mjs +20 -0
  319. package/ui/arkgraph/context-reveal.mjs +39 -0
  320. package/ui/arkgraph/details.css +3 -0
  321. package/ui/arkgraph/entry.jsx +28 -0
  322. package/ui/arkgraph/experiment-curves.mjs +17 -0
  323. package/ui/arkgraph/experiment-selection.mjs +15 -0
  324. package/ui/arkgraph/experiment-style.css +26 -0
  325. package/ui/arkgraph/experiment-ui.jsx +32 -0
  326. package/ui/arkgraph/frame.html +1 -0
  327. package/ui/arkgraph/graph-gestures.mjs +62 -0
  328. package/ui/arkgraph/label-layout.mjs +57 -0
  329. package/ui/arkgraph/locales/en.json +229 -0
  330. package/ui/arkgraph/locales/source-types.json +15 -0
  331. package/ui/arkgraph/localization-build.mjs +27 -0
  332. package/ui/arkgraph/material-build.mjs +23 -0
  333. package/ui/arkgraph/material-colors.mjs +39 -0
  334. package/ui/arkgraph/material-style.css +15 -0
  335. package/ui/arkgraph/open-graph.jsx +326 -0
  336. package/ui/arkgraph/outline.jsx +49 -0
  337. package/ui/arkgraph/package-lock.json +888 -0
  338. package/ui/arkgraph/package.json +17 -0
  339. package/ui/arkgraph/reading-layout.mjs +130 -0
  340. package/ui/arkgraph/reading-presentation.mjs +73 -0
  341. package/ui/arkgraph/record-detail.css +51 -0
  342. package/ui/arkgraph/record-details.jsx +29 -0
  343. package/ui/arkgraph/research-types.mjs +31 -0
  344. package/ui/arkgraph/selection-mark.jsx +6 -0
  345. package/ui/arkgraph/soft-spine.mjs +26 -0
  346. package/ui/arkgraph/steering-style.css +187 -0
  347. package/ui/arkgraph/style.css +272 -0
@@ -0,0 +1,51 @@
1
+ import { cp, mkdir, readFile, writeFile } from 'node:fs/promises';
2
+ import path from 'node:path';
3
+ import { assembleCapDirectory, CAP_PROFILE } from '../cap/v2/core.mjs';
4
+ import { canonicalJsonBytes } from '../cap/v2/canonical-json.mjs';
5
+ import { capRecordId } from '../cap/v2/research-records.mjs';
6
+ import { signCapArtifact, attachCapAttestation } from '../cap/v2/attestation.mjs';
7
+ import { verifyCapDirectory } from '../cap/v2/verify.mjs';
8
+ import { HANDOFF_ROLE, HANDOFF_SCHEMA, assertPortableHandoff, validateCodeFiles } from './handoff.mjs';
9
+ import { sha256Value } from '../util.mjs';
10
+
11
+ /** Add delivery metadata; all scientific Records and original Blobs stay identical. */
12
+ export async function assembleHandoffSupplement({ sourceDirectory, directory, handoff, signingKeyPath }) {
13
+ const source = await verifyCapDirectory(sourceDirectory);
14
+ if (!source.valid) throw Error('Invalid source CAP');
15
+ if (handoff.schema !== HANDOFF_SCHEMA || handoff.delivery?.replayReady !== true) throw Error('Handoff is not ready');
16
+ assertPortableHandoff(handoff); validateCodeFiles(handoff.files);
17
+ if (handoff.policy) throw Error('Do not publish historical private verification policies');
18
+ const manifest = source.manifest;
19
+ if (manifest.blobs.some(b => b.roles.includes(HANDOFF_ROLE))) throw Error('Source already has a handoff; select its original execution CAP');
20
+ const records = await Promise.all(manifest.records.map(r => readFile(path.join(sourceDirectory, r.path), 'utf8').then(JSON.parse)));
21
+ const execution = records.find(r => r.role === 'execution');
22
+ if ((execution.citeark.sharedExecutionRunId ?? execution.citeark.runId) !== handoff.sourceRunId) throw Error('Handoff execution mismatch');
23
+ const blobs = await Promise.all(manifest.blobs.map(async b => ({ ...b,
24
+ ...(b.availability === 'embedded' ? { bytes: await readFile(path.join(sourceDirectory, b.path)) } : {}) })));
25
+ blobs.push({ id: capRecordId('blob', `handoff:${sha256Value(handoff)}`), roles: [HANDOFF_ROLE],
26
+ mediaType: 'application/json', bytes: canonicalJsonBytes(handoff),
27
+ rights: { statement: 'CiteArk execution handoff; retained implementation keeps its original source rights.' } });
28
+ const built = await assembleCapDirectory({ directory,
29
+ artifact: { id: capRecordId('artifact', `${source.artifactDigest}:handoff:${sha256Value(handoff)}`),
30
+ createdAt: handoff.delivery.generatedAt, createdBy: manifest.artifact.createdBy.ref },
31
+ profiles: manifest.profiles.filter(p => p !== CAP_PROFILE.core),
32
+ roots: manifest.roots.map(({role,ref}) => ({role,ref})), records, blobs,
33
+ relations: [...manifest.relations, { relationship: 'derivedFrom', artifactDigest: source.artifactDigest,
34
+ summary: 'Adds execution handoff only. No experiment rerun, scientific Record change or reassessment.' }] });
35
+ for (const name of ['projections', 'preview']) await cp(path.join(sourceDirectory, name), path.join(directory, name), { recursive: true }).catch(error => { if (error.code !== 'ENOENT') throw error; });
36
+ if (manifest.profiles.includes(CAP_PROFILE.publicBundle)) {
37
+ await mkdir(directory, { recursive: true });
38
+ await writeFile(path.join(directory, 'ro-crate-metadata.json'), canonicalJsonBytes({
39
+ '@context': 'https://w3id.org/ro/crate/1.3/context', '@graph': [
40
+ {'@id':'ro-crate-metadata.json','@type':'CreativeWork',about:{'@id':'./'}},
41
+ {'@id':'./','@type':'Dataset',identifier:built.artifactDigest,hasPart:built.manifest.records.map(r=>({'@id':r.path}))},
42
+ ],
43
+ }));
44
+ }
45
+ await attachCapAttestation({ directory, attestation: await signCapArtifact({ artifactDigest: built.artifactDigest,
46
+ actor: { ref: manifest.artifact.createdBy.ref }, signingKeyPath }) });
47
+ const verified = await verifyCapDirectory(directory);
48
+ if (!verified.valid) throw Error(verified.issues.join('\n'));
49
+ if (manifest.records.some(r => !verified.manifest.records.some(next => next.id === r.id && next.digest === r.digest))) throw Error('Scientific Records changed');
50
+ return { ...built, sourceArtifactDigest: source.artifactDigest, sourceRunId: execution.citeark.runId };
51
+ }
@@ -0,0 +1,43 @@
1
+ import { createInterface } from 'node:readline';
2
+ import { Writable } from 'node:stream';
3
+
4
+ export function cancelled() {
5
+ return Object.assign(Error('Operation cancelled.'), { exitCode: 130 });
6
+ }
7
+
8
+ export function createTerminalPrompter({ input = process.stdin, output = process.stdout } = {}) {
9
+ return {
10
+ async ask(label, { defaultValue = '', secret = false } = {}) {
11
+ if (!input.isTTY) throw Error('An interactive terminal is required. For automation, provide model options or use --dry-run.');
12
+ let muted = false;
13
+ const display = new Writable({ write(chunk, encoding, callback) {
14
+ if (!muted) output.write(chunk, encoding);
15
+ callback();
16
+ } });
17
+ display.isTTY = output.isTTY;
18
+ display.columns = output.columns;
19
+ const rl = createInterface({ input, output: display, terminal: true });
20
+ const prompt = `${label}${defaultValue ? ` [${defaultValue}]` : ''}: `;
21
+ return await new Promise((resolve, reject) => {
22
+ let settled = false;
23
+ const abort = () => {
24
+ if (settled) return;
25
+ settled = true; muted = false; output.write('\n'); rl.close(); reject(cancelled());
26
+ };
27
+ rl.on('SIGINT', abort);
28
+ rl.on('close', abort);
29
+ rl.question(prompt, value => {
30
+ settled = true;
31
+ muted = false;
32
+ if (secret) output.write('\n');
33
+ rl.close();
34
+ resolve(value.trim() || defaultValue);
35
+ });
36
+ muted = secret;
37
+ });
38
+ },
39
+ async confirm(label) {
40
+ return /^(?:y|yes)$/i.test(await this.ask(`${label} [y/N]`));
41
+ },
42
+ };
43
+ }
@@ -0,0 +1,75 @@
1
+ import { redactString } from "../cap/redaction.mjs";
2
+
3
+ /**
4
+ * Turn the execution Agent's deliberately free-form analysis into a portable
5
+ * CAP preview. Scientific status remains canonical in the root Assessment
6
+ * Record; this file is an Agent interpretation intended for people.
7
+ */
8
+ export function renderAgentDiagnosis({ contract, result, assessment, integrity }) {
9
+ const claimId = contract.research?.claimId ?? contract.research?.claimVersionId ?? "unknown claim";
10
+ const agentText = freeformDiagnosis(result);
11
+ const comparisons = Array.isArray(assessment.measurementAssessments)
12
+ ? assessment.measurementAssessments
13
+ : [];
14
+ const lines = [
15
+ "# Agent diagnosis",
16
+ "",
17
+ "> This is the execution Agent's interpretation of the recorded run. Canonical observations and the reproduction conclusion remain in the CAP Evidence and root Assessment Records.",
18
+ "",
19
+ "## Outcome",
20
+ "",
21
+ `- Claim: \`${claimId}\``,
22
+ `- Execution: \`${result.execution?.status ?? "unknown"}\``,
23
+ `- Integrity: \`${integrity.status ?? "unknown"}\``,
24
+ `- Verification: \`${assessment.verificationStatus ?? "inconclusive"}\``,
25
+ ];
26
+ for (const item of comparisons) {
27
+ const comparison = item.comparison ?? {};
28
+ lines.push(
29
+ `- ${comparison.metric ?? item.measurementId ?? "measurement"}: reported ${display(comparison.reportedValue)} ${comparison.unit ?? ""}; observed ${display(comparison.observedValue)}; status \`${item.verificationStatus ?? "inconclusive"}\``,
30
+ );
31
+ }
32
+ lines.push(
33
+ "",
34
+ "## Analysis",
35
+ "",
36
+ redactString(agentText),
37
+ "",
38
+ );
39
+ if (Array.isArray(result.limitations) && result.limitations.length) {
40
+ lines.push("## Remaining uncertainty", "");
41
+ for (const limitation of result.limitations) lines.push(`- ${redactString(String(limitation))}`);
42
+ lines.push("");
43
+ }
44
+ const markdown = `${lines.join("\n").trim()}\n`;
45
+ return {
46
+ markdown,
47
+ summary: diagnosisSummary(agentText),
48
+ };
49
+ }
50
+
51
+ function freeformDiagnosis(result) {
52
+ if (typeof result?.diagnosis === "string" && result.diagnosis.trim()) {
53
+ return result.diagnosis.trim();
54
+ }
55
+ if (typeof result?.execution?.failure?.reason === "string") {
56
+ return result.execution.failure.reason.trim();
57
+ }
58
+ if (typeof result?.execution?.summary === "string") {
59
+ return result.execution.summary.trim();
60
+ }
61
+ return "The execution Agent did not identify a more specific cause from the available evidence.";
62
+ }
63
+
64
+ function diagnosisSummary(value) {
65
+ const paragraph = String(value)
66
+ .replace(/^#+\s.*$/gm, "")
67
+ .split(/\n\s*\n/)
68
+ .map((item) => item.replaceAll(/\s+/g, " ").trim())
69
+ .find(Boolean) ?? "The Agent did not identify a more specific cause.";
70
+ return redactString(paragraph).slice(0, 1_000);
71
+ }
72
+
73
+ function display(value) {
74
+ return typeof value === "number" && Number.isFinite(value) ? String(value) : "unavailable";
75
+ }
@@ -0,0 +1,46 @@
1
+ import { readFile, readdir } from "node:fs/promises";
2
+ import path from "node:path";
3
+
4
+ import { redactString } from "../cap/redaction.mjs";
5
+
6
+ /**
7
+ * Recover the best per-target explanation even when the wider pipeline exits
8
+ * before it can seal a CAP. Prefer the execution Agent's own result; otherwise
9
+ * retain a clearly labelled workflow error instead of silently leaving the
10
+ * claim in an unexplained pending state.
11
+ */
12
+ export async function readTargetFailureDiagnosis(pipelineDirectory, error) {
13
+ for (const resultPath of await findNamedFiles(pipelineDirectory, "result.json")) {
14
+ try {
15
+ const result = JSON.parse(await readFile(resultPath, "utf8"));
16
+ const agentText = typeof result?.diagnosis === "string" && result.diagnosis.trim()
17
+ ? result.diagnosis.trim()
18
+ : typeof result?.execution?.failure?.reason === "string" && result.execution.failure.reason.trim()
19
+ ? result.execution.failure.reason.trim()
20
+ : null;
21
+ if (agentText) {
22
+ return { text: redactString(agentText).slice(0, 4_000), source: "execution-agent" };
23
+ }
24
+ } catch {
25
+ // A partial result from one attempt must not hide a valid diagnosis from
26
+ // another attempt for the same target.
27
+ }
28
+ }
29
+ return {
30
+ text: redactString(error instanceof Error ? error.message : String(error)).slice(0, 4_000),
31
+ source: "workflow",
32
+ };
33
+ }
34
+
35
+ async function findNamedFiles(root, filename) {
36
+ const matches = [];
37
+ async function visit(directory) {
38
+ for (const entry of await readdir(directory, { withFileTypes: true }).catch(() => [])) {
39
+ const target = path.join(directory, entry.name);
40
+ if (entry.isDirectory()) await visit(target);
41
+ else if (entry.isFile() && entry.name === filename) matches.push(target);
42
+ }
43
+ }
44
+ await visit(root);
45
+ return matches.sort();
46
+ }
@@ -0,0 +1,54 @@
1
+ import { CiteArkError, safeRelativePath } from "../util.mjs";
2
+ import { fastTextClassificationParser } from "./parsers/fasttext-classification.mjs";
3
+ import { jsonScalarParser } from "./parsers/json-scalar.mjs";
4
+ import { simcseSentevalParser } from "./parsers/simcse-senteval.mjs";
5
+ import { starSpaceClassificationParser } from "./parsers/starspace-classification.mjs";
6
+
7
+ const PARSERS = new Map(
8
+ [jsonScalarParser, simcseSentevalParser, fastTextClassificationParser, starSpaceClassificationParser]
9
+ .map((parser) => [`${parser.id}@${parser.version}`, parser]),
10
+ );
11
+
12
+ /**
13
+ * Pure parser capability registry. It does not depend on execution contracts or
14
+ * research normalization, which keeps protocol validation acyclic.
15
+ */
16
+ export function registerEvidenceParser(parser) {
17
+ if (!parser?.id || !parser?.version || typeof parser.parse !== "function") {
18
+ throw new CiteArkError("证据解析器接口无效");
19
+ }
20
+ PARSERS.set(parserKey(parser), parser);
21
+ }
22
+
23
+ export function listEvidenceParsers() {
24
+ return [...PARSERS.values()].map(
25
+ ({ id, version, description, configSchema }) => ({
26
+ id,
27
+ version,
28
+ description,
29
+ configSchema,
30
+ }),
31
+ );
32
+ }
33
+
34
+ export function getEvidenceParser(descriptor) {
35
+ return PARSERS.get(parserKey(descriptor));
36
+ }
37
+
38
+ export function validateEvidenceParserDescriptor(descriptor, measurement = {}) {
39
+ if (!descriptor || typeof descriptor !== "object") {
40
+ return ["parser 必须是对象"];
41
+ }
42
+ if (!safeRelativePath(descriptor.evidencePath)) {
43
+ return ["evidencePath 必须是 output 内的安全相对路径"];
44
+ }
45
+ const parser = getEvidenceParser(descriptor);
46
+ if (!parser) {
47
+ return [`没有注册证据解析器:${descriptor.id}@${descriptor.version}`];
48
+ }
49
+ return parser.validateConfig?.(descriptor.config ?? {}, measurement) ?? [];
50
+ }
51
+
52
+ function parserKey(parser) {
53
+ return `${parser?.id}@${parser?.version}`;
54
+ }
@@ -0,0 +1,82 @@
1
+ import { CiteArkError } from "../../util.mjs";
2
+
3
+ export const fastTextClassificationParser = {
4
+ id: "fasttext.classification.test",
5
+ version: "1",
6
+ description: "从 fastText test 的原始 N、P@k、R@k 输出中确定性提取单标签分类准确率。",
7
+ configSchema: {
8
+ required: ["expectedExamples"],
9
+ properties: {
10
+ expectedExamples: "预期测试样本数,例如 AG News 为 7600",
11
+ k: "fastText 的 k;默认 1",
12
+ scale: "从 0-1 比例换算为目标单位的倍率;默认 100",
13
+ },
14
+ output: {
15
+ metric: "test_accuracy",
16
+ unit: "percentage_points",
17
+ },
18
+ },
19
+ validateConfig(config = {}, measurement = {}) {
20
+ const issues = [];
21
+ if (!Number.isInteger(config.expectedExamples) || config.expectedExamples <= 0) {
22
+ issues.push("config.expectedExamples 必须是正整数");
23
+ }
24
+ if (config.k !== undefined && (!Number.isInteger(config.k) || config.k <= 0)) {
25
+ issues.push("config.k 必须是正整数");
26
+ }
27
+ if (config.scale !== undefined && !Number.isFinite(config.scale)) {
28
+ issues.push("config.scale 必须是有限数值");
29
+ }
30
+ if (measurement.metric && measurement.metric !== "test_accuracy") {
31
+ issues.push("measurement.metric 必须是 test_accuracy");
32
+ }
33
+ if (measurement.unit && measurement.unit !== "percentage_points") {
34
+ issues.push("measurement.unit 必须是 percentage_points");
35
+ }
36
+ return issues;
37
+ },
38
+ parse(text, config = {}) {
39
+ const k = config.k ?? 1;
40
+ const scale = config.scale ?? 100;
41
+ const examples = integerLine(text, "N");
42
+ const precision = numericLine(text, `P@${k}`);
43
+ const recall = numericLine(text, `R@${k}`);
44
+ if (examples !== config.expectedExamples) {
45
+ throw new CiteArkError(`fastText 测试样本数不一致:observed=${examples}, expected=${config.expectedExamples}`);
46
+ }
47
+ if (precision < 0 || precision > 1 || recall < 0 || recall > 1) {
48
+ throw new CiteArkError("fastText P@k/R@k 必须位于 0 到 1 之间");
49
+ }
50
+ if (Math.abs(precision - recall) > 1e-12) {
51
+ throw new CiteArkError(`单标签 top-${k} 分类的 P@k 与 R@k 不一致:P=${precision}, R=${recall}`);
52
+ }
53
+ const accuracy = precision * scale;
54
+ return {
55
+ observations: [
56
+ { metric: "test_accuracy", value: accuracy, unit: "percentage_points" },
57
+ { metric: "test_examples", value: examples, unit: "count" },
58
+ ],
59
+ primary: { metric: "test_accuracy", value: accuracy, unit: "percentage_points" },
60
+ checks: [
61
+ { name: "expected_test_examples", status: "passed", observed: examples, expected: config.expectedExamples },
62
+ { name: "single_label_precision_recall_consistency", status: "passed", precision, recall, k },
63
+ { name: "proportion_to_percentage_points", status: "passed", scale, transformedValue: accuracy },
64
+ ],
65
+ };
66
+ },
67
+ };
68
+
69
+ function integerLine(text, label) {
70
+ const value = numericLine(text, label);
71
+ if (!Number.isInteger(value)) throw new CiteArkError(`fastText ${label} 不是整数`);
72
+ return value;
73
+ }
74
+
75
+ function numericLine(text, label) {
76
+ const escaped = label.replaceAll(/[.*+?^${}()|[\]\\]/g, "\\$&");
77
+ const matches = [...text.matchAll(new RegExp(`^\\s*${escaped}\\s+([0-9]+(?:\\.[0-9]+)?)\\s*$`, "gmi"))];
78
+ if (matches.length !== 1) throw new CiteArkError(`fastText 原始证据必须且只能包含一行 ${label}`);
79
+ const value = Number(matches[0][1]);
80
+ if (!Number.isFinite(value)) throw new CiteArkError(`fastText ${label} 不是有限数值`);
81
+ return value;
82
+ }
@@ -0,0 +1,96 @@
1
+ import { CiteArkError } from "../../util.mjs";
2
+
3
+ export const jsonScalarParser = {
4
+ id: "json.scalar",
5
+ version: "1",
6
+ description:
7
+ "从 JSON Pointer 指向的数值读取原始标量,并可用公开的 scale/offset 做确定性单位换算。",
8
+ configSchema: {
9
+ required: ["metric", "unit"],
10
+ properties: {
11
+ pointer: "JSON Pointer;默认空字符串表示整个文档",
12
+ metric: "解析后主指标名称,必须与 experiment measurement.metric 一致",
13
+ unit: "解析后单位,必须与 experiment measurement.unit 一致",
14
+ scale: "可选有限数值;parsed = raw * scale + offset,默认 1",
15
+ offset: "可选有限数值;parsed = raw * scale + offset,默认 0",
16
+ },
17
+ example: {
18
+ pointer: "/accuracy",
19
+ metric: "accuracy",
20
+ unit: "percentage_points",
21
+ scale: 100,
22
+ },
23
+ },
24
+ validateConfig(config = {}, measurement = {}) {
25
+ const issues = [];
26
+ if (typeof config.metric !== "string" || !config.metric) {
27
+ issues.push("config.metric 必须是非空字符串");
28
+ }
29
+ if (typeof config.unit !== "string" || !config.unit) {
30
+ issues.push("config.unit 必须是非空字符串");
31
+ }
32
+ if (config.pointer !== undefined && typeof config.pointer !== "string") {
33
+ issues.push("config.pointer 必须是字符串");
34
+ }
35
+ if (config.scale !== undefined && !Number.isFinite(config.scale)) {
36
+ issues.push("config.scale 必须是有限数值");
37
+ }
38
+ if (config.offset !== undefined && !Number.isFinite(config.offset)) {
39
+ issues.push("config.offset 必须是有限数值");
40
+ }
41
+ if (measurement.metric && config.metric !== measurement.metric) {
42
+ issues.push("config.metric 必须与 measurement.metric 一致");
43
+ }
44
+ if (measurement.unit && config.unit !== measurement.unit) {
45
+ issues.push("config.unit 必须与 measurement.unit 一致");
46
+ }
47
+ return issues;
48
+ },
49
+ parse(text, config = {}) {
50
+ let value;
51
+ try {
52
+ const document = JSON.parse(text);
53
+ value = resolvePointer(document, config.pointer ?? "");
54
+ } catch (error) {
55
+ throw new CiteArkError(`JSON 标量证据解析失败:${error.message}`, { cause: error });
56
+ }
57
+ if (typeof value !== "number" || !Number.isFinite(value)) throw new CiteArkError("JSON 指针没有指向有限数值");
58
+ const metric = config.metric;
59
+ const unit = config.unit;
60
+ if (typeof metric !== "string" || !metric || typeof unit !== "string" || !unit) {
61
+ throw new CiteArkError("json.scalar parser 需要 config.metric 和 config.unit");
62
+ }
63
+ const scale = config.scale ?? 1;
64
+ const offset = config.offset ?? 0;
65
+ if (!Number.isFinite(scale) || !Number.isFinite(offset)) {
66
+ throw new CiteArkError("json.scalar parser 的 scale 和 offset 必须是有限数值");
67
+ }
68
+ const rawValue = value;
69
+ value = rawValue * scale + offset;
70
+ if (!Number.isFinite(value)) throw new CiteArkError("JSON 标量单位换算后不是有限数值");
71
+ return {
72
+ observations: [{ metric, value, unit }],
73
+ primary: { metric, value, unit },
74
+ checks: [
75
+ { name: "finite_scalar", status: "passed", rawValue },
76
+ {
77
+ name: "affine_unit_transform",
78
+ status: "passed",
79
+ scale,
80
+ offset,
81
+ transformedValue: value,
82
+ },
83
+ ],
84
+ };
85
+ },
86
+ };
87
+
88
+ function resolvePointer(document, pointer) {
89
+ if (pointer === "") return document;
90
+ if (!pointer.startsWith("/")) throw new Error("pointer 必须是 JSON Pointer");
91
+ return pointer
92
+ .slice(1)
93
+ .split("/")
94
+ .map((part) => part.replaceAll("~1", "/").replaceAll("~0", "~"))
95
+ .reduce((value, key) => value[key], document);
96
+ }
@@ -0,0 +1,104 @@
1
+ import { CiteArkError } from "../../util.mjs";
2
+
3
+ export const simcseSentevalParser = {
4
+ id: "simcse.senteval.sts-table",
5
+ version: "1",
6
+ description:
7
+ "从 SimCSE SentEval 的七项 STS Markdown 汇总表提取分数,并独立复算算术平均值。",
8
+ configSchema: {
9
+ required: [],
10
+ properties: {
11
+ expectedTasks:
12
+ "可选任务名数组;默认 STS12-16、STSBenchmark、SICKRelatedness",
13
+ },
14
+ output: {
15
+ metric: "average_spearman",
16
+ unit: "percentage_points",
17
+ },
18
+ },
19
+ validateConfig(config = {}, measurement = {}) {
20
+ const issues = [];
21
+ if (
22
+ config.expectedTasks !== undefined &&
23
+ (!Array.isArray(config.expectedTasks) ||
24
+ config.expectedTasks.length === 0 ||
25
+ config.expectedTasks.some(
26
+ (task) => typeof task !== "string" || !task,
27
+ ))
28
+ ) {
29
+ issues.push("config.expectedTasks 必须是非空字符串数组");
30
+ }
31
+ if (measurement.metric && measurement.metric !== "average_spearman") {
32
+ issues.push("measurement.metric 必须是 average_spearman");
33
+ }
34
+ if (
35
+ measurement.unit &&
36
+ measurement.unit !== "percentage_points"
37
+ ) {
38
+ issues.push("measurement.unit 必须是 percentage_points");
39
+ }
40
+ return issues;
41
+ },
42
+ parse(text, config = {}) {
43
+ const expectedTasks = config.expectedTasks ?? [
44
+ "STS12",
45
+ "STS13",
46
+ "STS14",
47
+ "STS15",
48
+ "STS16",
49
+ "STSBenchmark",
50
+ "SICKRelatedness",
51
+ ];
52
+ const lines = text.split(/\r?\n/);
53
+ let headerIndex = -1;
54
+ let headers = [];
55
+ for (let index = 0; index < lines.length; index += 1) {
56
+ const cells = tableCells(lines[index]);
57
+ if (expectedTasks.every((task) => cells.includes(task)) && cells.some((cell) => /^Avg\.?$/i.test(cell))) {
58
+ headerIndex = index;
59
+ headers = cells;
60
+ }
61
+ }
62
+ if (headerIndex < 0) throw new CiteArkError("SimCSE 证据中没有找到七项 STS 汇总表头");
63
+ const valueLine = lines.slice(headerIndex + 1).find((line) => {
64
+ const cells = tableCells(line);
65
+ return cells.length === headers.length && cells.every((cell) => Number.isFinite(Number(cell)));
66
+ });
67
+ if (!valueLine) throw new CiteArkError("SimCSE 证据中没有找到 STS 汇总数值行");
68
+ const values = tableCells(valueLine).map(Number);
69
+ const byHeader = new Map(headers.map((header, index) => [header.replace(/\.$/, ""), values[index]]));
70
+ const taskValues = expectedTasks.map((task) => {
71
+ const value = byHeader.get(task);
72
+ if (!Number.isFinite(value)) throw new CiteArkError(`SimCSE 汇总表缺少 ${task}`);
73
+ return { metric: `spearman.${task}`, value, unit: "percentage_points", scope: task };
74
+ });
75
+ const printedAverage = byHeader.get("Avg");
76
+ if (!Number.isFinite(printedAverage)) throw new CiteArkError("SimCSE 汇总表缺少 Avg.");
77
+ const calculatedAverage = taskValues.reduce((sum, item) => sum + item.value, 0) / taskValues.length;
78
+ const roundingDifference = Math.abs(calculatedAverage - printedAverage);
79
+ if (roundingDifference > 0.011) {
80
+ throw new CiteArkError(`SimCSE Avg. 与七项算术平均不一致:printed=${printedAverage}, calculated=${calculatedAverage}`);
81
+ }
82
+ return {
83
+ observations: [
84
+ ...taskValues,
85
+ { metric: "average_spearman", value: printedAverage, unit: "percentage_points", scope: "seven_sts_tasks" },
86
+ ],
87
+ primary: { metric: "average_spearman", value: printedAverage, unit: "percentage_points" },
88
+ checks: [
89
+ { name: "expected_task_count", status: "passed", observed: taskValues.length, expected: expectedTasks.length },
90
+ { name: "arithmetic_mean", status: "passed", observed: calculatedAverage, printed: printedAverage, maximumRoundingDifference: 0.011 },
91
+ ],
92
+ };
93
+ },
94
+ };
95
+
96
+ function tableCells(line) {
97
+ if (!line.trim().startsWith("|")) return [];
98
+ return line
99
+ .trim()
100
+ .replace(/^\|/, "")
101
+ .replace(/\|$/, "")
102
+ .split("|")
103
+ .map((cell) => cell.trim());
104
+ }
@@ -0,0 +1,78 @@
1
+ import { CiteArkError } from "../../util.mjs";
2
+
3
+ export const starSpaceClassificationParser = {
4
+ id: "starspace.classification.test",
5
+ version: "1",
6
+ description: "从 StarSpace test 的原始 Evaluation Metrics 输出中确定性提取 hit@1 分类准确率。",
7
+ configSchema: {
8
+ required: ["expectedExamples"],
9
+ properties: {
10
+ expectedExamples: "预期测试样本数,例如 AG News 为 7600",
11
+ scale: "从 0-1 比例换算为目标单位的倍率;默认 100",
12
+ },
13
+ output: {
14
+ metric: "test_accuracy",
15
+ unit: "percentage_points",
16
+ },
17
+ },
18
+ validateConfig(config = {}, measurement = {}) {
19
+ const issues = [];
20
+ if (!Number.isInteger(config.expectedExamples) || config.expectedExamples <= 0) {
21
+ issues.push("config.expectedExamples 必须是正整数");
22
+ }
23
+ if (config.scale !== undefined && !Number.isFinite(config.scale)) {
24
+ issues.push("config.scale 必须是有限数值");
25
+ }
26
+ if (measurement.metric && measurement.metric !== "test_accuracy") {
27
+ issues.push("measurement.metric 必须是 test_accuracy");
28
+ }
29
+ if (measurement.unit && measurement.unit !== "percentage_points") {
30
+ issues.push("measurement.unit 必须是 percentage_points");
31
+ }
32
+ return issues;
33
+ },
34
+ parse(text, config = {}) {
35
+ const matches = [...text.matchAll(
36
+ /hit@1:\s*([0-9]+(?:\.[0-9]+)?)\s+hit@10:\s*([0-9]+(?:\.[0-9]+)?)\s+hit@20:\s*([0-9]+(?:\.[0-9]+)?)\s+hit@50:\s*([0-9]+(?:\.[0-9]+)?)\s+mean ranks\s*:\s*([0-9]+(?:\.[0-9]+)?)\s+Total examples\s*:\s*([0-9]+)/gi,
37
+ )];
38
+ if (matches.length !== 1) {
39
+ throw new CiteArkError("StarSpace 原始证据必须且只能包含一组 Evaluation Metrics");
40
+ }
41
+ const [, hit1Text, hit10Text, hit20Text, hit50Text, meanRankText, examplesText] = matches[0];
42
+ const hit1 = Number(hit1Text);
43
+ const hit10 = Number(hit10Text);
44
+ const hit20 = Number(hit20Text);
45
+ const hit50 = Number(hit50Text);
46
+ const meanRank = Number(meanRankText);
47
+ const examples = Number(examplesText);
48
+ const scale = config.scale ?? 100;
49
+ if (examples !== config.expectedExamples) {
50
+ throw new CiteArkError(`StarSpace 测试样本数不一致:observed=${examples}, expected=${config.expectedExamples}`);
51
+ }
52
+ for (const [name, value] of Object.entries({ hit1, hit10, hit20, hit50 })) {
53
+ if (!Number.isFinite(value) || value < 0 || value > 1) {
54
+ throw new CiteArkError(`StarSpace ${name} 必须位于 0 到 1 之间`);
55
+ }
56
+ }
57
+ if (!(hit1 <= hit10 && hit10 <= hit20 && hit20 <= hit50)) {
58
+ throw new CiteArkError("StarSpace hit@k 不满足随 k 单调不减");
59
+ }
60
+ if (!Number.isFinite(meanRank) || meanRank < 1) {
61
+ throw new CiteArkError("StarSpace mean ranks 必须是不小于 1 的有限数值");
62
+ }
63
+ const accuracy = hit1 * scale;
64
+ return {
65
+ observations: [
66
+ { metric: "test_accuracy", value: accuracy, unit: "percentage_points" },
67
+ { metric: "test_examples", value: examples, unit: "count" },
68
+ { metric: "mean_rank", value: meanRank, unit: "rank" },
69
+ ],
70
+ primary: { metric: "test_accuracy", value: accuracy, unit: "percentage_points" },
71
+ checks: [
72
+ { name: "expected_test_examples", status: "passed", observed: examples, expected: config.expectedExamples },
73
+ { name: "hit_at_k_monotonicity", status: "passed", hit1, hit10, hit20, hit50 },
74
+ { name: "proportion_to_percentage_points", status: "passed", scale, transformedValue: accuracy },
75
+ ],
76
+ };
77
+ },
78
+ };