@citeark/agent 0.3.20
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +202 -0
- package/README.md +128 -0
- package/data/dataset-source-registry.v1.json +300 -0
- package/dist/arkgraph/boot.js +6 -0
- package/dist/arkgraph/index.html +1 -0
- package/dist/arkgraph/viewer.css +1 -0
- package/dist/arkgraph/viewer.en.css +1 -0
- package/dist/arkgraph/viewer.en.js +49 -0
- package/dist/arkgraph/viewer.en.js.LEGAL.txt +56 -0
- package/dist/arkgraph/viewer.js +49 -0
- package/dist/arkgraph/viewer.js.LEGAL.txt +56 -0
- package/docker/claude-code/Dockerfile +97 -0
- package/docker/claude-code/codex-pro-relay.mjs +466 -0
- package/docker/claude-code/runtime-contract-check.mjs +79 -0
- package/docs/arkgraph-reading.md +79 -0
- package/docs/configuration.md +100 -0
- package/docs/integration.md +92 -0
- package/docs/maturity-plan.md +27 -0
- package/docs/npm-release.md +44 -0
- package/docs/paper-reading.md +40 -0
- package/docs/research-plan-granularity.md +27 -0
- package/docs/terminal.md +49 -0
- package/examples/toy-evaluation/compile-task.json +27 -0
- package/examples/toy-evaluation/paper.md +5 -0
- package/examples/toy-evaluation/repository/README.md +9 -0
- package/examples/toy-evaluation/repository/checkpoint.json +4 -0
- package/examples/toy-evaluation/repository/evaluate.py +17 -0
- package/examples/toy-evaluation/task.json +81 -0
- package/package.json +59 -0
- package/prompts/compile-research.md +58 -0
- package/prompts/execute-contract.md +72 -0
- package/prompts/execute-workspace-simple.md +51 -0
- package/prompts/execute-workspace.md +34 -0
- package/prompts/prepare-reproduction.md +82 -0
- package/prompts/repair-research.md +45 -0
- package/protocol/CAP.md +129 -0
- package/protocol/LICENSE +12 -0
- package/protocol/MAPPINGS.md +72 -0
- package/protocol/README.md +38 -0
- package/protocol/conformance-v2.0-alpha.1.json +36 -0
- package/protocol/examples/arkgraph/checkpoint-evaluation.json +309 -0
- package/protocol/examples/arkgraph/fixtures.mjs +49 -0
- package/protocol/examples/arkgraph/paper-free.json +291 -0
- package/protocol/examples/arkgraph/partial-failure.json +344 -0
- package/protocol/examples/arkgraph/training-evaluation.json +443 -0
- package/protocol/profiles/agent-trace.md +16 -0
- package/protocol/profiles/computational-run.md +16 -0
- package/protocol/profiles/core.md +15 -0
- package/protocol/profiles/public-bundle.md +18 -0
- package/protocol/profiles/reproduction.md +29 -0
- package/protocol/profiles/research-compilation.md +44 -0
- package/protocol/profiles/research-plan.md +39 -0
- package/protocol/profiles/restricted-evidence.md +15 -0
- package/runtime/bootstrap-autodl-runtime.sh +314 -0
- package/runtime/create-runtime-venv.sh +41 -0
- package/runtime/install-local-cpu-runtime.sh +23 -0
- package/runtime/install-scientific-runtime.sh +153 -0
- package/runtime/mineru/parse.py +62 -0
- package/runtime/mineru/requirements.txt +4 -0
- package/runtime/requirements-baseline.txt +38 -0
- package/schemas/cap/v2/activity.schema.json +47 -0
- package/schemas/cap/v2/agent.schema.json +32 -0
- package/schemas/cap/v2/assertion.schema.json +110 -0
- package/schemas/cap/v2/descriptor.schema.json +243 -0
- package/schemas/cap/v2/entity.schema.json +64 -0
- package/schemas/cap/v2/manifest.schema.json +67 -0
- package/schemas/cap/v2/relation.schema.json +82 -0
- package/schemas/compute-catalog.schema.json +63 -0
- package/schemas/compute-decision.schema.json +27 -0
- package/schemas/execution-contract.schema.json +1024 -0
- package/schemas/research-card.schema.json +30 -0
- package/schemas/research-inventory-draft.schema.json +366 -0
- package/schemas/research.schema.json +1044 -0
- package/schemas/result.schema.json +173 -0
- package/schemas/verification-policy.schema.json +47 -0
- package/schemas/verified-conclusion.schema.json +58 -0
- package/schemas/workspace-summary.schema.json +24 -0
- package/scripts/build-arkgraph-view.mjs +12 -0
- package/scripts/check-execution-feasibility.mjs +24 -0
- package/scripts/check-syntax.mjs +15 -0
- package/scripts/deterministic-asset-preparation.py +438 -0
- package/scripts/package-cap.mjs +23 -0
- package/scripts/package-local-agent.mjs +23 -0
- package/scripts/preview-arkgraph.mjs +25 -0
- package/scripts/replay-research-compiler-candidate.mjs +134 -0
- package/scripts/review-compiler-sources.mjs +44 -0
- package/scripts/run-asset-preparation.sh +17 -0
- package/scripts/run-research-plan.mjs +98 -0
- package/scripts/validate-asset-preparation.py +290 -0
- package/scripts/verify-local-runtime.mjs +57 -0
- package/scripts/verify-npm-package.mjs +57 -0
- package/src/adapters/paper2agent.mjs +107 -0
- package/src/assets/cache.mjs +159 -0
- package/src/assets/compute.mjs +98 -0
- package/src/assets/executor.mjs +145 -0
- package/src/assets/lifecycle.mjs +213 -0
- package/src/assets/manifest.mjs +242 -0
- package/src/assets/opportunistic-preparation.mjs +81 -0
- package/src/assets/plan.mjs +411 -0
- package/src/assets/prompts.mjs +29 -0
- package/src/assets/public-asset-probe.mjs +525 -0
- package/src/assets/qualification.mjs +119 -0
- package/src/assets/readiness.mjs +130 -0
- package/src/assets/reproduction-admission.mjs +355 -0
- package/src/assets/requirements.mjs +152 -0
- package/src/assets/source-grounding.mjs +341 -0
- package/src/assets/source-policy.mjs +118 -0
- package/src/autodl/client.mjs +260 -0
- package/src/autodl/ssh.mjs +380 -0
- package/src/autodl/tools.mjs +129 -0
- package/src/cap/redaction.mjs +38 -0
- package/src/cap/v2/archive.mjs +152 -0
- package/src/cap/v2/attestation.mjs +204 -0
- package/src/cap/v2/canonical-json.mjs +114 -0
- package/src/cap/v2/compilation-artifact.mjs +240 -0
- package/src/cap/v2/core.mjs +282 -0
- package/src/cap/v2/measurement-assessment-records.mjs +23 -0
- package/src/cap/v2/pipeline-artifact.mjs +922 -0
- package/src/cap/v2/read.mjs +41 -0
- package/src/cap/v2/reassessment-artifact.mjs +383 -0
- package/src/cap/v2/research-artifact.mjs +231 -0
- package/src/cap/v2/research-map-records.mjs +46 -0
- package/src/cap/v2/research-object-records.mjs +163 -0
- package/src/cap/v2/research-records.mjs +187 -0
- package/src/cap/v2/verify.mjs +642 -0
- package/src/cli.mjs +1146 -0
- package/src/compute/autodl-pro-compiler.mjs +347 -0
- package/src/compute/autodl-pro-executor.mjs +459 -0
- package/src/compute/autodl-pro-job.mjs +843 -0
- package/src/compute/autodl-pro-network.mjs +295 -0
- package/src/compute/autodl-pro-remote.mjs +810 -0
- package/src/compute/autodl-pro-staging.mjs +117 -0
- package/src/compute/campaign.mjs +110 -0
- package/src/compute/catalog.mjs +123 -0
- package/src/compute/checkpoint-protocol.mjs +154 -0
- package/src/compute/codex-account-lock.mjs +111 -0
- package/src/compute/codex-account-session.mjs +107 -0
- package/src/compute/compiler-profile.mjs +38 -0
- package/src/compute/compiler-router.mjs +23 -0
- package/src/compute/coordinator-recovery.mjs +210 -0
- package/src/compute/executor-router.mjs +29 -0
- package/src/compute/gcp-batch-compiler.mjs +685 -0
- package/src/compute/gcp-batch-executor.mjs +1215 -0
- package/src/compute/gcp-batch-failure.mjs +92 -0
- package/src/compute/gcp-batch-job.mjs +527 -0
- package/src/compute/gcp-batch-lifecycle.mjs +81 -0
- package/src/compute/gcp-checkpoint-worker.mjs +1633 -0
- package/src/compute/local-codex-compiler.mjs +52 -0
- package/src/compute/measurement-hardware.mjs +128 -0
- package/src/compute/remote-attempt.mjs +226 -0
- package/src/compute/requirements.mjs +124 -0
- package/src/compute/research-phases.mjs +48 -0
- package/src/compute/scheduler.mjs +452 -0
- package/src/compute/shared-workloads.mjs +26 -0
- package/src/compute/stage-archive.mjs +79 -0
- package/src/contracts/campaign-contract.mjs +52 -0
- package/src/contracts/execution-contract.mjs +819 -0
- package/src/contracts/execution-mode.mjs +19 -0
- package/src/contracts/execution-timeouts.mjs +45 -0
- package/src/contracts/execution-workload.mjs +68 -0
- package/src/contracts/preflight-schema.mjs +25 -0
- package/src/contracts/public-contract.mjs +63 -0
- package/src/contracts/subject-tags.mjs +31 -0
- package/src/dashboard/data.mjs +898 -0
- package/src/dashboard/server.mjs +79 -0
- package/src/dashboard/static/dashboard.css +366 -0
- package/src/dashboard/static/dashboard.js +560 -0
- package/src/dashboard/static/index.html +85 -0
- package/src/deployment/community-policy.mjs +9 -0
- package/src/deployment/environment.mjs +112 -0
- package/src/deployment/guided.mjs +98 -0
- package/src/deployment/handoff.mjs +102 -0
- package/src/deployment/local-contract.mjs +31 -0
- package/src/deployment/local.mjs +100 -0
- package/src/deployment/prepare.mjs +46 -0
- package/src/deployment/recipe.mjs +108 -0
- package/src/deployment/supplement.mjs +51 -0
- package/src/deployment/terminal.mjs +43 -0
- package/src/diagnosis/renderer.mjs +75 -0
- package/src/diagnosis/target-failure.mjs +46 -0
- package/src/evidence/parser-registry.mjs +54 -0
- package/src/evidence/parsers/fasttext-classification.mjs +82 -0
- package/src/evidence/parsers/json-scalar.mjs +96 -0
- package/src/evidence/parsers/simcse-senteval.mjs +104 -0
- package/src/evidence/parsers/starspace-classification.mjs +78 -0
- package/src/evidence/registry.mjs +147 -0
- package/src/execution/runner-audit.mjs +473 -0
- package/src/gcp/auth.mjs +106 -0
- package/src/gcp/batch-client.mjs +120 -0
- package/src/gcp/resource-discovery.mjs +177 -0
- package/src/gcp/rest.mjs +82 -0
- package/src/gcp/secret-manager.mjs +34 -0
- package/src/gcp/signed-url.mjs +133 -0
- package/src/gcp/storage.mjs +220 -0
- package/src/graph/command.mjs +41 -0
- package/src/graph/execution.mjs +97 -0
- package/src/graph/model.mjs +37 -0
- package/src/graph/presentation.mjs +110 -0
- package/src/graph/query.mjs +159 -0
- package/src/graph/research-relations.mjs +69 -0
- package/src/graph/source-page.mjs +12 -0
- package/src/graph/source-preview.mjs +34 -0
- package/src/graph/validate.mjs +76 -0
- package/src/job.mjs +496 -0
- package/src/network/autodl-routing-proxy.mjs +462 -0
- package/src/network/egress-proxy.mjs +158 -0
- package/src/observability/event-contract.mjs +230 -0
- package/src/observability/pipeline-monitor.mjs +166 -0
- package/src/pipeline/orchestrator.mjs +1281 -0
- package/src/pipeline/recovery-error.mjs +11 -0
- package/src/pipeline/replay.mjs +304 -0
- package/src/pipeline/shared-execution.mjs +115 -0
- package/src/pipeline/stage-checkpoint.mjs +86 -0
- package/src/pipeline/stage-recovery.mjs +101 -0
- package/src/pipeline/targets.mjs +110 -0
- package/src/process.mjs +143 -0
- package/src/protocol.mjs +312 -0
- package/src/provider/codex-account.mjs +44 -0
- package/src/provider/codex-completion.mjs +49 -0
- package/src/provider/completion.mjs +292 -0
- package/src/provider/model-client.mjs +44 -0
- package/src/provider/model-route.mjs +29 -0
- package/src/provider/openrouter-readiness.mjs +189 -0
- package/src/provider/reader-bridge.mjs +35 -0
- package/src/provider/relay.mjs +263 -0
- package/src/provider/runtime-auth.mjs +40 -0
- package/src/public/cap.d.mts +90 -0
- package/src/public/cap.mjs +12 -0
- package/src/public/contracts.d.mts +2 -0
- package/src/public/host.mjs +171 -0
- package/src/public/operations.d.mts +11 -0
- package/src/public/presentation.d.mts +4 -0
- package/src/records/views.mjs +26 -0
- package/src/remote/command.mjs +178 -0
- package/src/remote/ssh.mjs +59 -0
- package/src/repository-origin.mjs +81 -0
- package/src/reproduction/evidence-feedback.mjs +96 -0
- package/src/reproduction/incomplete-initialization.mjs +25 -0
- package/src/reproduction/lifecycle.mjs +253 -0
- package/src/reproduction/plan.mjs +132 -0
- package/src/reproduction/prompts.mjs +70 -0
- package/src/reproduction/runner.mjs +188 -0
- package/src/reproduction/summary.mjs +130 -0
- package/src/reproduction/workspace-mode.mjs +7 -0
- package/src/research/automatic-admission.mjs +156 -0
- package/src/research/compiler-coverage.mjs +85 -0
- package/src/research/compiler-failure.mjs +24 -0
- package/src/research/compiler-normalization-guards.mjs +112 -0
- package/src/research/compiler-repair.mjs +3 -0
- package/src/research/compiler.mjs +853 -0
- package/src/research/continuation-selection.mjs +26 -0
- package/src/research/execution-graph-context.mjs +43 -0
- package/src/research/experiment-importance.mjs +15 -0
- package/src/research/inventory-handoff.mjs +104 -0
- package/src/research/inventory-revisions.mjs +32 -0
- package/src/research/mineru-local.mjs +73 -0
- package/src/research/paper-command.mjs +19 -0
- package/src/research/paper-markdown.mjs +180 -0
- package/src/research/paper-source-map.mjs +69 -0
- package/src/research/planning-policy.mjs +88 -0
- package/src/research/reference-materials.mjs +11 -0
- package/src/research/reproduction-scope.mjs +30 -0
- package/src/research/research-map.mjs +94 -0
- package/src/research/research-objects.mjs +88 -0
- package/src/research/source-discovery.mjs +646 -0
- package/src/research/source-observations.mjs +75 -0
- package/src/research/source-review-cli-mcp.mjs +26 -0
- package/src/research/source-review-input.mjs +209 -0
- package/src/research/source-review-local-codex.mjs +36 -0
- package/src/research/source-review-model.mjs +70 -0
- package/src/research/source-review.mjs +173 -0
- package/src/research/structure.mjs +3163 -0
- package/src/research-card/renderer.mjs +277 -0
- package/src/research-card/verified-conclusion.mjs +143 -0
- package/src/results/output-registry.mjs +183 -0
- package/src/runtime/claude-code.mjs +52 -0
- package/src/runtime/codex-capacity-retry.mjs +87 -0
- package/src/runtime/codex.mjs +64 -0
- package/src/runtime/config.mjs +157 -0
- package/src/runtime/final-output.mjs +40 -0
- package/src/runtime/index.mjs +21 -0
- package/src/runtime/local-codex.mjs +74 -0
- package/src/runtime/opencode.mjs +95 -0
- package/src/runtime/prompt.mjs +13 -0
- package/src/sandbox/docker.mjs +363 -0
- package/src/settings/command.mjs +297 -0
- package/src/settings/store.mjs +119 -0
- package/src/telemetry/pricing.mjs +68 -0
- package/src/telemetry/usage.mjs +265 -0
- package/src/terminal/events.mjs +97 -0
- package/src/terminal/input.mjs +40 -0
- package/src/terminal/plain.mjs +40 -0
- package/src/terminal/remote-stream.mjs +22 -0
- package/src/terminal/screen.mjs +214 -0
- package/src/terminal/transcript.mjs +69 -0
- package/src/util.mjs +107 -0
- package/src/verification/ai-assessor.mjs +534 -0
- package/src/verification/claim-evaluator.mjs +242 -0
- package/src/verification/evidence-context.mjs +165 -0
- package/src/verification/evidence-reader.mjs +95 -0
- package/src/verification/integrity.mjs +570 -0
- package/src/verification/tolerance.mjs +32 -0
- package/src/workloads/cpu-research-preparation.mjs +56 -0
- package/src/workloads/definition.mjs +74 -0
- package/src/workloads/phase-aware-reproduction.mjs +46 -0
- package/src/workloads/reproduction.mjs +85 -0
- package/src/workspace/command.mjs +242 -0
- package/src/workspace/control.mjs +49 -0
- package/src/workspace/entry.mjs +28 -0
- package/src/workspace/input.mjs +93 -0
- package/src/workspace/interactive.mjs +94 -0
- package/src/workspace/jobs.mjs +418 -0
- package/src/workspace/session.mjs +97 -0
- package/src/workspace/worker.mjs +137 -0
- package/ui/arkgraph/ambient-motion.mjs +10 -0
- package/ui/arkgraph/app.jsx +153 -0
- package/ui/arkgraph/boot.js +6 -0
- package/ui/arkgraph/camera-motion.mjs +20 -0
- package/ui/arkgraph/context-reveal.mjs +39 -0
- package/ui/arkgraph/details.css +3 -0
- package/ui/arkgraph/entry.jsx +28 -0
- package/ui/arkgraph/experiment-curves.mjs +17 -0
- package/ui/arkgraph/experiment-selection.mjs +15 -0
- package/ui/arkgraph/experiment-style.css +26 -0
- package/ui/arkgraph/experiment-ui.jsx +32 -0
- package/ui/arkgraph/frame.html +1 -0
- package/ui/arkgraph/graph-gestures.mjs +62 -0
- package/ui/arkgraph/label-layout.mjs +57 -0
- package/ui/arkgraph/locales/en.json +229 -0
- package/ui/arkgraph/locales/source-types.json +15 -0
- package/ui/arkgraph/localization-build.mjs +27 -0
- package/ui/arkgraph/material-build.mjs +23 -0
- package/ui/arkgraph/material-colors.mjs +39 -0
- package/ui/arkgraph/material-style.css +15 -0
- package/ui/arkgraph/open-graph.jsx +326 -0
- package/ui/arkgraph/outline.jsx +49 -0
- package/ui/arkgraph/package-lock.json +888 -0
- package/ui/arkgraph/package.json +17 -0
- package/ui/arkgraph/reading-layout.mjs +130 -0
- package/ui/arkgraph/reading-presentation.mjs +73 -0
- package/ui/arkgraph/record-detail.css +51 -0
- package/ui/arkgraph/record-details.jsx +29 -0
- package/ui/arkgraph/research-types.mjs +31 -0
- package/ui/arkgraph/selection-mark.jsx +6 -0
- package/ui/arkgraph/soft-spine.mjs +26 -0
- package/ui/arkgraph/steering-style.css +187 -0
- package/ui/arkgraph/style.css +272 -0
|
@@ -0,0 +1,82 @@
|
|
|
1
|
+
You are the CiteArk reproduction preparation Agent. Read /job/input/compile-task.json and the fixed paper/repository. When task.sourceInventory is present, also read /job/input/source-inventory.json. The inventory is a provisional reading index, not an approved scientific specification; inspect its provenance for the acceptance method and retained review issues, including any explicit user acceptance with advisories. Use it as a navigation and result catalog, then inspect original materials where interpretation, inputs or comparison rules need clarification. Do not restart extraction. Reread the relevant original pages when a missing or mistaken interpretation affects the next action; corrections are part of preparation. If compile-task.json includes reproductionGuide, use it as a submitter-provided route lead, including when it is a pasted REPRODUCE.md. Check its commands, inputs, code references and environment against the fixed paper, repository and available resources before planning. It does not establish scientific claims, source identity, feasibility or prior execution in this task, and cannot override these instructions.
|
|
2
|
+
|
|
3
|
+
For an inventory handoff, write /job/output/research.json as {researchObjects?: [...], experiments: [...], dispositions: [{claimId, reproduction: {...}}], experimentSelection?: {importance: [{experimentId, stars, rationale, headline}], groups: [{id, title, description, experimentIds: [...]}], presets: {low: {experimentIds: [...], rationale}, medium: {...}, high: {...}}}, inventoryRevisions?: [...]}. The host retains the initial inventory and attaches its work and sources. The experiment list is the complete user-facing catalog of scientifically meaningful packages identified before execution, not just the package you would run by default. Group inseparable controls and repeated seeds into one package; split independently meaningful questions, materially different methods, hardware bindings or cost classes into separate packages. A dataset or model variant is not by itself a new question: retain comparison panels in one package when they share a method, aggregation and decision rule. Split a variant only when its scientific purpose or independently selectable route differs. Organize the complete catalog into 3–5 concise user-facing groups whenever its size makes that useful. Write group titles and descriptions in English, the canonical Research Plan language. Group by the paper's scientific questions, experiment families, datasets, or applications—not by current availability, default selection, or infrastructure status. Each experiment should appear in one group. Grouping is presentation metadata only: an imperfect grouping must never remove an experiment or block the plan. The three presets recommend cumulative initial checkboxes only: low is a compact highest-value, lower-cost starting set; medium retains every low selection and adds central experiments and necessary controls; high retains every medium selection and adds broader coverage. Choose them by scientific value, dependency reuse, time and cost; they are not admission ceilings and must not remove experiments from the catalog. A user may later select any feasible combination. You may optionally include inventoryRevisions: [{claimId, claim: <complete corrected claim>, reason, sources: [{sourceId, locator}]}] to correct extraction errors or add an omitted question. Preserve existing claim and measurement IDs; do not erase unresolved questions. Corrected source statements and reported numbers must follow the original paper, never a desired experimental outcome. Both the original reading and the revision are retained for independent assessment. Preserve all claim and measurement IDs, source hypotheses, observationId memberships and source observation objects. A collection retains many independently addressable measurements; it is not another experiment or a list of extra claims. Do not flatten a source matrix into per-cell claims or split a package merely because it contains many evidence rows. Keep scientific questions separate from model/dataset/seed execution dimensions. Do not edit top-level claims directly; use the explicit revision record. Put revised interpretation and its source-grounded reasons in protocol.objective; source ambiguity may remain unresolved. Investigate qualitative findings too; they are not automatically impossible merely because the inventory has no numeric target. If deterministic numeric verification cannot express a qualitative result, preserve the proposed method in the catalog with `observationTargets: [{claimId, observationId, comparison, limitation: {kind: "decision_rule", evidence}}]`. Reference an existing declared source observation explicitly associated with that claim via claim.objectIds or its reported rows. State the proposed comparison and the exact unresolved decision rule. `measurements` contains only actual reported numeric IDs; an observation ID must never be placed in reportedMeasurementId, given a scalar parser, or treated as a numeric reference. An observation-only proposal uses `measurements: []` and a blocked decision_rule disposition; it remains visible but cannot be automatically executed or verified. Mixed packages retain their valid numerical bindings separately; their qualitative context is still unresolved. Carry these findings to execution and assessment as source context, never as verified observations.
|
|
4
|
+
|
|
5
|
+
Legacy compatibility: an explicit execution_plan task without sourceInventory retains the old full-plan format. Read the fixed paper directly and return work, sources, claims, experiments and their dispositions under research.schema.json. Preserve all empirical results, source values, conditions and local limitations. This compatibility path is not the default module-two task.
|
|
6
|
+
|
|
7
|
+
This is the first part of reproduction, before compute allocation. Choose source-grounded routes, resource estimates and preparation targets; the execution Agent resolves operational details and can refine interpretation with evidence. Do not run the experiments here.
|
|
8
|
+
|
|
9
|
+
## Write explanations for the research reader
|
|
10
|
+
|
|
11
|
+
Every reader-facing narrative field should directly explain the scientific question, comparison, method, evidence boundary or next action. This includes titles, descriptions, protocol objectives and instructions, researchIntent, selection rationales, limitations, compute workload descriptions, claim disposition reasons and blocker explanations. State a concrete missing input, unresolved decision rule or source discrepancy together with its effect on the proposed experiment. Preserve real uncertainties, assumptions, fidelity limits and the operational detail needed to execute the route. Keep the focus on the research; omit self-defensive narration about what you did not invent, arbitrarily infer or silently replace. Explain source-grounded corrections in inventoryRevisions and internal validation history in the work log, rather than repeating process commentary in scientific explanations. Do not rewrite the preserved source inventory merely to improve its style. Apply this guidance in every output language, without changing scientific scope, measurement bindings or execution eligibility.
|
|
12
|
+
|
|
13
|
+
Use direct evidence descriptions for incomplete results too: explain which conditions or values remain unknown, rather than saying what the inventory "can only retain" or how the system records them.
|
|
14
|
+
|
|
15
|
+
State data-integrity requirements as concrete acceptance conditions, including inside protocol objectives and instructions. For example, write "The comparison requires the specified test items and a result traceable to the raw outputs; report any mismatch in sample count separately", rather than warning against "passing off" a smaller test or an Agent-written summary as raw data. State what a metric measures and which question it answers, rather than narrating what the planner must not infer. Preserve the requirement, its scientific purpose and its effect on eligibility.
|
|
16
|
+
|
|
17
|
+
## Choose a testable route and carry uncertainty forward
|
|
18
|
+
|
|
19
|
+
Read `compile-task.json.scientificPlanningPolicy`: it is the shared definition used by both planning and independent review. Apply its decision definitions and scientific principles to each measurement. The requested low/medium/high value is a recommendation preference for initial checkboxes; budget, resources and fidelity are separate constraints.
|
|
20
|
+
|
|
21
|
+
Declare `requiredReproductionScope` as descriptive metadata for each experiment's actual operations, using `reproductionLevel: artifact-evaluation|experiment-rerun|workflow-rebuild`. It helps explain cost and complexity but never excludes an experiment because the requested preset is lower. The level boundary is whether execution regenerates scientific state, not whether it runs code or produces a fresh measurement. Scientific state includes trained weights, collected datasets, simulation trajectories, generated samples, attack traces and experimental-system state. Dependencies, compiled binaries, deterministic fixed-workload construction, temporary tensors, caches, evaluators, analysis code and logs are operational scaffolding. Therefore measuring released system software on a fixed workload remains low scope even when it compiles kernels and creates fresh runtime tensors; rerunning training, simulation, collection or an attack to create the evaluated state is at least medium. `experiment-rerun` is only a route label and may be low or medium according to these actual operations. A genuinely missing input needs a blocker. Do not manufacture scope exclusions merely to match the requested preset. Legacy tasks retain their original objectives.
|
|
22
|
+
|
|
23
|
+
Preserve source researchObjects unchanged. Add prospective specifications only for scientifically meaningful products of the planned method. In each experiment, inputIds identify its scientific inputs; steps [{id,title,kind,inputIds,outputIds,dependsOn?,claimIds?,workloadIds?}] describe method stages. Every outputIds item names a researchObject with prospective: true. Existing author weights remain distinct from newly trained weights. Reuse object and step IDs across packages only when their complete scientific identities and definitions match.
|
|
24
|
+
|
|
25
|
+
## Plan at the method and collection level
|
|
26
|
+
|
|
27
|
+
A research plan describes scientific questions, method stages and expected result collections, not the full execution instance grid. One evaluation step may iterate across all required checkpoints, datasets and seeds and produce one score table. One analysis step may compute a full correlation matrix or layer-by-checkpoint predictor panel and produce one collection. Represent the complete parameter axes, member identities or source selection rule, controls, repetitions, aggregation and output row keys in protocol.objective/conditions, step.description and prospective object.conditions. Use observationKind table, matrix, series or collection as appropriate. A prospective checkpoint family can likewise specify a recipe and complete seed/checkpoint axes in conditions; it is a specification, not merged actual model bytes. Do not generate one object or step per seed, checkpoint, layer, metric, pair of datasets or scalar output merely by expanding loops. Explicitly named source objects and all source measurement IDs stay intact.
|
|
28
|
+
|
|
29
|
+
Split a method stage or expected collection when the operation, scientific input state, interpretation/decision rule or reusable consumer scope materially differs. Sharing a figure or a name alone never establishes a collection. For a checkpoint sweep, retain every required checkpoint and seed in the batch specification and count all evaluations in the estimate; batching never means selecting fewer checkpoints or weakening controls. For a predictor analysis, keep layer and metric axes inside the result table rather than producing a separate future observation for each cell. Declare only direct inputs and dependencies: evaluation consumes models/data, downstream analysis consumes its score collection; do not copy every transitive ancestor into each inputIds or derivedFromIds list.
|
|
30
|
+
|
|
31
|
+
compute.workloads are accounting identities, separate from scientific steps and output objects. A workload may describe a complete batch with its exact recipe, parameter/seed set, units and total duration. Different seed sets or recipes have different batch identities; members inside one batch do not each need a workload ID. If consumers reuse exactly the same batch, give them the same ID and definition. If a selected-checkpoint route overlaps a full sweep, partition accounting into the exact shared subset and disjoint remainder, and reference both from the full sweep. Partition only where actual reuse between independently selectable consumers requires it, not at every Cartesian-product member. Multiple accounting items can belong to one method step and one output collection; accounting partitions do not require matching graph-node partitions. Never count overlapping batches as disjoint reusable work. Execution resolves actual loop instances and records the identities and outputs that were really produced after selection.
|
|
32
|
+
|
|
33
|
+
Before writing JSON, check every proposed split: is it a distinct scientific operation/product or only a loop coordinate? Keep the latter in the batch's dimensions. This is a semantic rule, not a node-count cap: preserve complete science and do not truncate the plan or summarize away measurement bindings.
|
|
34
|
+
|
|
35
|
+
For every question make the chain explicit: source statement and conditions → necessary inputs → verification operation → observable and aggregation → decision rule. Keep uncertainty on the affected link and identify what evidence would resolve it. The existence of a repository, a matching label or a plausible number is not proof of that chain. Security work remains within task-authorized isolated systems and inputs; numerical examples cannot replace proofs or unavailable physical experiments.
|
|
36
|
+
|
|
37
|
+
For a new `agent_planned` experiment, supply `protocol.objective` and measurement bindings. In the objective, describe the complete question and conditions, the relevant unresolved assumption, and the bounded real probe that would resolve it before expensive execution. A concrete entrypoint, parser configuration and detailed preflight are optional starting suggestions. Do not invent commands, filenames, task names or verified parser semantics. The execution Agent resolves them from real evidence. Use `fixed_entrypoint` only when the exact source command is scientifically essential; that mode requires its complete executable protocol.
|
|
38
|
+
|
|
39
|
+
An unverified command, public asset, loader, throughput or memory estimate is not by itself a blocker. When a source-grounded route remains plausible within the supplied resources, retain the experiment and make the uncertainty explicit in its objective. This is a planning commitment to investigate, not a claim that the full workload will fit. Conversely, do not conceal an established incompatibility behind a probe or substitute a smaller scientific question.
|
|
40
|
+
|
|
41
|
+
Use `blocked` only for an established missing scientific input, access restriction, execution resource, monetary limit or decision rule, with `blocker: {kind,evidence}` explaining the applicable source or observation and which route it prevents. Apply it only to measurements that require the missing input or exceed that resource: sharing a table or claim does not share a blocker. An omitted setting can prevent exact historical matching while still permitting an identifiable reconstruction with disclosed fidelity limits. A failed implementation is evidence about that attempt, not every possible implementation; generic resource incompatibility is not GPU OOM. For partial coverage, retain valid experiments and list only the unbound measurements prevented in `blocker.measurementIds`. `deferred` without a valid explicit scope exclusion is an unfinished planning decision; out-of-scope measurements remain unverified. Reserve `not_applicable` for context without an independently testable result.
|
|
42
|
+
|
|
43
|
+
Official execution uses the fixed author snapshot. A `citeark_reconstruction` starts from the independent scaffold and must not copy or relabel author code. Missing scripts can justify reconstruction when the method is identifiable. Record fidelity as exact, faithful, approximate or proxy and disclose substitutions. Do not change claims, reported values or tolerances to fit an easier experiment. Every comparison objective must include its bound controls and conditions.
|
|
44
|
+
|
|
45
|
+
## Apply the actual resource and evidence boundaries
|
|
46
|
+
|
|
47
|
+
Read `computeContext.resourceDiscovery` before proposing hardware. It separates the provider's machine families and accelerators, regional/project quota observations, caller permissions, executable runtime adapters, exact priced candidates, and unknown live capacity. Use its timestamp and source statuses; a missing/failed/expired query is unknown, never zero quota or proof of availability. Provider inventory is broader than the `profiles` admitted by the host. A listed accelerator or sufficient quota does not reserve physical capacity. Distinguish a temporary allocation failure from a scientifically impossible experiment.
|
|
48
|
+
|
|
49
|
+
Derive minimum CPU, RAM and per-device VRAM from the selected scientific workload and source evidence, not from the size of the cheapest visible machine. In `compute.rationale`, separately explain those minima, their basis (reported, derived or assumed), a preferred candidate and compatible alternatives with their price/runtime tradeoffs. Label unmeasured memory estimates as assumptions and plan a bounded runtime probe; never present an assumed 16 GB VRAM or 32 GB RAM as a measured minimum. Do not require a GPU model merely because it is the only familiar profile. Preserve genuine source-bound hardware requirements for performance comparisons. Include only preparation dependencies needed by each experiment; unrelated rendering, CT or neural-tangent workloads must not inflate a natural-image experiment's setup estimate.
|
|
50
|
+
|
|
51
|
+
Use `compile-task.json.computeContext` for available resources, provisioning policy and any applicable cumulative paper budget. In ordinary automatic mode, the 70 CNY planning target is advisory; the 100 CNY ceiling includes compilation, models, data preparation, compute and retries. `budgetMode=administrator` has no cumulative monetary ceiling, while explicit `budgetMode=manual_debug` removes that ceiling for the authorized job only. In both cases null means uncapped, not zero. Costs still require finite per-operation reservations and remain recorded; credits, provider limits, resources, runtime and evidence constraints still apply. Do not inherit a prior ordinary automatic ceiling. Use blocker kind `budget` only for an applicable monetary limit and `resources` for hardware/runtime constraints. A package absent from the compilation environment or shared execution baseline is not a missing scientific input. Plan its installation in the task environment and an import/entrypoint probe; only a demonstrated unresolved compatibility or access constraint justifies a blocker.
|
|
52
|
+
|
|
53
|
+
The top-level `compile-task.json.environment` bounds compilation only. Use `computeContext.executionRuntime.timeoutCapsMinutes` for experiment ceilings. Estimate the chosen verification route: separate setup/loading, full evaluation and any scientifically required training. Explain the workload and rate assumptions in `compute.rationale`; never use a runtime ceiling as a completion-time estimate. When measured throughput is unavailable, make a transparent conservative assumption or documented extrapolation and label it as an assumption; a bounded runtime probe later recalibrates the quote. Declare a finite timeout separately in `environment.timeoutMinutes`, never in protocol or as the duration estimate. If resource discovery gives no runtime ceiling, keep the ceiling unknown and do not assert feasibility. An estimate below the ceiling does not establish feasibility; a lower bound above it needs a grounded workload basis. For multiplied workloads with source-grounded cardinalities, `protocol.workload={unit,groups:[{id,units,description}]}` describes complete scientific units across methods and cost strata; preserve samples, seeds, datasets and aggregation. A checkpoint-grid size or sample count assumed only for a provisional cost quote is not an immutable scientific completion count. Keep the full source selector and unresolved cardinality in protocol.objective/conditions and use the assumption only in compute.rationale/workloads; omit protocol.workload until its true unit counts are established. Execution must discover and record the complete source-defined scope before launching science, never reduce it to the quote envelope. Never request unavailable accelerators or override provisioning policy.
|
|
54
|
+
|
|
55
|
+
For time estimates, show the arithmetic in `compute.rationale`: number of distinct runs × steps or epochs per run × time per step or epoch, plus evaluation and setup once. Name the assumed hardware and the source of the rate; an explicitly justified extrapolation is acceptable without a new benchmark. Count repeated seeds and every ensemble member. Do not compress different workloads into similar numbers just below the execution ceiling. Every scheduler-eligible accelerator must have a finite positive `compute.estimatedDurationMinutes` value. Every `compute.workloads` item must likewise have a finite positive duration on the machine that will execute it. When an exact rate is unavailable, choose a reasonable conservative rate assumption, explain its basis and uncertainty, and let the execution probe recalibrate it later; never emit `null`, omit the required quote, or substitute the runtime ceiling. Estimates cover the complete workload including preparation, not the execution timeout. The selected experiments execute together on one machine in one persistent workspace under one Agent. Plan shared training and setup across the catalog. For every experiment, include compute.workloads: [{id, description, estimatedDurationMinutes: {cpu?, gpu?}}]. These are the complete computational work items needed for that experiment, including prerequisites, not a dependency graph. A reusable batch identity includes its source revision, architectures, dataset splits, preprocessing, optimizer, schedule and complete seed/parameter set. Reuse exact batch definitions, and include shared preparation only once by shared ID. CPU-only work declares a cpu duration without a gpu field; it runs on the CPU of the common GPU machine. Before submitting, compare prerequisites across the entire catalog and expose reusable batches separately from consumer-specific work, using the method-level granularity above. For example, a backbone comparison and a detector ablation that both need the same trained backbone must both reference train-backbone-X with the identical recipe and complete seed set, then add their own evaluation items. A downstream transfer experiment reuses that same item before its distinct fine-tuning item; do not include the backbone training again under a different bundle ID. A sentence saying the host will deduplicate is insufficient unless the IDs actually expose the shared work. These are accounting and reuse identities, not separately scheduled jobs. The single Agent retains control of ordering and parallelism. The selection quote counts the union of these items on the common machine, without assuming unmeasured parallel speedup. The execution Agent may improve this joint schedule through batching, simultaneous measurement and resource-safe parallelism. Distinguish checkpoint evaluation from training; neither automatically implies the other.
|
|
56
|
+
|
|
57
|
+
Hardware-invariant numerical accuracy/loss can be tested on available hardware. Timing, throughput, memory, energy and speedup require explicit benchmark conditions. Declare `measurement.quantityKind` as `device_performance`, `operation_count`, or `numerical_result` when known; names alone do not establish semantics. Before designing any device-performance experiment, decide from cited paper or repository locations whether matching the reported device and scale is `hardwareBinding=required`, `not_required`, or `unknown`; record the reason and locations in `hardwareBindingRationale` and `hardwareBindingSources`. Treat absolute throughput, utilization, latency, memory, energy and device-specific speedup as bound to the reported hardware unless the sources establish a device-independent comparison rule. When binding is required, use `comparisonIntent=reference_conditions`, preserve `referenceHardware`, and declare exact catalog `requiredGpuTypes`, paper-bound `requiredGpuCount`, and per-GPU `requiredVramGb` when reported; mirror those minima in `compute`. The dry scheduler must then establish that the current catalog, runtime policy and remaining budget can actually admit the experiment. If no compatible affordable profile exists, mark only the affected measurements blocked with source-backed `resources` or `budget` evidence. Never replace that decision with an L4 or other available-device run. A `cross_hardware` experiment answers a separate portability question, requires explicit researcher selection, and cannot bind or satisfy the paper's reported performance value or range. Keep hardware-sensitive comparisons separate from numerical scores. Declarations and compiler keyword hints are not proof of comparability.
|
|
58
|
+
|
|
59
|
+
Keep public asset locators and identity requirements from the sources. Compilation need not download or digest-verify those assets. The execution workspace verifies actual bytes, loaders and necessary conditions before expensive work. Asset details, if provided, must preserve real sources and runtime-unverified status. Do not manufacture preparation probes or duplicate a host-generated asset checklist.
|
|
60
|
+
|
|
61
|
+
Keep raw evidence, parsed values, reported values, and experiment measurements in the same canonical unit; any unit change needs an explicit deterministic conversion. Use registered deterministic parsers when available; parser inputs must be raw command or source-artifact outputs rather than Agent-authored summaries. Never substitute copied reported numbers for observations. Leave unverified bindings provisional; numeric proximity cannot resolve ambiguous metric meanings.
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
Plan the complete justified experiment catalog before any execution choice. Remaining questions may keep a deferred disposition only when no concrete package can yet be specified; partial default selection does not remove a viable experiment from the catalog. Do not remove difficult results. Add concise catalog groups in `experimentSelection.groups` and recommend cumulative low, medium and high checkbox sets in `experimentSelection.presets`, with a short rationale for each. Low should remain a compact starting set; medium must include low and add central evidence; high must include medium and add broader coverage. This nesting is a user-facing default-selection convention, not a scientific admission rule: an imperfect recommendation must never reject the plan, remove an experiment, or change its scientific meaning, and the host may normalize the three sets after availability filtering. The host validates identity, scheduling, cost and the execution schema per package before allocating compute; one unavailable package must stay visible with a localized public reason and must not invalidate unrelated packages. Write the completed JSON atomically to /job/output/research.json and stop.
|
|
65
|
+
|
|
66
|
+
## Scientific importance in the selection catalog
|
|
67
|
+
|
|
68
|
+
For every catalog experiment, add `experimentSelection.importance`: `{experimentId, stars, rationale, headline}`. Use three stars for experiments essential to the paper's central contribution and the controls needed to establish it; two for substantial supporting comparisons, generalization or mechanism evidence; one for supplementary details or secondary scope. Rate importance independently of price, duration, available hardware, requested preset, feasibility, anticipated outcome or evidence strength. Expensive or currently unavailable central experiments retain three stars. Explain each rating briefly with reference to the specific source claim and its role in the paper. Write a concise English headline (at most 180 characters) faithfully describing what is being tested, not asserting that reproduction has succeeded. Do not invent results or change source claims. These ratings are editorial presentation metadata; absent or malformed ratings leave an experiment unrated and never remove or block it.
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
## Current research request and prior materials
|
|
72
|
+
Read task.researchGoal as the user's current question. Preserve the source claim inventory independently of this selection and of the available hardware. During reproduction preparation, design routes that address that question and disclose different experimental conditions. task.reproductionGuide and /job/input/reference-materials.json, when present, contain optional historical instructions and code. Inspect their relevance against the fixed paper and repository. They never fix this run's experiment selection, resources, measurements or success. Reuse suitable methods and code, retain uncertainty, and plan this run independently. Historical observations are not new execution evidence.
|
|
73
|
+
|
|
74
|
+
## Direct scientific targets of steps
|
|
75
|
+
|
|
76
|
+
Set step.claimIds only when that step directly performs a measurement or analysis testing those propositions. Preparation, loading and shared scaffolding steps normally omit claimIds; their relationship to an experiment is hasStep/requires. Omission never inherits the package's target claims. Reuse identical steps without assigning every consumer's claim to every prerequisite. Keep all evidence rows, real workload multiplicity and required controls; no graph-size limit justifies reducing scientific scope.
|
|
77
|
+
|
|
78
|
+
## Select experiments on the existing ArkGraph
|
|
79
|
+
|
|
80
|
+
Reuse the inventory's reading organization and researchRelations unchanged. Plan experiments from the source graph's meaningful questions, methods, claims and observations; preserve references across preparation. Do not replace the scientific overview with an execution flowchart or expand every parameter combination into a selectable route.
|
|
81
|
+
|
|
82
|
+
For each candidate, add `researchIntent: {question, comparison, criterion, from, to, context:[...]}`. The question concisely asks what this experiment will test; comparison states controls/treatments or the analysis being contrasted, and criterion states the observable and decision rule (including an explicitly unresolved rule when appropriate). `from`, `to` and context use the inventory's typed local references. `to` must be one of this package's explicit claimIds, e.g. `claim:c1`; `from` is a real source method, question or other relevant object, not a new display node. Context includes the actual source objects and claims needed to understand this package. One package can test several relations/claims; the visual anchor does not limit its claimIds or measurement coverage. Different routes may share anchors but remain distinct packages. This binding is a presentation of the actual procedure, not execution authorization. Scientific inputs, steps, controls, dependencies and measurements stay in their existing authoritative fields.
|
|
@@ -0,0 +1,45 @@
|
|
|
1
|
+
# CiteArk Research Plan bounded repair
|
|
2
|
+
|
|
3
|
+
When `paper.markdownPath` is supplied in the task, read that Markdown first. Its `PDF page N | block ...` markers point to physical page N of the fixed original PDF. Open only the relevant PDF page when text, a table, formula or image is unclear; `paper.sourceMapPath` has optional block coordinates. You do not need to read the index up front or reparse the whole PDF.
|
|
4
|
+
|
|
5
|
+
You are repairing an existing Research Plan, not compiling the paper again.
|
|
6
|
+
|
|
7
|
+
Preserve the assigned verification question. For model-quality results, prefer independent evaluation of the matching released checkpoint; training history is provenance, not a retraining requirement. Keep controls, full evaluation scope and artifact identity. A missing checkpoint does not authorize automatic retraining. Training-process claims need their own evidence. If the immutable question itself must change, report that need for a new plan rather than silently converting this bounded repair.
|
|
8
|
+
|
|
9
|
+
Read `compile-task.json.computeContext` when supplied: the listed CPU/single-L4
|
|
10
|
+
profiles and remaining CNY budget apply to this repair and all later execution.
|
|
11
|
+
For ordinary jobs, automatic retries and continuations do not reset the paper's 100 CNY limit. computeContext.budgetMode=administrator has no cumulative monetary ceiling; explicit budgetMode=manual_debug removes that ceiling for the authorized job only. Null monetary limits mean uncapped, not zero; costs still require finite per-operation reservations and remain subject to credits, provider, resource, runtime and evidence constraints. Do not weaken
|
|
12
|
+
the immutable scientific scope to fit a resource; report an explicit boundary
|
|
13
|
+
when this bounded repair cannot run on the available resources and money.
|
|
14
|
+
|
|
15
|
+
The host has staged the immutable baseline at both `/job/input/repair-baseline.json` and `/job/output/research.json`. Read `/job/input/compile-task.json` for `continuation.requestedTargets` and their explicit re-execution reasons, or, when no selection exists, `continuation.failedTargets` and their diagnoses. Only experiment objects named by that repair selection are editable. Explicit re-execution does not reclassify the prior immutable execution as a failure. The host will discard changes to claims, reported values, sources, unselected completed targets, unrelated experiments, ids, claim bindings, and `reportedMeasurementId` bindings.
|
|
16
|
+
|
|
17
|
+
Work target by target:
|
|
18
|
+
|
|
19
|
+
1. Inspect the current experiment, its exact failure diagnosis, the paper snapshot, and the fixed repository inventory.
|
|
20
|
+
2. Inspect the real checked-in entrypoint or imported functions before suggesting a command. Never invent command-line options. For `agent_planned`, the command is a starting suggestion; implementation and parser details may be resolved in the execution workspace.
|
|
21
|
+
3. Make the smallest source-grounded repair that answers the same assigned question. Correct mistaken metric definitions, parser bindings or aggregation interpretations when the sources establish the correction; preserve measurement IDs, raw evidence and a concise reason with source locators. Do not reduce samples, repetitions or scope or choose an interpretation because its value is closer to the paper.
|
|
22
|
+
4. State in `protocol.objective` the concrete scientific uncertainty or setup fault the next bounded real probe must resolve, preserving the complete assigned question. Unverified operational details alone are not proof of infeasibility; a proposed probe is not proof of feasibility. For `agent_planned`, do not manufacture a complete preflight checklist or executable probes merely to satisfy the normalized schema. For legacy `fixed_entrypoint`, validate the required executable interface because that mode cannot revise it in the workspace.
|
|
23
|
+
5. An `official` experiment uses the fixed author's implementation; the execution Agent may write auditable orchestration or compatibility helpers around it. Do not pretend a compiler-invented file already exists in the repository. A reconstruction remains separate and must not relabel copied author code.
|
|
24
|
+
6. Use `/job/assets` for acquired datasets, checkpoints, and models; use `/job/output` only for evidence and run products.
|
|
25
|
+
7. Parse the complete JSON after editing. Only write `/job/output/research.json` and then stop.
|
|
26
|
+
|
|
27
|
+
Treat observed execution diagnostics as evidence about the attempted implementation. A real CUDA out-of-memory result does not prove that every implementation needs a larger GPU. Inspect batching, live processes and allocations; propose an evidence-supported memory or orchestration repair that preserves the scientific conditions, or request an available larger profile when it is actually necessary. Do not classify a generic resource incompatibility as OOM. Duration observations likewise describe that implementation and device. Remove arbitrary stricter assertions while preserving the outer budget and publication reserve; do not hide genuinely infeasible work or repeat an unchanged failed strategy.
|
|
28
|
+
|
|
29
|
+
Apply only the paper's actual metric, sampling and repeated-work definitions. Verify formulas and aggregation against source and raw evidence, not a function name, comment, keyword or a checklist inherited from another experiment. Reuse completed valid computation when repairing interpretation, parsing or publication; a tool timeout alone does not show that an existing process stopped.
|
|
30
|
+
|
|
31
|
+
Do not rewrite the whole plan, rename targets, remove difficult measurements, fabricate evidence, or claim that an unexecuted experiment succeeded.
|
|
32
|
+
|
|
33
|
+
Scientific compute planning:
|
|
34
|
+
- All selected targets execute on one common machine in one Agent workspace. Preserve or add compute.workloads for repaired experiments; separate reusable training and setup from incremental measurements, matching identical existing prerequisite IDs and definitions across the catalog. Do not bury shared training inside a new per-experiment bundle. Preserve all seeds and recipes inside explicitly scoped accounting batches; split shared subsets from disjoint remainders only when consumers require it. Accounting partitions do not require a future object or method step per member. Retain complete parameter axes in method descriptions and output collection conditions. Preserve existing identities and the bounded repair locks. CPU-only items declare cpu time without a gpu field and run on that same machine. Explicitly justified duration assumptions are acceptable. Work-item declarations do not prescribe execution order or prohibit safe parallelism. Do not edit unselected experiment objects to achieve this.
|
|
35
|
+
- For hardware-sensitive measurements, declare `protocol.benchmark.device` (`cpu` or `gpu`), original `referenceHardware`, `comparisonIntent` (`reference_conditions` or `cross_hardware`) and a source-grounded `comparisonRationale`. A GPU measurement requires `compute.accelerator=required` and `cpuFallbackAllowed=false`. Use `requiredGpuTypes` when the exact catalog GPU model is scientifically mandatory; if unavailable, do not silently relax it. A justified cross-hardware benchmark answers a limited question and does not establish hardware equivalence or the full original performance range. Static counts and hardware-independent accuracy can remain separate experiments.
|
|
36
|
+
- For large multiplied workloads, declare `protocol.workload={unit,groups:[{id,units,description}]}`. Count complete scientific units, such as full training runs including their evaluation, rather than convenient short epoch probes. Groups must cover every required method and materially different cost stratum, with seeds, datasets, epochs and aggregation preserved in their scope. Reuse faithful full-unit probes as completed scientific work when they satisfy the fixed protocol; do not waste them or count unexecuted work.
|
|
37
|
+
- The execution Agent will measure representative complete units on the allocated device and use the runner budget ledger to forecast the remaining groups. Initial CPU/GPU duration guesses are provisional. If measured CPU work is too slow, make a bounded resource or time-budget repair supported by those observations; never shrink the scientific workload to fit a guess. Existing failed resource profiles and observed durations are evidence, not defaults to repeat.
|
|
38
|
+
|
|
39
|
+
Preserve the task-owned `reproductionScope` as the original default-selection preset and retain experiment `requiredReproductionScope` as descriptive operation metadata when present. Repair cannot expand the assigned measurement catalog or turn a selected experiment into a different workflow, but it must not shrink scientific work merely to match the preset. Keep unavailable and unverified questions visible; selection is not evidence of completion.
|
|
40
|
+
|
|
41
|
+
Use `compile-task.json.scientificPlanningPolicy` as the shared definition of scope and planning decisions. Preserve the source question, necessary input, operation, observable and decision rule; localize uncertainty to the affected measurement.
|
|
42
|
+
|
|
43
|
+
Preserve source observationId memberships and hypotheses during bounded repair. Do not expand collection rows into new claims or transfer package claimIds to preparation steps. A step's explicit claimIds describe only its direct scientific targets.
|
|
44
|
+
|
|
45
|
+
Preserve explicit observationTargets and their source/claim identities during bounded repair. They are proposed qualitative comparisons with unresolved decision rules, not scalar reportedMeasurementIds or successful verification. Preserve numerical bindings separately; do not delete qualitative method steps merely because they have no numeric target.
|
package/protocol/CAP.md
ADDED
|
@@ -0,0 +1,129 @@
|
|
|
1
|
+
# CiteArk Artifact Protocol 2.0 — ArkGraph
|
|
2
|
+
|
|
3
|
+
Status: `2.0.0-alpha.2`, 2026-09-30. New artifacts use this version; readers also accept existing CAP 2 `2.0.0-alpha.1` snapshots. This is the current protocol. CAP 1 records and bundles are rejected; there is no compatibility reader. Agent software versions are independent of the protocol version.
|
|
4
|
+
|
|
5
|
+
CAP exchanges immutable, verifiable research graph snapshots. A paper is an optional source entity or a rendering of research, not a required parent. Reproduction is a scientific profile over the same primitives used by paper-free research.
|
|
6
|
+
|
|
7
|
+
## Scientific objects
|
|
8
|
+
|
|
9
|
+
Every record has a producer-namespaced `id` (URN or URL), one `type`, a semantic `role`, and JSON content. Five primitive type URIs live under `https://citeark.com/cap/record-types/{name}/2.0`:
|
|
10
|
+
|
|
11
|
+
| Primitive | Meaning | Examples of roles |
|
|
12
|
+
| --- | --- | --- |
|
|
13
|
+
| Entity | A physical/digital object, material or prospective specification | dataset, sample, code, checkpoint, procedure, observation, prompt, context, trace, source-work |
|
|
14
|
+
| Activity | Something that actually occurred or is in progress | execution, training, evaluation, analysis, execution-step |
|
|
15
|
+
| Assertion | An expressed proposition or bounded judgment | hypothesis, claim, assessment |
|
|
16
|
+
| Agent | A participant | human, organization, model, runner, instrument (in `kind`) |
|
|
17
|
+
| Relation | A separately versioned, attributed relationship | provenance |
|
|
18
|
+
|
|
19
|
+
The old domain record types do not exist. In the reproduction profile, `procedure` is an Entity, `execution` is an Activity, `observation` is an Entity, and `claim`/`assessment` are Assertions. `scientificProjections` provides these semantic views to existing scientific modules; these views are not additional wire formats.
|
|
20
|
+
|
|
21
|
+
A logical ID identifies an object across revisions; the record SHA-256 digest identifies its exact content version. The same ID can have different versions in different artifacts. A snapshot contains at most one version per logical ID; another version can be referenced through another artifact. Identical titles, statements, metrics or output bytes do not merge logical objects or prove independent execution. Shared executions use the actual run identity across claim-specific assessments.
|
|
22
|
+
|
|
23
|
+
Activity status is `running`, `succeeded`, `failed`, `cancelled`, `partial`, `timedOut` or `unknown`. A prospective procedure must not become an Activity merely because it was planned. Activity success is operational, not scientific support. Scientific verdict fields are forbidden on an Activity.
|
|
24
|
+
|
|
25
|
+
An observation records its basis (`observed`, `declared`, `inferred`, `attested`), raw source and limitations. A measurement uses `metric`, `value.decimal`, `unit` and, when known, `scope`, `uncertainty` and `method`; scientific decimal values are strings. Missing uncertainty or data remains unknown. A failure observation must not contain a fabricated measurement.
|
|
26
|
+
|
|
27
|
+
An assessment has `subject` (the reproduction profile uses `claim`), `evidence`, `method`, `performedBy`, `scope`, `limitations` and `conclusion`. Conclusions are `supports`, `challenges`, `contradicts`, `inconclusive`. Different assessments can conflict. Absence of assessment is a view state, not a scientific verdict. Reproduction assessments additionally bind the procedure, actual execution and precommitted verification policy; the policy commitment algorithm remains `citeark-policy-commitment-v1`.
|
|
28
|
+
|
|
29
|
+
## Relations and provenance
|
|
30
|
+
|
|
31
|
+
A Relation is a content-addressed record with `predicate`, exact `subject` and `object` references, `basis`, exact `attributedTo` Agent reference, `sources` and optional `qualifiers`. A reference is `{ref, digest, recordType}` plus `artifactDigest` for an external record. The record endpoints and attribution have to bind exact versions. The relation's own digest is not embedded in its endpoints, avoiding a hash cycle.
|
|
32
|
+
|
|
33
|
+
| Predicate | Direction |
|
|
34
|
+
| --- | --- |
|
|
35
|
+
| used / generated | Activity → input/output Entity |
|
|
36
|
+
| follows | actual Activity → procedure Entity |
|
|
37
|
+
| associatedWith | Activity → Agent |
|
|
38
|
+
| partOf | component Activity → containing Activity |
|
|
39
|
+
| requires | prospective procedure Entity → required Entity |
|
|
40
|
+
| hasStep | procedure Entity → prospective procedure-step Entity |
|
|
41
|
+
| expects | prospective procedure Entity → prospective output specification Entity |
|
|
42
|
+
| about | scientific object → context Entity |
|
|
43
|
+
| plannedFor | procedure Entity → target Assertion |
|
|
44
|
+
| derivedFrom | derived Entity → source Entity |
|
|
45
|
+
| assertedIn | Assertion → source Entity |
|
|
46
|
+
| attributedTo | scientific object → Agent |
|
|
47
|
+
| evidenceFor | observation Entity → Assertion (relevance, not support) |
|
|
48
|
+
| assesses | assessment Assertion → subject Assertion |
|
|
49
|
+
| basedOn | assessment Assertion → scientific basis/context |
|
|
50
|
+
| describes | Entity → scientific object |
|
|
51
|
+
| revises / supersedes | new version → old version of the same primitive |
|
|
52
|
+
|
|
53
|
+
Reproduction-profile embedded bindings (e.g. `generatedBy`, `tests`, `evidence`) are materialized into Relation records by the assembler. Those local bindings may use `{ref}` because the immutable manifest fixes the snapshot; Relation endpoints always include exact digests. Cross-artifact references always include artifact, record digest and primitive. Reproduction-specific external bindings also state their semantic role.
|
|
54
|
+
|
|
55
|
+
The assembler identifies each embedded array binding by its exact source pointer, including the array index (for example `/sourceContext/1`). Multiple locations in the same source, or multiple input roles referring to the same object, retain separate relations and their qualifiers. Repeated declarations are preserved; duplicate explicit logical record IDs remain invalid. Existing immutable snapshots retain their original relation identities.
|
|
56
|
+
|
|
57
|
+
An external reference is a declaration of identity, not proof that the dependency was obtained, is authentic or is readable. Verification checks its structure but does not fetch it. Queries resolve only against the exact authorized artifact set supplied by their caller. A hidden or absent dependency remains `unavailable`; hashes do not grant access. A signature authenticates a statement and its signer, never upgrades `declared` into `observed`.
|
|
58
|
+
|
|
59
|
+
## Immutable artifact and bundle
|
|
60
|
+
|
|
61
|
+
`cap-manifest.json` contains:
|
|
62
|
+
|
|
63
|
+
- `$schema`: `https://citeark.com/schemas/cap/v2/manifest.schema.json`;
|
|
64
|
+
- `specVersion`: `2.0.0-alpha.2`;
|
|
65
|
+
- `mediaType`: `application/vnd.citeark.cap.manifest.v2+json`;
|
|
66
|
+
- `artifact`: creation timestamp, creator reference, optional logical artifact ID;
|
|
67
|
+
- `profiles`, explicitly versioned under `/cap/profiles/{name}/2.0`;
|
|
68
|
+
- `roots`: nonempty list of role and exact local reference;
|
|
69
|
+
- `records`: ID, primitive, role, media type, schema, SHA-256 digest, byte size and path;
|
|
70
|
+
- `blobs`: digest-bound materials and availability;
|
|
71
|
+
- `relations`: artifact-level links (`derivedFrom`, `revises`, `reproduces`, `extends`, `supersedes`, `refersTo`). These links differ from scientific Relation records.
|
|
72
|
+
|
|
73
|
+
Record JSON and manifests use RFC 8785 canonical JSON: finite JSON values, I-JSON-safe strings, deterministic key ordering and number serialization. Record bytes live at `records/sha256/<first-two-hex>/<hex>.json`; embedded Blob bytes at `blobs/sha256/<first-two-hex>/<hex>`. The artifact digest is SHA-256 of manifest bytes, independent of archive metadata and detached attachments. Each changed scientific fact creates a new record and artifact digest.
|
|
74
|
+
|
|
75
|
+
Blobs have `embedded`, `external` or `withheld` availability, an exact byte size, digest, media type and roles. External/withheld blobs have no bundled path. Access policy and per-resource rights never grant permissions the producer does not hold. A digest-only prompt identity may be an Entity without a Blob when its original byte size/material is not known. This must not be advertised as retrievable content.
|
|
76
|
+
|
|
77
|
+
The transport is a `.cap` tar+gzip archive with media type `application/vnd.citeark.cap+tar+gzip;version=2`. The archive digest binds transport bytes; it is not the artifact digest. Relative paths must be safe; duplicate paths, traversal, symlinks, devices and undeclared scientific files are rejected. `attestations/`, `preview/`, `projections/`, `cap-locations.json` and `ro-crate-metadata.json` are detached attachments. Their bytes are bound by archive/content digests, not by the scientific manifest. Previews, layouts and location hints cannot modify scientific meaning.
|
|
78
|
+
|
|
79
|
+
## Profiles and trust
|
|
80
|
+
|
|
81
|
+
Core requires the five-primitive model, hashes, exact relation endpoints, creator and roots. It requires neither a paper, a hosted job, an execution, nor an assessment.
|
|
82
|
+
|
|
83
|
+
Research Compilation describes source-grounded interpretation and retains compiler output/inventory, without inventing execution. Research Plan roots all planned claims and procedures and may retain declared source measurements. Computational Run includes a procedure, actual execution and observations. Reproduction binds reference work, targets, execution and assessments; it reveals and verifies the private policy commitment after execution. Agent Trace adds structured ordered trace events. Restricted Evidence describes unavailable materials and access policy. Public Bundle requires per-resource rights, a usable RO-Crate projection and at least one valid detached root attestation. See `profiles/` for profile-specific requirements.
|
|
84
|
+
|
|
85
|
+
Attestations remain separate from scientific records. They use Ed25519 over an in-toto Statement v1 in a DSSE envelope. CAP's predicate type is `https://citeark.com/cap/attestations/artifact/v2` and attestation media type is `application/vnd.citeark.cap.attestation.v2+json`. The signed subject is the artifact digest. Actor, role, timestamp and optional hosting-task binding are signed metadata. A new attestation does not change scientific identity. A public-key digest identifies signing material; trust lists and hosted-task admission are application policy. Signature validity, trusted signer, authentic execution and scientific support are distinct findings.
|
|
86
|
+
|
|
87
|
+
## Agent trace and execution
|
|
88
|
+
|
|
89
|
+
Preserve observable inputs, model/provider and version when available, workflow prompts/templates, tool calls/results, commands, patches, decisions, errors and retries that affect results. Large/full transcripts stay in blobs and are optional. Hidden chain of thought is not required or claimed to be captured.
|
|
90
|
+
|
|
91
|
+
Disclosure distinguishes `minimal`, `audit` and fuller retained material; label each actual retained source rather than claiming completeness. Redacted artifacts disclose redaction and original digest when known. The execution producer retains staged workflow instructions, runner commands, workspace changes, assessment context and digest-only model prompts where that is all the assessor recorded.
|
|
92
|
+
|
|
93
|
+
Helper step logs are Agent-writable declarations. They identify actual reported invocations, fixed plans, raw log outputs, exit status and missing completion, without being upgraded to independent runner observation. Commands are not heuristically relabeled as training/evaluation. Uncaptured fine-grained scientific phases stay explicit gaps. Merely reassessing retained evidence creates new assessment records and relations, preserving Activity and observation bytes without rerunning scientific commands.
|
|
94
|
+
|
|
95
|
+
## Research-object granularity (alpha.2)
|
|
96
|
+
|
|
97
|
+
Every explicitly identified dataset, split/sample, model architecture, checkpoint, method and code revision can be its own Entity. `researchObjects` is the Agent's structured producer input: explicit IDs, semantic roles, known identity fields, source locators, conditions and derivation. A source mention is not byte identity. `sourceId` aliases the exact material source; paper citations use `sourceLocators`. Equal names never merge objects. Planned output specifications carry `prospective: true` and `availability: not-yet-generated`, never actual byte digests or Activity success.
|
|
98
|
+
|
|
99
|
+
An executable experiment remains a `procedure` package. Its independently identified `procedure-step` children describe preparation, preprocessing, training, evaluation or analysis, with explicit inputs, expected outputs and dependencies. Shared step IDs must describe the same recipe. `hasStep` does not authorize running a child separately. `expects` always targets a prospective specification and never means an output was generated. Source measurements are separate declared observation Entities in both Compilation and Plan artifacts; `about` connects claims and observations to their context objects without asserting readiness or scientific support.
|
|
100
|
+
|
|
101
|
+
Plan producer granularity follows scientific methods and coherent products. A procedure-step may iterate over complete checkpoint/dataset/seed/layer axes and expect a result collection or trained-state family whose conditions preserve all dimensions and member identity rules. A loop coordinate alone is not a new prospective step or Entity. Split materially different operations, scientific state, decision rules or reusable consumer scope; preserve all already explicit source identities. Accounting batches may partition exact shared subsets and disjoint remainders across consumers without creating matching graph partitions. Inputs and derivation describe direct scientific dependencies. This convention does not cap node counts, reduce experimental scope, or merge distinct actual observations/checkpoints; execution records actual members and their identities.
|
|
102
|
+
|
|
103
|
+
A claim represents a scientifically meaningful proposition, not each supporting numerical cell. Producer measurements may explicitly reference a source `researchObjects` observation through `observationId`. This produces one declared observation Entity with `kind: source-observation-collection`, optional `observationKind` (table, matrix, series, collection or scalar), and a `measurements` array retaining each measurement ID, decimal value, metric, unit, dimensions, conditions, source locator, context objects and extraction uncertainty. Distinct claims can cite the same collection via `evidenceFor`; numerical rows do not create extra observation nodes. Measurement IDs remain globally unique and continue to bind execution contracts and individual assessments. Existing scalar inputs remain separate observations. Collections are never inferred from shared titles, pages or figure numbers; neither a fixed claim count nor one claim per figure is required.
|
|
104
|
+
|
|
105
|
+
Source-proposed mechanisms use optional `hypotheses` and become Assertions with role `hypothesis`, fixed source locators and proposed status. They carry no reproduction disposition or scientific verdict and are excluded from empirical claim targets. Source auditing checks whether an explanation is faithfully represented as proposed; audit acceptance is not support for that explanation.
|
|
106
|
+
|
|
107
|
+
An experiment package keeps its own declared claim targets. A procedure-step has only explicitly supplied `claimIds`; omission means no direct claim target, rather than inheritance from every consuming package. Shared preparation/loading steps retain their inputs, expected outputs, dependencies and package membership without generating a claim edge for every consumer. These producer conventions are additive within alpha.2 and do not rewrite existing immutable artifacts.
|
|
108
|
+
|
|
109
|
+
The executor can bind a real helper invocation with `run-research-plan.mjs research-plan.json --step <id> <command...>`. This adds a declared `follows` relation to the planned step; the helper log remains Agent-writable. Unbound commands retain their captured process identity without guessed scientific phases. Actual model/checkpoint/dataset outputs receive their own observed material Entities and generated relations, while model bytes remain withheld. An explicit output `scientificObjectId` links a prospective specification to actual bytes through an attributed declared binding, preserving both identities.
|
|
110
|
+
|
|
111
|
+
Each available measurement judgment becomes a `measurement-assessment` Assertion with its exact measurement scope, supporting observations, reason, assessor and limitations. The compound `assessment` preserves its original verdict and links the component judgments; an individually supported measurement does not promote an incompletely tested claim to support. Consumers projecting experiment packages or compound verdicts continue to select exact roles `procedure` and `assessment`.
|
|
112
|
+
|
|
113
|
+
## Query and views
|
|
114
|
+
|
|
115
|
+
The lightweight package exports `createArkGraph`, `queryArkGraph`, `getArkGraphRecord`, `deriveArkRoute`, `compareArkRoutes` and `traceArkGraphProvenance`. Input artifacts must already be verified and authorized. The rule version is `arkgraph/1`. Queries return exact inputs, scope, missing roots, dependency status, pagination and truncation. A route is a relevant subgraph with branches and shared inputs, not an enumeration of every path.
|
|
116
|
+
|
|
117
|
+
Comparison uses exact identities and a caller-specified target set. It returns shared/changed versions, assessment coverage and unknowns, preserving all conflicting judgments. It does not match measurements by display name, infer support from tolerance, count partial results as complete coverage, or infer independence from different digests. Layout and route selections are derived views; future editable selections must be separately versioned.
|
|
118
|
+
|
|
119
|
+
Local CLI: `citeark-agent graph --cap result.cap --operation subgraph`; supply `--query query.json` for record, route, compare and provenance selections. Remote queries use `--server https://citeark.co` and optional `CITEARK_API_KEY`. Local research and graph reading require no platform account.
|
|
120
|
+
|
|
121
|
+
## Conformance examples
|
|
122
|
+
|
|
123
|
+
`examples/arkgraph/` contains four synthetic input fixtures and their builder: checkpoint reuse/evaluation, training/evaluation with shared targets and conflicting assessments, partial failure, and paper-free research with an unresolved earlier procedure version. Tests assemble, sign, archive, read and query all four. They validate software semantics; their values are not scientific experiment results.
|
|
124
|
+
|
|
125
|
+
科研计划的观察比较提案可保留在 procedure 的 `citeark.observationTargets`,其显式来源观察以标准 `about` 关系绑定。此字段不是数值测量或判定结果,只有该目标的提案不能产生数值执行合同;来源身份与主张归属必须验证。详见 [规划粒度](../docs/research-plan-granularity.md)。
|
|
126
|
+
|
|
127
|
+
## Research reading projection
|
|
128
|
+
|
|
129
|
+
Producers may retain a source-grounded twelve-category reading organization in the source work's `citeark.reading` metadata. Mainline/branch identities and display labels are editorial views, never scientific edges or execution authorization. New context roles and source scientific relations retain fixed source provenance. A source-reported `research-activity` is a declared Activity with unknown runtime status; it is not this compiler's execution. Source evaluation prose uses `source-assessment`, distinct from reproduction assessments and their evidence-bound verdicts. See [ArkGraph reading](../docs/arkgraph-reading.md) and the Compilation Profile for admissibility. Consumers must use a reader supporting these roles before enabling this producer output; existing immutable artifacts are unchanged.
|
package/protocol/LICENSE
ADDED
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
CAP specification dedication
|
|
2
|
+
|
|
3
|
+
The CiteArk Artifact Protocol specification text, JSON Schemas, and
|
|
4
|
+
conformance vectors are dedicated to the public domain under the Creative
|
|
5
|
+
Commons CC0 1.0 Universal dedication.
|
|
6
|
+
|
|
7
|
+
https://creativecommons.org/publicdomain/zero/1.0/
|
|
8
|
+
|
|
9
|
+
This dedication applies to the protocol materials only. It does not grant any
|
|
10
|
+
rights in packaged papers, source code, datasets, models, checkpoints, or other
|
|
11
|
+
third-party materials, and it does not change the license of the CiteArk
|
|
12
|
+
software implementation.
|
|
@@ -0,0 +1,72 @@
|
|
|
1
|
+
# CAP 2.0 standard mappings
|
|
2
|
+
|
|
3
|
+
CAP defines scientific Claim, Evidence, Assessment and Agent-execution
|
|
4
|
+
semantics. This document specifies interoperability boundaries with established
|
|
5
|
+
standards. A projection never overrides canonical CAP Records.
|
|
6
|
+
|
|
7
|
+
## RO-Crate 1.3
|
|
8
|
+
|
|
9
|
+
The Public Bundle Profile emits `ro-crate-metadata.json`:
|
|
10
|
+
|
|
11
|
+
| CAP | RO-Crate projection |
|
|
12
|
+
| --- | --- |
|
|
13
|
+
| Artifact | Root Data Entity (`Dataset`) |
|
|
14
|
+
| Record | contextual entity with stable `@id` |
|
|
15
|
+
| Blob | file/data entity with `contentSize`, `encodingFormat` and checksum |
|
|
16
|
+
| Profile | `conformsTo` |
|
|
17
|
+
| Artifact relation | contextual relationship or qualified entity |
|
|
18
|
+
|
|
19
|
+
The CAP Artifact digest and Record digests remain authoritative for integrity.
|
|
20
|
+
JSON-LD identifiers and current content URLs are not substitutes for digests.
|
|
21
|
+
|
|
22
|
+
## Workflow Run RO-Crate
|
|
23
|
+
|
|
24
|
+
When an Experiment is implemented as a computational workflow, prospective
|
|
25
|
+
workflow structure and retrospective run provenance SHOULD be projected using
|
|
26
|
+
Workflow Run RO-Crate. CAP does not require every Agent command to be modeled as
|
|
27
|
+
a workflow step. The original workflow crate can also be referenced as a
|
|
28
|
+
digest-bound Blob or SourceWork source.
|
|
29
|
+
|
|
30
|
+
## W3C PROV
|
|
31
|
+
|
|
32
|
+
| CAP | W3C PROV |
|
|
33
|
+
| --- | --- |
|
|
34
|
+
| Record or Blob | `prov:Entity` |
|
|
35
|
+
| Execution | `prov:Activity` |
|
|
36
|
+
| Actor | `prov:Agent` |
|
|
37
|
+
| Execution input | `prov:used` |
|
|
38
|
+
| Evidence generated by Execution | `prov:wasGeneratedBy` |
|
|
39
|
+
| Record origin actor | `prov:wasAttributedTo` |
|
|
40
|
+
| Artifact or Evidence derivation | `prov:wasDerivedFrom` |
|
|
41
|
+
|
|
42
|
+
The mapping is an export view. CAP's observed, declared, inferred and attested
|
|
43
|
+
basis remains necessary because a generic PROV relation does not by itself state
|
|
44
|
+
the capture trust boundary.
|
|
45
|
+
|
|
46
|
+
## in-toto and DSSE
|
|
47
|
+
|
|
48
|
+
CAP Artifact Attestations use an in-toto Statement v1 and predicate type
|
|
49
|
+
`https://citeark.com/cap/attestations/artifact/v2`. The subject SHA-256 is the
|
|
50
|
+
hexadecimal portion of `artifactDigest`. DSSE is the preferred authentication
|
|
51
|
+
envelope.
|
|
52
|
+
|
|
53
|
+
CAP's reference bundle can carry an Ed25519 public key for offline integrity
|
|
54
|
+
verification. Sigstore Bundle verification material, institutional PKI, DID or
|
|
55
|
+
external key lookup are consumer policies and do not change the Statement.
|
|
56
|
+
|
|
57
|
+
## SPDX 3.0.1
|
|
58
|
+
|
|
59
|
+
SPDX expressions MAY be embedded for simple resource rights. A detailed SPDX
|
|
60
|
+
3.0.1 document SHOULD be included as a projection or digest-bound Blob when the
|
|
61
|
+
Artifact needs software, build, model, dataset, dependency, provenance or
|
|
62
|
+
license inventory.
|
|
63
|
+
|
|
64
|
+
Each CAP SourceWork source and Blob keeps its own rights reference. A package-
|
|
65
|
+
wide SPDX document MUST NOT erase or override resource-level differences.
|
|
66
|
+
|
|
67
|
+
## OCI
|
|
68
|
+
|
|
69
|
+
CAP descriptors use the OCI-compatible concepts of digest, byte size and media
|
|
70
|
+
type. A future OCI Bundle transport may publish the canonical Manifest as an OCI
|
|
71
|
+
manifest or subject and Records/Blobs as content-addressed blobs. CAP 2.0 does
|
|
72
|
+
not require an OCI Registry and the `.cap` transport remains `tar+gzip`.
|
|
@@ -0,0 +1,38 @@
|
|
|
1
|
+
# CAP implementer entry point
|
|
2
|
+
|
|
3
|
+
CAP is an open, Agent-neutral protocol for immutable, content-addressed and
|
|
4
|
+
independently verifiable scientific snapshots.
|
|
5
|
+
|
|
6
|
+
Start with:
|
|
7
|
+
|
|
8
|
+
1. [`CAP.md`](CAP.md) — normative CAP `2.0.0-alpha.2` specification;
|
|
9
|
+
2. [`profiles/`](profiles/) — Core, Research Plan, Computational Run,
|
|
10
|
+
Reproduction, Agent Trace, Public Bundle and Restricted Evidence constraints;
|
|
11
|
+
3. [`MAPPINGS.md`](MAPPINGS.md) — RO-Crate, W3C PROV, in-toto/DSSE, SPDX and
|
|
12
|
+
OCI interoperability boundaries;
|
|
13
|
+
4. [`conformance-v2.0-alpha.1.json`](conformance-v2.0-alpha.1.json) — RFC 8785,
|
|
14
|
+
verification-policy commitment and DSSE test vectors;
|
|
15
|
+
5. [`../schemas/cap/v2/`](../schemas/cap/v2/) — the nine Core JSON Schemas;
|
|
16
|
+
6. [`../src/cap/v2/`](../src/cap/v2/) — reference assembler, verifier,
|
|
17
|
+
attestation and `tar+gzip` Bundle implementation.
|
|
18
|
+
|
|
19
|
+
A minimal Core consumer needs safe archive extraction, RFC 8785 canonical JSON,
|
|
20
|
+
SHA-256 and the active Profile rules. Signatures are detached: DSSE and in-toto
|
|
21
|
+
are used when an Attestation is present, but an unsigned private Core Artifact
|
|
22
|
+
can still be structurally and cryptographically content-valid.
|
|
23
|
+
|
|
24
|
+
The reference CLI accepts CAP 2.0 only:
|
|
25
|
+
|
|
26
|
+
```bash
|
|
27
|
+
citeark-agent cap verify --dir ./artifact
|
|
28
|
+
citeark-agent cap pack --dir ./artifact --output ./artifact.cap
|
|
29
|
+
citeark-agent cap verify --file ./artifact.cap
|
|
30
|
+
```
|
|
31
|
+
|
|
32
|
+
There is no legacy parser, format auto-detection, `.citeark` alias or migration
|
|
33
|
+
mode. Unsupported versions are rejected. Importing non-CAP research material
|
|
34
|
+
means creating a new CAP 2.0 Artifact with explicit source and provenance
|
|
35
|
+
Records, not relabeling the source bytes.
|
|
36
|
+
|
|
37
|
+
Protocol text, schemas and conformance vectors are released under CC0 1.0 so
|
|
38
|
+
independent implementations can reuse them without asking permission.
|
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
{
|
|
2
|
+
"specVersion": "2.0.0-alpha.1",
|
|
3
|
+
"canonicalJson": {
|
|
4
|
+
"input": {
|
|
5
|
+
"z": 1,
|
|
6
|
+
"a": {
|
|
7
|
+
"😀": "ok",
|
|
8
|
+
"b": [true, null, 1e-7]
|
|
9
|
+
}
|
|
10
|
+
},
|
|
11
|
+
"canonical": "{\"a\":{\"b\":[true,null,1e-7],\"😀\":\"ok\"},\"z\":1}",
|
|
12
|
+
"sha256": "sha256:41c375394ddf9f1f80033cde1fb81c1f250be02281665fe89ac3a2335cc973d0"
|
|
13
|
+
},
|
|
14
|
+
"policyCommitment": {
|
|
15
|
+
"nonce": "MDEyMzQ1Njc4OWFiY2RlZg",
|
|
16
|
+
"policy": {
|
|
17
|
+
"reportedValue": {
|
|
18
|
+
"decimal": "82.1",
|
|
19
|
+
"unit": "percent"
|
|
20
|
+
},
|
|
21
|
+
"criterion": {
|
|
22
|
+
"operator": "absoluteDifferenceLte",
|
|
23
|
+
"value": {
|
|
24
|
+
"decimal": "0.5",
|
|
25
|
+
"unit": "percentage-point"
|
|
26
|
+
}
|
|
27
|
+
}
|
|
28
|
+
},
|
|
29
|
+
"sha256": "sha256:9bc637a31d9ceb4a0cdf081b9e5a3cc212e3f324811e6d8c24cd7ba2b60ef78f"
|
|
30
|
+
},
|
|
31
|
+
"dsse": {
|
|
32
|
+
"payloadType": "application/vnd.in-toto+json",
|
|
33
|
+
"payloadBase64": "e30=",
|
|
34
|
+
"paeBase64": "RFNTRXYxIDI4IGFwcGxpY2F0aW9uL3ZuZC5pbi10b3RvK2pzb24gMiB7fQ=="
|
|
35
|
+
}
|
|
36
|
+
}
|