@citeark/agent 0.3.20
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +202 -0
- package/README.md +128 -0
- package/data/dataset-source-registry.v1.json +300 -0
- package/dist/arkgraph/boot.js +6 -0
- package/dist/arkgraph/index.html +1 -0
- package/dist/arkgraph/viewer.css +1 -0
- package/dist/arkgraph/viewer.en.css +1 -0
- package/dist/arkgraph/viewer.en.js +49 -0
- package/dist/arkgraph/viewer.en.js.LEGAL.txt +56 -0
- package/dist/arkgraph/viewer.js +49 -0
- package/dist/arkgraph/viewer.js.LEGAL.txt +56 -0
- package/docker/claude-code/Dockerfile +97 -0
- package/docker/claude-code/codex-pro-relay.mjs +466 -0
- package/docker/claude-code/runtime-contract-check.mjs +79 -0
- package/docs/arkgraph-reading.md +79 -0
- package/docs/configuration.md +100 -0
- package/docs/integration.md +92 -0
- package/docs/maturity-plan.md +27 -0
- package/docs/npm-release.md +44 -0
- package/docs/paper-reading.md +40 -0
- package/docs/research-plan-granularity.md +27 -0
- package/docs/terminal.md +49 -0
- package/examples/toy-evaluation/compile-task.json +27 -0
- package/examples/toy-evaluation/paper.md +5 -0
- package/examples/toy-evaluation/repository/README.md +9 -0
- package/examples/toy-evaluation/repository/checkpoint.json +4 -0
- package/examples/toy-evaluation/repository/evaluate.py +17 -0
- package/examples/toy-evaluation/task.json +81 -0
- package/package.json +59 -0
- package/prompts/compile-research.md +58 -0
- package/prompts/execute-contract.md +72 -0
- package/prompts/execute-workspace-simple.md +51 -0
- package/prompts/execute-workspace.md +34 -0
- package/prompts/prepare-reproduction.md +82 -0
- package/prompts/repair-research.md +45 -0
- package/protocol/CAP.md +129 -0
- package/protocol/LICENSE +12 -0
- package/protocol/MAPPINGS.md +72 -0
- package/protocol/README.md +38 -0
- package/protocol/conformance-v2.0-alpha.1.json +36 -0
- package/protocol/examples/arkgraph/checkpoint-evaluation.json +309 -0
- package/protocol/examples/arkgraph/fixtures.mjs +49 -0
- package/protocol/examples/arkgraph/paper-free.json +291 -0
- package/protocol/examples/arkgraph/partial-failure.json +344 -0
- package/protocol/examples/arkgraph/training-evaluation.json +443 -0
- package/protocol/profiles/agent-trace.md +16 -0
- package/protocol/profiles/computational-run.md +16 -0
- package/protocol/profiles/core.md +15 -0
- package/protocol/profiles/public-bundle.md +18 -0
- package/protocol/profiles/reproduction.md +29 -0
- package/protocol/profiles/research-compilation.md +44 -0
- package/protocol/profiles/research-plan.md +39 -0
- package/protocol/profiles/restricted-evidence.md +15 -0
- package/runtime/bootstrap-autodl-runtime.sh +314 -0
- package/runtime/create-runtime-venv.sh +41 -0
- package/runtime/install-local-cpu-runtime.sh +23 -0
- package/runtime/install-scientific-runtime.sh +153 -0
- package/runtime/mineru/parse.py +62 -0
- package/runtime/mineru/requirements.txt +4 -0
- package/runtime/requirements-baseline.txt +38 -0
- package/schemas/cap/v2/activity.schema.json +47 -0
- package/schemas/cap/v2/agent.schema.json +32 -0
- package/schemas/cap/v2/assertion.schema.json +110 -0
- package/schemas/cap/v2/descriptor.schema.json +243 -0
- package/schemas/cap/v2/entity.schema.json +64 -0
- package/schemas/cap/v2/manifest.schema.json +67 -0
- package/schemas/cap/v2/relation.schema.json +82 -0
- package/schemas/compute-catalog.schema.json +63 -0
- package/schemas/compute-decision.schema.json +27 -0
- package/schemas/execution-contract.schema.json +1024 -0
- package/schemas/research-card.schema.json +30 -0
- package/schemas/research-inventory-draft.schema.json +366 -0
- package/schemas/research.schema.json +1044 -0
- package/schemas/result.schema.json +173 -0
- package/schemas/verification-policy.schema.json +47 -0
- package/schemas/verified-conclusion.schema.json +58 -0
- package/schemas/workspace-summary.schema.json +24 -0
- package/scripts/build-arkgraph-view.mjs +12 -0
- package/scripts/check-execution-feasibility.mjs +24 -0
- package/scripts/check-syntax.mjs +15 -0
- package/scripts/deterministic-asset-preparation.py +438 -0
- package/scripts/package-cap.mjs +23 -0
- package/scripts/package-local-agent.mjs +23 -0
- package/scripts/preview-arkgraph.mjs +25 -0
- package/scripts/replay-research-compiler-candidate.mjs +134 -0
- package/scripts/review-compiler-sources.mjs +44 -0
- package/scripts/run-asset-preparation.sh +17 -0
- package/scripts/run-research-plan.mjs +98 -0
- package/scripts/validate-asset-preparation.py +290 -0
- package/scripts/verify-local-runtime.mjs +57 -0
- package/scripts/verify-npm-package.mjs +57 -0
- package/src/adapters/paper2agent.mjs +107 -0
- package/src/assets/cache.mjs +159 -0
- package/src/assets/compute.mjs +98 -0
- package/src/assets/executor.mjs +145 -0
- package/src/assets/lifecycle.mjs +213 -0
- package/src/assets/manifest.mjs +242 -0
- package/src/assets/opportunistic-preparation.mjs +81 -0
- package/src/assets/plan.mjs +411 -0
- package/src/assets/prompts.mjs +29 -0
- package/src/assets/public-asset-probe.mjs +525 -0
- package/src/assets/qualification.mjs +119 -0
- package/src/assets/readiness.mjs +130 -0
- package/src/assets/reproduction-admission.mjs +355 -0
- package/src/assets/requirements.mjs +152 -0
- package/src/assets/source-grounding.mjs +341 -0
- package/src/assets/source-policy.mjs +118 -0
- package/src/autodl/client.mjs +260 -0
- package/src/autodl/ssh.mjs +380 -0
- package/src/autodl/tools.mjs +129 -0
- package/src/cap/redaction.mjs +38 -0
- package/src/cap/v2/archive.mjs +152 -0
- package/src/cap/v2/attestation.mjs +204 -0
- package/src/cap/v2/canonical-json.mjs +114 -0
- package/src/cap/v2/compilation-artifact.mjs +240 -0
- package/src/cap/v2/core.mjs +282 -0
- package/src/cap/v2/measurement-assessment-records.mjs +23 -0
- package/src/cap/v2/pipeline-artifact.mjs +922 -0
- package/src/cap/v2/read.mjs +41 -0
- package/src/cap/v2/reassessment-artifact.mjs +383 -0
- package/src/cap/v2/research-artifact.mjs +231 -0
- package/src/cap/v2/research-map-records.mjs +46 -0
- package/src/cap/v2/research-object-records.mjs +163 -0
- package/src/cap/v2/research-records.mjs +187 -0
- package/src/cap/v2/verify.mjs +642 -0
- package/src/cli.mjs +1146 -0
- package/src/compute/autodl-pro-compiler.mjs +347 -0
- package/src/compute/autodl-pro-executor.mjs +459 -0
- package/src/compute/autodl-pro-job.mjs +843 -0
- package/src/compute/autodl-pro-network.mjs +295 -0
- package/src/compute/autodl-pro-remote.mjs +810 -0
- package/src/compute/autodl-pro-staging.mjs +117 -0
- package/src/compute/campaign.mjs +110 -0
- package/src/compute/catalog.mjs +123 -0
- package/src/compute/checkpoint-protocol.mjs +154 -0
- package/src/compute/codex-account-lock.mjs +111 -0
- package/src/compute/codex-account-session.mjs +107 -0
- package/src/compute/compiler-profile.mjs +38 -0
- package/src/compute/compiler-router.mjs +23 -0
- package/src/compute/coordinator-recovery.mjs +210 -0
- package/src/compute/executor-router.mjs +29 -0
- package/src/compute/gcp-batch-compiler.mjs +685 -0
- package/src/compute/gcp-batch-executor.mjs +1215 -0
- package/src/compute/gcp-batch-failure.mjs +92 -0
- package/src/compute/gcp-batch-job.mjs +527 -0
- package/src/compute/gcp-batch-lifecycle.mjs +81 -0
- package/src/compute/gcp-checkpoint-worker.mjs +1633 -0
- package/src/compute/local-codex-compiler.mjs +52 -0
- package/src/compute/measurement-hardware.mjs +128 -0
- package/src/compute/remote-attempt.mjs +226 -0
- package/src/compute/requirements.mjs +124 -0
- package/src/compute/research-phases.mjs +48 -0
- package/src/compute/scheduler.mjs +452 -0
- package/src/compute/shared-workloads.mjs +26 -0
- package/src/compute/stage-archive.mjs +79 -0
- package/src/contracts/campaign-contract.mjs +52 -0
- package/src/contracts/execution-contract.mjs +819 -0
- package/src/contracts/execution-mode.mjs +19 -0
- package/src/contracts/execution-timeouts.mjs +45 -0
- package/src/contracts/execution-workload.mjs +68 -0
- package/src/contracts/preflight-schema.mjs +25 -0
- package/src/contracts/public-contract.mjs +63 -0
- package/src/contracts/subject-tags.mjs +31 -0
- package/src/dashboard/data.mjs +898 -0
- package/src/dashboard/server.mjs +79 -0
- package/src/dashboard/static/dashboard.css +366 -0
- package/src/dashboard/static/dashboard.js +560 -0
- package/src/dashboard/static/index.html +85 -0
- package/src/deployment/community-policy.mjs +9 -0
- package/src/deployment/environment.mjs +112 -0
- package/src/deployment/guided.mjs +98 -0
- package/src/deployment/handoff.mjs +102 -0
- package/src/deployment/local-contract.mjs +31 -0
- package/src/deployment/local.mjs +100 -0
- package/src/deployment/prepare.mjs +46 -0
- package/src/deployment/recipe.mjs +108 -0
- package/src/deployment/supplement.mjs +51 -0
- package/src/deployment/terminal.mjs +43 -0
- package/src/diagnosis/renderer.mjs +75 -0
- package/src/diagnosis/target-failure.mjs +46 -0
- package/src/evidence/parser-registry.mjs +54 -0
- package/src/evidence/parsers/fasttext-classification.mjs +82 -0
- package/src/evidence/parsers/json-scalar.mjs +96 -0
- package/src/evidence/parsers/simcse-senteval.mjs +104 -0
- package/src/evidence/parsers/starspace-classification.mjs +78 -0
- package/src/evidence/registry.mjs +147 -0
- package/src/execution/runner-audit.mjs +473 -0
- package/src/gcp/auth.mjs +106 -0
- package/src/gcp/batch-client.mjs +120 -0
- package/src/gcp/resource-discovery.mjs +177 -0
- package/src/gcp/rest.mjs +82 -0
- package/src/gcp/secret-manager.mjs +34 -0
- package/src/gcp/signed-url.mjs +133 -0
- package/src/gcp/storage.mjs +220 -0
- package/src/graph/command.mjs +41 -0
- package/src/graph/execution.mjs +97 -0
- package/src/graph/model.mjs +37 -0
- package/src/graph/presentation.mjs +110 -0
- package/src/graph/query.mjs +159 -0
- package/src/graph/research-relations.mjs +69 -0
- package/src/graph/source-page.mjs +12 -0
- package/src/graph/source-preview.mjs +34 -0
- package/src/graph/validate.mjs +76 -0
- package/src/job.mjs +496 -0
- package/src/network/autodl-routing-proxy.mjs +462 -0
- package/src/network/egress-proxy.mjs +158 -0
- package/src/observability/event-contract.mjs +230 -0
- package/src/observability/pipeline-monitor.mjs +166 -0
- package/src/pipeline/orchestrator.mjs +1281 -0
- package/src/pipeline/recovery-error.mjs +11 -0
- package/src/pipeline/replay.mjs +304 -0
- package/src/pipeline/shared-execution.mjs +115 -0
- package/src/pipeline/stage-checkpoint.mjs +86 -0
- package/src/pipeline/stage-recovery.mjs +101 -0
- package/src/pipeline/targets.mjs +110 -0
- package/src/process.mjs +143 -0
- package/src/protocol.mjs +312 -0
- package/src/provider/codex-account.mjs +44 -0
- package/src/provider/codex-completion.mjs +49 -0
- package/src/provider/completion.mjs +292 -0
- package/src/provider/model-client.mjs +44 -0
- package/src/provider/model-route.mjs +29 -0
- package/src/provider/openrouter-readiness.mjs +189 -0
- package/src/provider/reader-bridge.mjs +35 -0
- package/src/provider/relay.mjs +263 -0
- package/src/provider/runtime-auth.mjs +40 -0
- package/src/public/cap.d.mts +90 -0
- package/src/public/cap.mjs +12 -0
- package/src/public/contracts.d.mts +2 -0
- package/src/public/host.mjs +171 -0
- package/src/public/operations.d.mts +11 -0
- package/src/public/presentation.d.mts +4 -0
- package/src/records/views.mjs +26 -0
- package/src/remote/command.mjs +178 -0
- package/src/remote/ssh.mjs +59 -0
- package/src/repository-origin.mjs +81 -0
- package/src/reproduction/evidence-feedback.mjs +96 -0
- package/src/reproduction/incomplete-initialization.mjs +25 -0
- package/src/reproduction/lifecycle.mjs +253 -0
- package/src/reproduction/plan.mjs +132 -0
- package/src/reproduction/prompts.mjs +70 -0
- package/src/reproduction/runner.mjs +188 -0
- package/src/reproduction/summary.mjs +130 -0
- package/src/reproduction/workspace-mode.mjs +7 -0
- package/src/research/automatic-admission.mjs +156 -0
- package/src/research/compiler-coverage.mjs +85 -0
- package/src/research/compiler-failure.mjs +24 -0
- package/src/research/compiler-normalization-guards.mjs +112 -0
- package/src/research/compiler-repair.mjs +3 -0
- package/src/research/compiler.mjs +853 -0
- package/src/research/continuation-selection.mjs +26 -0
- package/src/research/execution-graph-context.mjs +43 -0
- package/src/research/experiment-importance.mjs +15 -0
- package/src/research/inventory-handoff.mjs +104 -0
- package/src/research/inventory-revisions.mjs +32 -0
- package/src/research/mineru-local.mjs +73 -0
- package/src/research/paper-command.mjs +19 -0
- package/src/research/paper-markdown.mjs +180 -0
- package/src/research/paper-source-map.mjs +69 -0
- package/src/research/planning-policy.mjs +88 -0
- package/src/research/reference-materials.mjs +11 -0
- package/src/research/reproduction-scope.mjs +30 -0
- package/src/research/research-map.mjs +94 -0
- package/src/research/research-objects.mjs +88 -0
- package/src/research/source-discovery.mjs +646 -0
- package/src/research/source-observations.mjs +75 -0
- package/src/research/source-review-cli-mcp.mjs +26 -0
- package/src/research/source-review-input.mjs +209 -0
- package/src/research/source-review-local-codex.mjs +36 -0
- package/src/research/source-review-model.mjs +70 -0
- package/src/research/source-review.mjs +173 -0
- package/src/research/structure.mjs +3163 -0
- package/src/research-card/renderer.mjs +277 -0
- package/src/research-card/verified-conclusion.mjs +143 -0
- package/src/results/output-registry.mjs +183 -0
- package/src/runtime/claude-code.mjs +52 -0
- package/src/runtime/codex-capacity-retry.mjs +87 -0
- package/src/runtime/codex.mjs +64 -0
- package/src/runtime/config.mjs +157 -0
- package/src/runtime/final-output.mjs +40 -0
- package/src/runtime/index.mjs +21 -0
- package/src/runtime/local-codex.mjs +74 -0
- package/src/runtime/opencode.mjs +95 -0
- package/src/runtime/prompt.mjs +13 -0
- package/src/sandbox/docker.mjs +363 -0
- package/src/settings/command.mjs +297 -0
- package/src/settings/store.mjs +119 -0
- package/src/telemetry/pricing.mjs +68 -0
- package/src/telemetry/usage.mjs +265 -0
- package/src/terminal/events.mjs +97 -0
- package/src/terminal/input.mjs +40 -0
- package/src/terminal/plain.mjs +40 -0
- package/src/terminal/remote-stream.mjs +22 -0
- package/src/terminal/screen.mjs +214 -0
- package/src/terminal/transcript.mjs +69 -0
- package/src/util.mjs +107 -0
- package/src/verification/ai-assessor.mjs +534 -0
- package/src/verification/claim-evaluator.mjs +242 -0
- package/src/verification/evidence-context.mjs +165 -0
- package/src/verification/evidence-reader.mjs +95 -0
- package/src/verification/integrity.mjs +570 -0
- package/src/verification/tolerance.mjs +32 -0
- package/src/workloads/cpu-research-preparation.mjs +56 -0
- package/src/workloads/definition.mjs +74 -0
- package/src/workloads/phase-aware-reproduction.mjs +46 -0
- package/src/workloads/reproduction.mjs +85 -0
- package/src/workspace/command.mjs +242 -0
- package/src/workspace/control.mjs +49 -0
- package/src/workspace/entry.mjs +28 -0
- package/src/workspace/input.mjs +93 -0
- package/src/workspace/interactive.mjs +94 -0
- package/src/workspace/jobs.mjs +418 -0
- package/src/workspace/session.mjs +97 -0
- package/src/workspace/worker.mjs +137 -0
- package/ui/arkgraph/ambient-motion.mjs +10 -0
- package/ui/arkgraph/app.jsx +153 -0
- package/ui/arkgraph/boot.js +6 -0
- package/ui/arkgraph/camera-motion.mjs +20 -0
- package/ui/arkgraph/context-reveal.mjs +39 -0
- package/ui/arkgraph/details.css +3 -0
- package/ui/arkgraph/entry.jsx +28 -0
- package/ui/arkgraph/experiment-curves.mjs +17 -0
- package/ui/arkgraph/experiment-selection.mjs +15 -0
- package/ui/arkgraph/experiment-style.css +26 -0
- package/ui/arkgraph/experiment-ui.jsx +32 -0
- package/ui/arkgraph/frame.html +1 -0
- package/ui/arkgraph/graph-gestures.mjs +62 -0
- package/ui/arkgraph/label-layout.mjs +57 -0
- package/ui/arkgraph/locales/en.json +229 -0
- package/ui/arkgraph/locales/source-types.json +15 -0
- package/ui/arkgraph/localization-build.mjs +27 -0
- package/ui/arkgraph/material-build.mjs +23 -0
- package/ui/arkgraph/material-colors.mjs +39 -0
- package/ui/arkgraph/material-style.css +15 -0
- package/ui/arkgraph/open-graph.jsx +326 -0
- package/ui/arkgraph/outline.jsx +49 -0
- package/ui/arkgraph/package-lock.json +888 -0
- package/ui/arkgraph/package.json +17 -0
- package/ui/arkgraph/reading-layout.mjs +130 -0
- package/ui/arkgraph/reading-presentation.mjs +73 -0
- package/ui/arkgraph/record-detail.css +51 -0
- package/ui/arkgraph/record-details.jsx +29 -0
- package/ui/arkgraph/research-types.mjs +31 -0
- package/ui/arkgraph/selection-mark.jsx +6 -0
- package/ui/arkgraph/soft-spine.mjs +26 -0
- package/ui/arkgraph/steering-style.css +187 -0
- package/ui/arkgraph/style.css +272 -0
|
@@ -0,0 +1,74 @@
|
|
|
1
|
+
const AGENT_MODES = new Set(["autonomous", "bounded"]);
|
|
2
|
+
|
|
3
|
+
const REQUIRED_HOOKS = [
|
|
4
|
+
"instructions",
|
|
5
|
+
"prompt",
|
|
6
|
+
"resumePrompt",
|
|
7
|
+
"validate",
|
|
8
|
+
"finalize",
|
|
9
|
+
"isSucceeded",
|
|
10
|
+
];
|
|
11
|
+
|
|
12
|
+
/**
|
|
13
|
+
* Define the contract between an autonomous/bounded Agent workload and the
|
|
14
|
+
* infrastructure runner. The runner owns isolation, checkpoints and transport;
|
|
15
|
+
* the workload owns the Agent's task, decisions and result semantics.
|
|
16
|
+
*/
|
|
17
|
+
export function defineAgentWorkload(definition) {
|
|
18
|
+
if (!definition || typeof definition !== "object" || Array.isArray(definition)) {
|
|
19
|
+
throw new TypeError("Agent 工作负载定义必须是对象");
|
|
20
|
+
}
|
|
21
|
+
if (typeof definition.workload !== "string" || !definition.workload.trim()) {
|
|
22
|
+
throw new TypeError("Agent 工作负载必须声明 workload");
|
|
23
|
+
}
|
|
24
|
+
validateAgentBoundary(definition.agentBoundary);
|
|
25
|
+
for (const hook of REQUIRED_HOOKS) {
|
|
26
|
+
if (typeof definition[hook] !== "function") {
|
|
27
|
+
throw new TypeError(`Agent 工作负载 ${definition.workload} 缺少 ${hook} hook`);
|
|
28
|
+
}
|
|
29
|
+
}
|
|
30
|
+
return Object.freeze({
|
|
31
|
+
validateScientificComputeDecision: true,
|
|
32
|
+
controllerValidationRetries: 0,
|
|
33
|
+
completionRelativePath: "output/result.json",
|
|
34
|
+
completionHasEvidence: true,
|
|
35
|
+
recoverFinalMessage: true,
|
|
36
|
+
...definition,
|
|
37
|
+
agentBoundary: Object.freeze({
|
|
38
|
+
...definition.agentBoundary,
|
|
39
|
+
decisionRights: Object.freeze([...definition.agentBoundary.decisionRights]),
|
|
40
|
+
immutableConstraints: Object.freeze([...definition.agentBoundary.immutableConstraints]),
|
|
41
|
+
}),
|
|
42
|
+
});
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
export function agentWorkloadMetadata(workload) {
|
|
46
|
+
const definition = defineAgentWorkload(workload);
|
|
47
|
+
return {
|
|
48
|
+
schemaVersion: "1.0",
|
|
49
|
+
workload: definition.workload,
|
|
50
|
+
agentBoundary: {
|
|
51
|
+
mode: definition.agentBoundary.mode,
|
|
52
|
+
decisionRights: [...definition.agentBoundary.decisionRights],
|
|
53
|
+
immutableConstraints: [...definition.agentBoundary.immutableConstraints],
|
|
54
|
+
},
|
|
55
|
+
};
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
function validateAgentBoundary(boundary) {
|
|
59
|
+
if (!boundary || typeof boundary !== "object" || Array.isArray(boundary)) {
|
|
60
|
+
throw new TypeError("Agent 工作负载必须声明 agentBoundary");
|
|
61
|
+
}
|
|
62
|
+
if (!AGENT_MODES.has(boundary.mode)) {
|
|
63
|
+
throw new TypeError("agentBoundary.mode 必须是 autonomous 或 bounded");
|
|
64
|
+
}
|
|
65
|
+
for (const field of ["decisionRights", "immutableConstraints"]) {
|
|
66
|
+
if (
|
|
67
|
+
!Array.isArray(boundary[field])
|
|
68
|
+
|| boundary[field].length === 0
|
|
69
|
+
|| boundary[field].some((item) => typeof item !== "string" || !item.trim())
|
|
70
|
+
) {
|
|
71
|
+
throw new TypeError(`agentBoundary.${field} 必须是非空字符串数组`);
|
|
72
|
+
}
|
|
73
|
+
}
|
|
74
|
+
}
|
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
import path from "node:path";
|
|
2
|
+
import { updateRunState } from "../job.mjs";
|
|
3
|
+
import { resolveResearchPlan } from "../reproduction/plan.mjs";
|
|
4
|
+
import { readAgentUsage } from "../telemetry/usage.mjs";
|
|
5
|
+
import { CiteArkError, readJson } from "../util.mjs";
|
|
6
|
+
import { createAutonomousReproductionWorkload } from "./reproduction.mjs";
|
|
7
|
+
import { defineAgentWorkload } from "./definition.mjs";
|
|
8
|
+
|
|
9
|
+
const handoffInstruction = `This GPU phase resumes the same CPU-prepared workspace. Read output/cpu-preparation.json and progress.md, inspect live work and reuse prepared inputs. Use the GPU for actual scientific execution and necessary device-specific debugging. If a substantial download, build or other CPU-only preparation remains, stop GPU work and request a CPU continuation by writing output/result.json as {"schemaVersion":"2.0","status":"handoff","nextPhase":"cpu","summary":"Concrete remaining CPU work and why the GPU can be released"}. Preserve all raw results and the current plan first; do not keep an idle GPU rented while doing that preparation. A handoff is not a scientific result. Once the actual computation is done, save its raw outputs and finish with the normal completed/partial/failed summary; CPU coordination performs parsing, independent assessment and publication. All phases share the original cumulative budget.`;
|
|
10
|
+
|
|
11
|
+
export function createPhaseAwareReproductionWorkload() {
|
|
12
|
+
const actor = createAutonomousReproductionWorkload();
|
|
13
|
+
return defineAgentWorkload({
|
|
14
|
+
...actor,
|
|
15
|
+
allowCpuHandoff: true,
|
|
16
|
+
completionSuccessJq: `(${actor.completionSuccessJq}) or (.schemaVersion == "2.0" and .status == "handoff" and .nextPhase == "cpu")`,
|
|
17
|
+
prompt: (options) => `${actor.prompt(options)}\n${handoffInstruction}`,
|
|
18
|
+
resumePrompt: (options) => `${actor.resumePrompt(options)}\n${handoffInstruction}`,
|
|
19
|
+
async validate(options) {
|
|
20
|
+
const result = await readJson(options.completionPath).catch(() => null);
|
|
21
|
+
if (result?.status !== "handoff") return actor.validate(options);
|
|
22
|
+
const issues = [];
|
|
23
|
+
if (result.schemaVersion !== "2.0" || result.nextPhase !== "cpu"
|
|
24
|
+
|| typeof result.summary !== "string" || !result.summary.trim()) issues.push("Invalid CPU continuation request");
|
|
25
|
+
try { await resolveResearchPlan(options.bundle.runtimeTask, options.bundle.directories.output); }
|
|
26
|
+
catch (error) { issues.push(error.message); }
|
|
27
|
+
return { result, issues, phaseHandoff: true, protocolVerification: null };
|
|
28
|
+
},
|
|
29
|
+
recover: (options) => options.validation.phaseHandoff ? options.validation : actor.recover(options),
|
|
30
|
+
// Handoffs retain their durable checkpoint; normal terminal outcomes use
|
|
31
|
+
// the existing completion/retention policy.
|
|
32
|
+
isSucceeded: (result, issues) => result?.status !== "handoff" && actor.isSucceeded(result, issues),
|
|
33
|
+
async finalize(options) {
|
|
34
|
+
if (!options.validation.phaseHandoff) return actor.finalize(options);
|
|
35
|
+
const { bundle, validation, checkpoint, attempts, timedOut, cloudExecution } = options;
|
|
36
|
+
if (validation.issues.length || timedOut || checkpoint?.kind !== "full") throw new CiteArkError(
|
|
37
|
+
"GPU 转 CPU 需要有效交接记录和完整检查点", { failureCode: "research.handoff_incomplete", retryable: false });
|
|
38
|
+
const usage = await readAgentUsage(path.join(bundle.directories.execution, "trace.jsonl"), {
|
|
39
|
+
runtimeHome: bundle.directories.runtimeHome, model: bundle.runtimeAgent.model,
|
|
40
|
+
});
|
|
41
|
+
await updateRunState(bundle, { status: "completed", finishedAt: new Date().toISOString(),
|
|
42
|
+
usage, attempts, cloudExecution, phaseHandoff: validation.result });
|
|
43
|
+
return { bundle, checkpoint, phaseHandoff: validation.result };
|
|
44
|
+
},
|
|
45
|
+
});
|
|
46
|
+
}
|
|
@@ -0,0 +1,85 @@
|
|
|
1
|
+
import { readAgentInstructions, ensureFinalWorkspaceCapture } from "../job.mjs";
|
|
2
|
+
import { usesWorkspaceSummary } from "../reproduction/workspace-mode.mjs";
|
|
3
|
+
import { validateResultFile } from "../protocol.mjs";
|
|
4
|
+
import {
|
|
5
|
+
finalizeReproduction,
|
|
6
|
+
recoverInvalidExecutionResult,
|
|
7
|
+
} from "../reproduction/lifecycle.mjs";
|
|
8
|
+
import {
|
|
9
|
+
initialReproductionPrompt,
|
|
10
|
+
retryReproductionPrompt,
|
|
11
|
+
} from "../reproduction/prompts.mjs";
|
|
12
|
+
import {
|
|
13
|
+
infrastructureResumePrompt,
|
|
14
|
+
targetRecoveryPrompt,
|
|
15
|
+
} from "../compute/checkpoint-protocol.mjs";
|
|
16
|
+
import { defineAgentWorkload } from "./definition.mjs";
|
|
17
|
+
|
|
18
|
+
/**
|
|
19
|
+
* The scientific reproduction Agent is intentionally one complete autonomous
|
|
20
|
+
* actor. Platform modules may prepare inputs and verify outputs, but they do not
|
|
21
|
+
* decompose or prescribe the Agent's reasoning and experimental loop.
|
|
22
|
+
*/
|
|
23
|
+
export function createAutonomousReproductionWorkload() {
|
|
24
|
+
return defineAgentWorkload({
|
|
25
|
+
workload: "reproduction",
|
|
26
|
+
agentBoundary: {
|
|
27
|
+
mode: "autonomous",
|
|
28
|
+
decisionRights: [
|
|
29
|
+
"inspect_fixed_research_materials",
|
|
30
|
+
"install_or_replace_task_local_dependencies",
|
|
31
|
+
"plan_and_revise_execution_commands",
|
|
32
|
+
"revise_source_grounded_research_plan_when_enabled",
|
|
33
|
+
"author_and_debug_helper_implementations",
|
|
34
|
+
"run_bounded_probes_and_iterate",
|
|
35
|
+
"acquire_and_verify_public_transports",
|
|
36
|
+
"delegate_within_the_agent_runtime_budget",
|
|
37
|
+
"collect_and_diagnose_execution_evidence",
|
|
38
|
+
],
|
|
39
|
+
immutableConstraints: [
|
|
40
|
+
"original_research_goal_and_input_identity",
|
|
41
|
+
"allocated_compute_and_time_ceiling",
|
|
42
|
+
"sandbox_security_boundary",
|
|
43
|
+
"auditable_evidence_and_result_protocol",
|
|
44
|
+
],
|
|
45
|
+
},
|
|
46
|
+
validateScientificComputeDecision: true,
|
|
47
|
+
completionRelativePath: "output/result.json",
|
|
48
|
+
completionSuccessJq: '.execution.status == "succeeded" or (.schemaVersion == "2.0" and (.status == "completed" or .status == "partial" or .status == "failed"))',
|
|
49
|
+
completionHasEvidence: true,
|
|
50
|
+
recoverFinalMessage: true,
|
|
51
|
+
instructions: (bundle) => readAgentInstructions(bundle.runtimeTask),
|
|
52
|
+
prompt({ bundle, attempt, validation }) {
|
|
53
|
+
return attempt === 1
|
|
54
|
+
? initialReproductionPrompt(bundle)
|
|
55
|
+
: retryReproductionPrompt(validation.issues, bundle.runtimeTask);
|
|
56
|
+
},
|
|
57
|
+
resumePrompt({ bundle, targetRecovery }) {
|
|
58
|
+
return targetRecovery
|
|
59
|
+
? targetRecoveryPrompt(bundle)
|
|
60
|
+
: infrastructureResumePrompt(bundle);
|
|
61
|
+
},
|
|
62
|
+
async validate({ bundle, completionPath }) {
|
|
63
|
+
if (usesWorkspaceSummary(bundle.runtimeTask)) await ensureFinalWorkspaceCapture(bundle);
|
|
64
|
+
return await validateResultFile(
|
|
65
|
+
completionPath,
|
|
66
|
+
bundle.runtimeTask,
|
|
67
|
+
bundle.directories.output,
|
|
68
|
+
);
|
|
69
|
+
},
|
|
70
|
+
async recover({ bundle, validation, attempts, timedOut }) {
|
|
71
|
+
return await recoverInvalidExecutionResult({
|
|
72
|
+
bundle,
|
|
73
|
+
validation,
|
|
74
|
+
attempts,
|
|
75
|
+
timedOut,
|
|
76
|
+
});
|
|
77
|
+
},
|
|
78
|
+
async finalize(options) {
|
|
79
|
+
return await finalizeReproduction(options);
|
|
80
|
+
},
|
|
81
|
+
isSucceeded(result, issues) {
|
|
82
|
+
return issues.length === 0 && result?.execution?.status === "succeeded";
|
|
83
|
+
},
|
|
84
|
+
});
|
|
85
|
+
}
|
|
@@ -0,0 +1,242 @@
|
|
|
1
|
+
import { readSettings, resolveModelProfile, savedModel } from '../settings/store.mjs';
|
|
2
|
+
import { resolveAgentRuntime } from '../runtime/config.mjs';
|
|
3
|
+
import path from 'node:path';
|
|
4
|
+
import { readFileSync } from 'node:fs';
|
|
5
|
+
import { TerminalScreen } from '../terminal/screen.mjs';
|
|
6
|
+
import { PlainSession } from '../terminal/plain.mjs';
|
|
7
|
+
import { withSessionEvents, sessionEvent } from '../terminal/events.mjs';
|
|
8
|
+
import { promptResearchEntry } from './entry.mjs';
|
|
9
|
+
import { readFile, mkdir } from 'node:fs/promises';
|
|
10
|
+
import { prepareResearchInput } from './input.mjs';
|
|
11
|
+
import { planWorkspace, executeWorkspace, ensureSigningKey } from './session.mjs';
|
|
12
|
+
import { createTerminalPrompter, cancelled } from '../deployment/terminal.mjs';
|
|
13
|
+
import { configureLocalModel, readLocalPreferences, writeLocalPreferences } from '../deployment/guided.mjs';
|
|
14
|
+
import { inspectLocalDocker, setupLocalRuntime } from '../deployment/environment.mjs';
|
|
15
|
+
import { captureProcess } from '../process.mjs';
|
|
16
|
+
import { readJson, writeJson, pathExists } from '../util.mjs';
|
|
17
|
+
import { publishLocalCap } from '../deployment/local.mjs';
|
|
18
|
+
|
|
19
|
+
export async function startResearch(argv, helpers) {
|
|
20
|
+
if (!process.env.CITEARK_JOB_WORKER) {
|
|
21
|
+
const settings = await readSettings();
|
|
22
|
+
const i = argv.indexOf('--compute'); const name = i >= 0 ? argv[i + 1] : argv.includes('--resume') ? undefined : settings.defaultCompute;
|
|
23
|
+
const hasInput = ['--paper', '--cap', '--repository-id', '--resume'].some(flag => argv.includes(flag)) || (argv[0] && !argv[0].startsWith('-'));
|
|
24
|
+
if (hasInput && name && settings.compute[name]?.type === 'ssh') return (await import('../remote/command.mjs')).startRemoteResearch(argv, settings.compute[name], name);
|
|
25
|
+
if (argv.includes('--background')) {
|
|
26
|
+
const job = await (await import('./jobs.mjs')).launchJob(argv);
|
|
27
|
+
console.log(JSON.stringify(job)); return job;
|
|
28
|
+
}
|
|
29
|
+
}
|
|
30
|
+
const interactive = process.stdin.isTTY && process.stdout.isTTY && process.env.TERM !== 'dumb'
|
|
31
|
+
&& !argv.some(value => ['--plain', '--non-interactive', '--dry-run'].includes(value));
|
|
32
|
+
const args = argv.filter(value => value !== '--plain');
|
|
33
|
+
if (argv.includes('--dry-run')) return runResearch(args, helpers);
|
|
34
|
+
if (!interactive) {
|
|
35
|
+
const output = new PlainSession();
|
|
36
|
+
try { return await withSessionEvents(event => output.event(event), () => runResearch(args, helpers, null, output)); }
|
|
37
|
+
finally { await output.close(); }
|
|
38
|
+
}
|
|
39
|
+
const version = JSON.parse(readFileSync(new URL('../../package.json', import.meta.url), 'utf8')).version;
|
|
40
|
+
const screen = new TerminalScreen({ version, onExit: async () => {
|
|
41
|
+
screen.event({ kind: 'status', text: 'CLI disconnected. Compute may continue; use the workspace to resume.' });
|
|
42
|
+
await screen.close();
|
|
43
|
+
console.log('CLI disconnected. Compute may continue; use the workspace to resume.');
|
|
44
|
+
process.exit(130);
|
|
45
|
+
} });
|
|
46
|
+
screen.start();
|
|
47
|
+
try {
|
|
48
|
+
const result = await withSessionEvents(event => screen.event(event), () => runResearch(args, helpers, screen, screen));
|
|
49
|
+
if (!screen.closed) await screen.finish(result?.phase === 'completed' ? 'Completed' : 'Plan saved');
|
|
50
|
+
return result;
|
|
51
|
+
} catch (error) {
|
|
52
|
+
screen.event({ kind: 'error', text: error.message });
|
|
53
|
+
if (error.exitCode !== 130) await screen.finish('Stopped');
|
|
54
|
+
throw error;
|
|
55
|
+
} finally { await screen.close(); }
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
async function runResearch(argv, { parseOptions, agentEntryOptions, gcpEntryOptions, autoDlEntryOptions,
|
|
59
|
+
agentRuntimeOptions, executorOptions }, screen = null, output = null) {
|
|
60
|
+
let reference = argv[0] && !argv[0].startsWith('-') ? argv.shift() : null;
|
|
61
|
+
const optionalAgent = Object.fromEntries(Object.entries(agentEntryOptions()).map(([k,v]) => [k, { ...v, required: false }]));
|
|
62
|
+
const options = parseOptions(argv, { ...optionalAgent, ...gcpEntryOptions(), ...autoDlEntryOptions(),
|
|
63
|
+
'--paper': { type: 'string' }, '--repository': { type: 'string' }, '--cap': { type: 'string' },
|
|
64
|
+
'--repository-id': { type: 'string' }, '--run': { type: 'string' }, '--goal': { type: 'string' },
|
|
65
|
+
'--guide': { type: 'string' }, '--experiments': { type: 'string' }, '--work-dir': { type: 'string' },
|
|
66
|
+
'--compute-catalog': { type: 'string' }, '--compute-policy': { type: 'string' }, '--compute-profile': { type: 'string' },
|
|
67
|
+
'--platform': { type: 'string', default: 'https://citeark.com' },
|
|
68
|
+
'--device': { type: 'string', default: 'cpu' }, '--image': { type: 'string' },
|
|
69
|
+
'--resume': { type: 'boolean', default: false }, '--plan-only': { type: 'boolean', default: false },
|
|
70
|
+
'--non-interactive': { type: 'boolean', default: false }, '--dry-run': { type: 'boolean', default: false },
|
|
71
|
+
'--signing-key': { type: 'string' },
|
|
72
|
+
'--model-profile': { type: 'string' }, '--assessment-profile': { type: 'string' }, '--compute': { type: 'string' },
|
|
73
|
+
'--change-model': { type: 'boolean', default: false }, '--background': { type: 'boolean', default: false },
|
|
74
|
+
});
|
|
75
|
+
if (reference && options['--repository-id']) throw Error('Provide a paper ID or --repository-id, not both.');
|
|
76
|
+
if (!['cpu', 'cuda'].includes(options['--device'])) throw Error('--device must be cpu or cuda');
|
|
77
|
+
const dryRun = options['--dry-run'];
|
|
78
|
+
const prompt = !dryRun && !options['--non-interactive'] && process.stdin.isTTY && process.stdout.isTTY
|
|
79
|
+
? screen ?? createTerminalPrompter() : null;
|
|
80
|
+
reference = await promptResearchEntry({ options, reference, prompt });
|
|
81
|
+
if (!reference && !options['--repository-id'] && !options['--paper'] && !options['--cap'] && !options['--resume']) {
|
|
82
|
+
throw Error('Provide --paper, --cap, or a paper ID. Run citeark in a terminal for guided setup.');
|
|
83
|
+
}
|
|
84
|
+
if (options['--resume'] && !options['--work-dir']) throw Error('Provide --work-dir to resume a workspace.');
|
|
85
|
+
const settings = await readSettings();
|
|
86
|
+
const chosenCompute = options['--compute'] ?? (!options['--resume'] ? settings.defaultCompute : undefined);
|
|
87
|
+
if (!process.env.CITEARK_JOB_WORKER && settings.compute[chosenCompute]?.type === 'ssh') {
|
|
88
|
+
if (options['--goal'] === undefined && prompt) options['--goal'] = await prompt.ask('What would you like to investigate? (optional)');
|
|
89
|
+
const remoteArgs = [...(reference ? [reference] : []), ...Object.entries(options).flatMap(([key, value]) => value === undefined || value === false ? [] : value === true ? [key] : [key, String(value)])];
|
|
90
|
+
await screen?.close();
|
|
91
|
+
return (await import('../remote/command.mjs')).startRemoteResearch(remoteArgs, settings.compute[chosenCompute], chosenCompute);
|
|
92
|
+
}
|
|
93
|
+
const root = path.resolve(options['--work-dir'] ?? `citeark-research-${Date.now()}`);
|
|
94
|
+
const sessionPath = path.join(root, 'workspace.json');
|
|
95
|
+
if (!options['--resume'] && await pathExists(root) && !process.env.CITEARK_JOB_WORKER) throw Error('Choose a new workspace directory or use --resume to continue.');
|
|
96
|
+
let saved = options['--resume'] ? await readJson(sessionPath) : null;
|
|
97
|
+
if (saved?.executorSettings) for (const [key, value] of Object.entries(saved.executorSettings)) if (!argv.includes(key)) options[key] = value;
|
|
98
|
+
if (output && saved) await output.attach(path.join(root, 'session.events.jsonl'));
|
|
99
|
+
if (saved?.phase === 'completed') {
|
|
100
|
+
console.log(`Research completed: ${saved.artifacts.join('\n')}`);
|
|
101
|
+
return saved;
|
|
102
|
+
}
|
|
103
|
+
if (options['--resume'] && [reference, options['--paper'], options['--cap'], options['--goal'], options['--repository']].some(Boolean)) {
|
|
104
|
+
throw Error('Resuming uses the saved sources and goal. Create a new workspace to change them.');
|
|
105
|
+
}
|
|
106
|
+
await mkdir(root, { recursive: true });
|
|
107
|
+
if (output && !saved) await output.attach(path.join(root, 'session.events.jsonl'));
|
|
108
|
+
if (screen) screen.context = root;
|
|
109
|
+
sessionEvent({ kind: 'stage', stage: 'intake', status: 'Running' });
|
|
110
|
+
let goal = options['--goal'];
|
|
111
|
+
if (!saved && goal === undefined && prompt) goal = await prompt.ask('What would you like to investigate? (optional)');
|
|
112
|
+
const input = saved?.input ?? await prepareResearchInput({ directory: path.join(root, 'input'),
|
|
113
|
+
paper: options['--paper'], repository: options['--repository'], capPath: options['--cap'],
|
|
114
|
+
reference: options['--repository-id'] ?? reference, baseUrl: options['--platform'], runId: options['--run'],
|
|
115
|
+
goal, guide: options['--guide'] ? await readFile(options['--guide'], 'utf8') : '' });
|
|
116
|
+
if (screen) screen.context = `${input.title ?? 'Research'} · ${root}`;
|
|
117
|
+
if (saved && ['--compute', '--compute-catalog', '--compute-policy', '--compute-profile', '--image'].some(key => options[key])) throw Error('Resuming uses the saved compute configuration. Start a new workspace to select different resources.');
|
|
118
|
+
const computeName = options['--compute'] ?? (!saved ? settings.defaultCompute : undefined);
|
|
119
|
+
const compute = computeName ? settings.compute[computeName] : null;
|
|
120
|
+
if (computeName && !compute) throw Error(`Unknown compute profile: ${computeName}`);
|
|
121
|
+
if (compute?.type === 'ssh') throw Error('Use the SSH session entry to start research on this server.');
|
|
122
|
+
if (compute?.type === 'catalog' && !saved) {
|
|
123
|
+
options['--compute-catalog'] ??= compute.catalog; options['--compute-policy'] ??= compute.policy; options['--compute-profile'] ??= compute.profile;
|
|
124
|
+
}
|
|
125
|
+
if (compute?.type === 'local') { options['--device'] = compute.device; options['--image'] ??= compute.image; }
|
|
126
|
+
let catalogPath = saved?.computeCatalogPath ?? options['--compute-catalog'];
|
|
127
|
+
if (!catalogPath && prompt) catalogPath = await prompt.ask('Compute configuration file (leave empty for local Docker)');
|
|
128
|
+
const profile = saved?.computeProfile ?? options['--compute-profile'];
|
|
129
|
+
const policyPath = saved?.computePolicyPath ?? options['--compute-policy'];
|
|
130
|
+
if (!catalogPath && !dryRun) {
|
|
131
|
+
const host = await inspectLocalDocker();
|
|
132
|
+
const gpu = options['--device'] === 'cuda';
|
|
133
|
+
if (gpu && (host.hostPlatform === 'darwin' || host.architecture !== 'amd64')) throw Error('CUDA is unavailable on this machine. Select remote compute or a CPU experiment.');
|
|
134
|
+
const image = options['--image'] ?? `citeark-agent/runtime:${gpu ? 'cuda' : 'cpu'}`;
|
|
135
|
+
const available = await captureProcess('docker', ['image', 'inspect', image]);
|
|
136
|
+
if (available.code) {
|
|
137
|
+
if (!prompt || !await prompt.confirm('Prepare the local research environment?')) throw Error(`Run citeark setup first${gpu ? ' --gpu' : ''}`);
|
|
138
|
+
await setupLocalRuntime({ gpu });
|
|
139
|
+
}
|
|
140
|
+
const catalog = { schemaVersion: '0.1', profiles: [{ id: 'user-local', label: 'Local Docker', executor: 'local-docker',
|
|
141
|
+
accelerator: gpu ? 'gpu' : 'cpu', image, cpuCores: host.cpus, memoryGb: Math.floor(host.memoryGb), hourlyUsd: 0, available: true }] };
|
|
142
|
+
if (gpu) {
|
|
143
|
+
const { checkLocalExecution } = await import('../deployment/environment.mjs');
|
|
144
|
+
const check = await checkLocalExecution({ environment: { image, gpu: 'all', cpus: 1, memoryGb: 1 }, compute: {} });
|
|
145
|
+
const gpus = check.observed.gpus;
|
|
146
|
+
catalog.profiles[0].gpu = { type: gpus[0].name, count: gpus.length, vramGb: Math.floor(Math.min(...gpus.map(g => g.memoryGb))) };
|
|
147
|
+
}
|
|
148
|
+
catalogPath = path.join(root, 'compute.json');
|
|
149
|
+
await writeJson(catalogPath, catalog);
|
|
150
|
+
}
|
|
151
|
+
catalogPath = catalogPath ? path.resolve(catalogPath) : undefined;
|
|
152
|
+
let policy = policyPath ? path.resolve(policyPath) : undefined;
|
|
153
|
+
if (!saved) {
|
|
154
|
+
if (catalogPath) { const catalog = await readJson(catalogPath); catalogPath = path.join(root, 'compute.json'); await writeJson(catalogPath, catalog); }
|
|
155
|
+
if (policy) { const limits = await readJson(policy); policy = path.join(root, 'compute-policy.json'); await writeJson(policy, limits); }
|
|
156
|
+
}
|
|
157
|
+
if (!saved) {
|
|
158
|
+
const task = await readJson(input.compileTaskPath);
|
|
159
|
+
if (catalogPath) {
|
|
160
|
+
const catalog = await readJson(catalogPath);
|
|
161
|
+
const limits = policy ? await readJson(policy) : {};
|
|
162
|
+
task.computeContext = { profiles: catalog.profiles, limits: { maximumCost: limits.maxEstimatedComputeUsd ?? null }, computePolicy: limits };
|
|
163
|
+
}
|
|
164
|
+
if (catalogPath) {
|
|
165
|
+
const catalog = await readJson(catalogPath);
|
|
166
|
+
const compiler = catalog.profiles.find(item => item.id === profile)
|
|
167
|
+
?? catalog.profiles.find(item => item.available !== false && item.accelerator === 'cpu')
|
|
168
|
+
?? catalog.profiles.find(item => item.available !== false);
|
|
169
|
+
if (compiler?.executor === 'local-docker' && compiler.image) task.environment.image = compiler.image;
|
|
170
|
+
}
|
|
171
|
+
if (options['--image']) task.environment.image = options['--image'];
|
|
172
|
+
await writeJson(input.compileTaskPath, task);
|
|
173
|
+
}
|
|
174
|
+
const overridingModel = ['--model-profile', '--assessment-profile', '--agent', '--model', '--api-base-url', '--effort', '--subagent-model', '--model-budget'].some(k => options[k]);
|
|
175
|
+
if (saved?.modelConfiguration && overridingModel && !options['--change-model']) throw Error('Resuming uses the saved model. Add --change-model to record an explicit model change.');
|
|
176
|
+
const selected = saved?.modelConfiguration && !options['--change-model'] ? saved.modelConfiguration
|
|
177
|
+
: resolveModelProfile(settings, options['--model-profile']) ?? await readLocalPreferences();
|
|
178
|
+
const keyEnvironment = options['--api-key-env'] ?? selected?.apiKeyEnv;
|
|
179
|
+
const explicit = { ...selected, runtime: options['--agent'] ?? selected?.runtime,
|
|
180
|
+
apiBaseUrl: options['--api-base-url'] ?? selected?.apiBaseUrl ?? process.env.CITEARK_MODEL_API_URL,
|
|
181
|
+
model: options['--model'] ?? selected?.model ?? process.env.CITEARK_MODEL,
|
|
182
|
+
apiKeyEnv: keyEnvironment, apiKey: options['--api-key'] ?? (keyEnvironment ? process.env[keyEnvironment] : selected?.apiKey ?? process.env.CITEARK_MODEL_API_KEY),
|
|
183
|
+
apiKeySource: keyEnvironment ? `env:${keyEnvironment}` : options['--api-key'] ? 'parameter' : process.env.CITEARK_MODEL_API_KEY ? 'env:CITEARK_MODEL_API_KEY' : 'local-session',
|
|
184
|
+
subagentModel: options['--subagent-model'] ?? selected?.subagentModel,
|
|
185
|
+
effort: options['--effort'] ?? selected?.effort,
|
|
186
|
+
maxBudgetUsd: options['--model-budget'] ? Number(options['--model-budget']) : selected?.maxBudgetUsd };
|
|
187
|
+
const runtime = prompt ? await configureLocalModel({ options: explicit, saved: selected ?? {}, prompt })
|
|
188
|
+
: { ...explicit, runtime: explicit.runtime ?? 'opencode', apiBaseUrl: explicit.apiBaseUrl ?? (dryRun ? 'https://api.invalid' : undefined), model: explicit.model ?? (dryRun ? 'dry-run' : undefined) };
|
|
189
|
+
resolveAgentRuntime({ ...runtime, requireApi: !dryRun && !options['--gcp-provider-secret'] && !process.env.CITEARK_GCP_PROVIDER_SECRET });
|
|
190
|
+
if (prompt) await writeLocalPreferences(runtime);
|
|
191
|
+
const assessmentConfig = options['--assessment-profile'] ? resolveModelProfile(settings, options['--assessment-profile']) : saved?.assessmentConfiguration;
|
|
192
|
+
const assessmentRuntime = assessmentConfig ? { ...assessmentConfig, apiKey: assessmentConfig.apiKeyEnv ? process.env[assessmentConfig.apiKeyEnv] : assessmentConfig.apiKey } : runtime;
|
|
193
|
+
if (assessmentConfig) resolveAgentRuntime({ ...assessmentRuntime, requireApi: !dryRun });
|
|
194
|
+
const previous = saved ?? { schemaVersion: '1.0', input, computeCatalogPath: catalogPath, computePolicyPath: policy, computeProfile: profile, phase: 'planning', artifacts: [] };
|
|
195
|
+
if (options['--change-model'] && saved?.modelConfiguration) previous.modelChanges = [...(previous.modelChanges ?? []), { at: new Date().toISOString(), previous: saved.modelConfiguration, next: savedModel(runtime) }];
|
|
196
|
+
previous.executorSettings = Object.fromEntries(Object.entries(options).filter(([key, value]) => value !== undefined && (key.startsWith('--gcp-') || key.startsWith('--autodl-') || key === '--allow-remote-model-key')));
|
|
197
|
+
previous.modelConfiguration = savedModel(runtime);
|
|
198
|
+
if (assessmentConfig) previous.assessmentConfiguration = savedModel(assessmentRuntime);
|
|
199
|
+
await writeJson(sessionPath, previous);
|
|
200
|
+
const signingKeyPath = dryRun ? options['--signing-key'] : await ensureSigningKey(options['--signing-key']);
|
|
201
|
+
if (screen) screen.context += ` · ${runtime.model}`;
|
|
202
|
+
if (screen && !process.env.CITEARK_JOB_WORKER) {
|
|
203
|
+
const { interactiveResearch } = await import('./interactive.mjs');
|
|
204
|
+
const completed = await interactiveResearch({ root, options, screen, runtime, assessmentRuntime, signingKeyPath });
|
|
205
|
+
if (completed?.phase === 'completed') await offerPublish(completed, { prompt, input, options });
|
|
206
|
+
return completed;
|
|
207
|
+
}
|
|
208
|
+
sessionEvent({ kind: 'stage', stage: 'research_compilation', status: 'Running' });
|
|
209
|
+
const planned = await planWorkspace({ directory: root, input, agentRuntime: runtime,
|
|
210
|
+
computeCatalogPath: catalogPath, computePolicyPath: policy, computeProfile: profile,
|
|
211
|
+
executorOptions: executorOptions(options), signingKeyPath, dryRun, ...(output ? { progress: () => {} } : {}) });
|
|
212
|
+
if (dryRun) { console.log(`Research inputs prepared (dry run): ${root}`); return planned; }
|
|
213
|
+
sessionEvent({ kind: 'stage', stage: 'awaiting_selection', status: 'Waiting for experiment selection' });
|
|
214
|
+
const experiments = planned.research.experiments;
|
|
215
|
+
console.log(`\nWorkspace: ${root}\nAvailable experiments:`);
|
|
216
|
+
experiments.forEach((item, index) => console.log(`${index + 1}. ${item.id} — ${item.title}`));
|
|
217
|
+
let experimentIds = options['--experiments']?.split(',').map(id => id.trim()).filter(Boolean);
|
|
218
|
+
if (!planned.session.targets && !experimentIds?.length && !options['--plan-only'] && prompt) {
|
|
219
|
+
const selected = await prompt.ask('Select experiment numbers (comma-separated; leave empty to save the plan)');
|
|
220
|
+
experimentIds = selected ? selected.split(',').map(value => experiments[Number(value.trim()) - 1]?.id ?? value.trim()) : [];
|
|
221
|
+
}
|
|
222
|
+
if (options['--plan-only'] || (!planned.session.targets && !experimentIds?.length)) {
|
|
223
|
+
console.log(`Plan saved. Continue with citeark start --resume --work-dir ${JSON.stringify(root)} --experiments <experiment-id>`);
|
|
224
|
+
return planned.session;
|
|
225
|
+
}
|
|
226
|
+
if (prompt && !await prompt.confirm('Run the selected experiments?')) throw cancelled();
|
|
227
|
+
const completed = await executeWorkspace({ directory: root, experimentIds, agentRuntime: runtime, assessmentRuntime,
|
|
228
|
+
executorOptions: executorOptions(options), signingKeyPath, ...(output ? { progress: () => {} } : {}) });
|
|
229
|
+
console.log(`Signed research artifacts saved locally:\n${completed.artifacts.join('\n')}`);
|
|
230
|
+
await offerPublish(completed, { prompt, input, options });
|
|
231
|
+
sessionEvent({ kind: 'stage', stage: 'completed', status: 'Completed' });
|
|
232
|
+
return completed;
|
|
233
|
+
}
|
|
234
|
+
|
|
235
|
+
async function offerPublish(completed, { prompt, input, options }) {
|
|
236
|
+
if (prompt && await prompt.confirm('Upload this plan and its results to CiteArk?')) {
|
|
237
|
+
const token = process.env.CITEARK_API_KEY ?? await prompt.ask('CiteArk API key (hidden)', { secret: true });
|
|
238
|
+
const visibility = input.platformRepositoryId && await prompt.confirm('Publish as a public community result for this paper?') ? 'public' : 'private';
|
|
239
|
+
for (const capPath of completed.artifacts) console.log(JSON.stringify(await publishLocalCap({ capPath,
|
|
240
|
+
repositoryId: input.platformRepositoryId, baseUrl: options['--platform'], token, visibility })));
|
|
241
|
+
}
|
|
242
|
+
}
|
|
@@ -0,0 +1,49 @@
|
|
|
1
|
+
import { AsyncLocalStorage } from "node:async_hooks";
|
|
2
|
+
const context = new AsyncLocalStorage();
|
|
3
|
+
export const withResearchControl = (control, action) =>
|
|
4
|
+
context.run(control, action);
|
|
5
|
+
export const researchInstructions = async () =>
|
|
6
|
+
(await context.getStore()?.instructions?.()) ?? "";
|
|
7
|
+
export const recordResearchCleanup = async (record) => {
|
|
8
|
+
await context.getStore()?.cleanup?.(record);
|
|
9
|
+
};
|
|
10
|
+
export const researchSignal = () => context.getStore()?.signal;
|
|
11
|
+
export async function researchBoundary(stage) {
|
|
12
|
+
await context.getStore()?.boundary?.(stage);
|
|
13
|
+
assertResearchActive();
|
|
14
|
+
}
|
|
15
|
+
export function assertResearchActive() {
|
|
16
|
+
if (researchSignal()?.aborted) throw researchSignal().reason;
|
|
17
|
+
}
|
|
18
|
+
export function cancellationError() {
|
|
19
|
+
const error = Error("Research cancelled by the user.");
|
|
20
|
+
error.exitCode = 130;
|
|
21
|
+
error.processingLeaseLost = true;
|
|
22
|
+
error.researchCancelled = true;
|
|
23
|
+
return error;
|
|
24
|
+
}
|
|
25
|
+
export function controlledExecutors(options = {}) {
|
|
26
|
+
if (!context.getStore()) return options;
|
|
27
|
+
return {
|
|
28
|
+
...options,
|
|
29
|
+
...Object.fromEntries(
|
|
30
|
+
["gcpBatch", "autoDl"].map((key) => [
|
|
31
|
+
key,
|
|
32
|
+
{
|
|
33
|
+
...options[key],
|
|
34
|
+
assertActive: async () => {
|
|
35
|
+
assertResearchActive();
|
|
36
|
+
await options[key]?.assertActive?.();
|
|
37
|
+
},
|
|
38
|
+
...(key === "gcpBatch"
|
|
39
|
+
? { keepJob: false }
|
|
40
|
+
: {
|
|
41
|
+
keepInstance: false,
|
|
42
|
+
keepFailedInstance: false,
|
|
43
|
+
releaseOnFailure: true,
|
|
44
|
+
}),
|
|
45
|
+
},
|
|
46
|
+
]),
|
|
47
|
+
),
|
|
48
|
+
};
|
|
49
|
+
}
|
|
@@ -0,0 +1,28 @@
|
|
|
1
|
+
// Collect missing entry information before entering the existing research flow.
|
|
2
|
+
export async function promptResearchEntry({ options, reference, prompt }) {
|
|
3
|
+
if (!prompt || reference || ['--paper', '--cap', '--repository-id', '--resume'].some(key => options[key])) return reference;
|
|
4
|
+
if (prompt.openResearch) return prompt.openResearch(options);
|
|
5
|
+
const choice = prompt.choose ? await prompt.choose('Start a research session', [
|
|
6
|
+
{ value: '1', label: 'Open a paper' }, { value: '2', label: 'Open a CiteArk paper' },
|
|
7
|
+
{ value: '3', label: 'Use a CAP artifact' }, { value: '4', label: 'Resume a workspace' },
|
|
8
|
+
]) : await prompt.ask('Start research: 1 Paper; 2 CiteArk ID; 3 CAP artifact; 4 Resume', { defaultValue: '1' });
|
|
9
|
+
if (choice === '4') {
|
|
10
|
+
options['--resume'] = true;
|
|
11
|
+
options['--work-dir'] = await prompt.ask('Workspace directory');
|
|
12
|
+
return reference;
|
|
13
|
+
}
|
|
14
|
+
if (choice === '2') return await prompt.ask('CiteArk paper ID');
|
|
15
|
+
if (choice === '1') {
|
|
16
|
+
options['--paper'] = await prompt.ask('Paper file path or URL');
|
|
17
|
+
if (!options['--paper']) throw Error('Enter a paper file path or URL.');
|
|
18
|
+
} else if (choice === '3') {
|
|
19
|
+
options['--cap'] = await prompt.ask('CAP file path');
|
|
20
|
+
if (!options['--cap']) throw Error('Enter a CAP file path.');
|
|
21
|
+
options['--paper'] = await prompt.ask('Paper file or URL (leave empty to use the CAP source)') || undefined;
|
|
22
|
+
} else {
|
|
23
|
+
throw Error('Choose 1, 2, 3, or 4.');
|
|
24
|
+
}
|
|
25
|
+
options['--repository'] ??= await prompt.ask('Code directory or repository URL (optional)') || undefined;
|
|
26
|
+
options['--guide'] ??= await prompt.ask('Research guide file (optional)') || undefined;
|
|
27
|
+
return reference;
|
|
28
|
+
}
|
|
@@ -0,0 +1,93 @@
|
|
|
1
|
+
import path from 'node:path';
|
|
2
|
+
import { mkdir, writeFile } from 'node:fs/promises';
|
|
3
|
+
import { readCapArchive, readCapBlob } from '../cap/v2/read.mjs';
|
|
4
|
+
import { validateCodeFiles } from '../deployment/handoff.mjs';
|
|
5
|
+
import { normalizeResearchStructure } from '../research/structure.mjs';
|
|
6
|
+
import { sha256Value, sha256File, writeJson } from '../util.mjs';
|
|
7
|
+
|
|
8
|
+
const source = value => /^https?:\/\//.test(value) ? { url: value } : { path: path.resolve(value) };
|
|
9
|
+
|
|
10
|
+
export async function prepareResearchInput({ directory, paper, repository, capPath, reference,
|
|
11
|
+
baseUrl = 'https://citeark.com', runId, goal = '', guide = '', token = process.env.CITEARK_API_KEY, fetchImpl = fetch }) {
|
|
12
|
+
await mkdir(directory, { recursive: true });
|
|
13
|
+
const platformRequest = url => fetchImpl(url, token && new URL(url).origin === new URL(baseUrl).origin ? { headers: { authorization: `Bearer ${token}` } } : undefined);
|
|
14
|
+
let platformRepositoryId = null;
|
|
15
|
+
let title = 'Independent research';
|
|
16
|
+
let platformPaper = null;
|
|
17
|
+
if (reference) {
|
|
18
|
+
const response = await platformRequest(new URL(`/api/v1/repositories/${encodeURIComponent(reference)}/deployment`, baseUrl));
|
|
19
|
+
if (!response.ok) throw Error(`Unable to read paper details: HTTP ${response.status}`);
|
|
20
|
+
const { data } = await response.json();
|
|
21
|
+
platformRepositoryId = data.repositoryId;
|
|
22
|
+
title = data.title;
|
|
23
|
+
platformPaper = data.researchInput?.paper ?? null;
|
|
24
|
+
paper ??= platformPaper?.url ? new URL(platformPaper.url, baseUrl).href : undefined;
|
|
25
|
+
repository ??= data.researchInput?.repository?.url;
|
|
26
|
+
const previous = runId ? data.choices?.find(item => item.runId === runId) : data.choices?.[0];
|
|
27
|
+
if (runId && !previous) throw Error('The selected run has no readable reference CAP.');
|
|
28
|
+
if (previous && !capPath) {
|
|
29
|
+
const download = await platformRequest(new URL(previous.archiveUrl, baseUrl));
|
|
30
|
+
if (!download.ok) throw Error(`Unable to download reference CAP: HTTP ${download.status}`);
|
|
31
|
+
capPath = path.join(directory, 'reference.cap');
|
|
32
|
+
await writeFile(capPath, Buffer.from(await download.arrayBuffer()));
|
|
33
|
+
}
|
|
34
|
+
}
|
|
35
|
+
let inventory = null;
|
|
36
|
+
let materials = null;
|
|
37
|
+
let artifact = null;
|
|
38
|
+
if (capPath) {
|
|
39
|
+
artifact = await readCapArchive(capPath, path.join(directory, 'reference-artifact'));
|
|
40
|
+
const handoff = await readCapBlob(artifact, 'deployment-handoff');
|
|
41
|
+
const plan = await readCapBlob(artifact, 'research-plan-input');
|
|
42
|
+
inventory = await readCapBlob(artifact, 'research-inventory')
|
|
43
|
+
?? plan?.sourceInventory?.research ?? handoff?.research?.sourceInventory?.research
|
|
44
|
+
?? artifact.records.sourceWork[0]?.citeark?.sourceInventory?.research ?? null;
|
|
45
|
+
const work = artifact.records.sourceWork[0];
|
|
46
|
+
title = work?.title ?? title;
|
|
47
|
+
// A received archive may suggest public sources, never paths on this host.
|
|
48
|
+
paper ??= work?.sources?.find(item => item.kind === 'paper' && /^https?:\/\//.test(item.uri))?.uri;
|
|
49
|
+
repository ??= work?.sources?.find(item => item.kind === 'repository' && /^https?:\/\//.test(item.uri))?.uri;
|
|
50
|
+
materials = { schemaVersion: '1.0', sourceArtifactDigest: artifact.artifactDigest,
|
|
51
|
+
instructions: handoff?.instructions ?? '', files: handoff ? validateCodeFiles(handoff.files) : [],
|
|
52
|
+
usage: 'Historical reference only. Choose and plan this research independently; collect fresh observations.' };
|
|
53
|
+
if (handoff?.instructions) guide = [guide, handoff.instructions].filter(Boolean).join('\n\n');
|
|
54
|
+
}
|
|
55
|
+
if (!paper) throw Error('Provide a paper file or URL. This CAP has no accessible paper source.');
|
|
56
|
+
let paperSource = { ...source(paper), title };
|
|
57
|
+
if (reference && paperSource.url && new URL(paperSource.url).origin === new URL(baseUrl).origin) {
|
|
58
|
+
const response = await platformRequest(paperSource.url);
|
|
59
|
+
if (!response.ok) throw Error(`Unable to read the paper: HTTP ${response.status}`);
|
|
60
|
+
const filename = path.join(directory, 'paper.pdf');
|
|
61
|
+
await writeFile(filename, Buffer.from(await response.arrayBuffer()));
|
|
62
|
+
if (platformPaper?.digest && `sha256:${await sha256File(filename)}` !== platformPaper.digest) throw Error('The downloaded paper does not match its source digest.');
|
|
63
|
+
paperSource = { path: filename, sourceUrl: paperSource.url, title };
|
|
64
|
+
}
|
|
65
|
+
const repositorySource = repository ? source(repository) : undefined;
|
|
66
|
+
const retainedRepository = inventory?.provenance?.sourceIdentity?.repository;
|
|
67
|
+
if (repositorySource?.url && retainedRepository?.url === repositorySource.url && retainedRepository.commit) {
|
|
68
|
+
repositorySource.commit = retainedRepository.commit;
|
|
69
|
+
}
|
|
70
|
+
const task = { schemaVersion: '0.1', name: 'independent-research', paper: paperSource,
|
|
71
|
+
...(repositorySource ? { repository: repositorySource } : {}),
|
|
72
|
+
researchGoal: goal, reproductionGuide: guide,
|
|
73
|
+
compilationStage: inventory ? 'execution_plan' : 'inventory',
|
|
74
|
+
environment: { image: 'citeark-agent/runtime:cpu', cpus: 2, memoryGb: 4, timeoutMinutes: 45, gpu: 'none' } };
|
|
75
|
+
if (inventory) {
|
|
76
|
+
inventory = normalizeResearchStructure(inventory);
|
|
77
|
+
if (inventory.compilationStage !== 'inventory') throw Error('The CAP does not contain a source inventory.');
|
|
78
|
+
const inventoryPath = path.join(directory, 'inventory.json');
|
|
79
|
+
await writeJson(inventoryPath, inventory);
|
|
80
|
+
task.sourceInventory = { path: inventoryPath, digest: `sha256:${sha256Value(inventory)}` };
|
|
81
|
+
}
|
|
82
|
+
const referenceMaterialsPath = materials ? path.join(directory, 'reference-materials.json') : null;
|
|
83
|
+
if (materials) {
|
|
84
|
+
await writeJson(referenceMaterialsPath, materials);
|
|
85
|
+
task.referenceMaterialsPath = referenceMaterialsPath;
|
|
86
|
+
}
|
|
87
|
+
const compileTaskPath = path.join(directory, 'task.json');
|
|
88
|
+
await writeJson(compileTaskPath, task);
|
|
89
|
+
return { compileTaskPath, referenceMaterialsPath, platformRepositoryId, title,
|
|
90
|
+
sourceArtifactDigests: artifact ? [artifact.artifactDigest] : [], paper: paperSource,
|
|
91
|
+
sourceCompilationDigests: artifact?.kind === 'research-compilation' ? [artifact.artifactDigest] : [] };
|
|
92
|
+
}
|
|
93
|
+
|