@citeark/agent 0.3.20
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +202 -0
- package/README.md +128 -0
- package/data/dataset-source-registry.v1.json +300 -0
- package/dist/arkgraph/boot.js +6 -0
- package/dist/arkgraph/index.html +1 -0
- package/dist/arkgraph/viewer.css +1 -0
- package/dist/arkgraph/viewer.en.css +1 -0
- package/dist/arkgraph/viewer.en.js +49 -0
- package/dist/arkgraph/viewer.en.js.LEGAL.txt +56 -0
- package/dist/arkgraph/viewer.js +49 -0
- package/dist/arkgraph/viewer.js.LEGAL.txt +56 -0
- package/docker/claude-code/Dockerfile +97 -0
- package/docker/claude-code/codex-pro-relay.mjs +466 -0
- package/docker/claude-code/runtime-contract-check.mjs +79 -0
- package/docs/arkgraph-reading.md +79 -0
- package/docs/configuration.md +100 -0
- package/docs/integration.md +92 -0
- package/docs/maturity-plan.md +27 -0
- package/docs/npm-release.md +44 -0
- package/docs/paper-reading.md +40 -0
- package/docs/research-plan-granularity.md +27 -0
- package/docs/terminal.md +49 -0
- package/examples/toy-evaluation/compile-task.json +27 -0
- package/examples/toy-evaluation/paper.md +5 -0
- package/examples/toy-evaluation/repository/README.md +9 -0
- package/examples/toy-evaluation/repository/checkpoint.json +4 -0
- package/examples/toy-evaluation/repository/evaluate.py +17 -0
- package/examples/toy-evaluation/task.json +81 -0
- package/package.json +59 -0
- package/prompts/compile-research.md +58 -0
- package/prompts/execute-contract.md +72 -0
- package/prompts/execute-workspace-simple.md +51 -0
- package/prompts/execute-workspace.md +34 -0
- package/prompts/prepare-reproduction.md +82 -0
- package/prompts/repair-research.md +45 -0
- package/protocol/CAP.md +129 -0
- package/protocol/LICENSE +12 -0
- package/protocol/MAPPINGS.md +72 -0
- package/protocol/README.md +38 -0
- package/protocol/conformance-v2.0-alpha.1.json +36 -0
- package/protocol/examples/arkgraph/checkpoint-evaluation.json +309 -0
- package/protocol/examples/arkgraph/fixtures.mjs +49 -0
- package/protocol/examples/arkgraph/paper-free.json +291 -0
- package/protocol/examples/arkgraph/partial-failure.json +344 -0
- package/protocol/examples/arkgraph/training-evaluation.json +443 -0
- package/protocol/profiles/agent-trace.md +16 -0
- package/protocol/profiles/computational-run.md +16 -0
- package/protocol/profiles/core.md +15 -0
- package/protocol/profiles/public-bundle.md +18 -0
- package/protocol/profiles/reproduction.md +29 -0
- package/protocol/profiles/research-compilation.md +44 -0
- package/protocol/profiles/research-plan.md +39 -0
- package/protocol/profiles/restricted-evidence.md +15 -0
- package/runtime/bootstrap-autodl-runtime.sh +314 -0
- package/runtime/create-runtime-venv.sh +41 -0
- package/runtime/install-local-cpu-runtime.sh +23 -0
- package/runtime/install-scientific-runtime.sh +153 -0
- package/runtime/mineru/parse.py +62 -0
- package/runtime/mineru/requirements.txt +4 -0
- package/runtime/requirements-baseline.txt +38 -0
- package/schemas/cap/v2/activity.schema.json +47 -0
- package/schemas/cap/v2/agent.schema.json +32 -0
- package/schemas/cap/v2/assertion.schema.json +110 -0
- package/schemas/cap/v2/descriptor.schema.json +243 -0
- package/schemas/cap/v2/entity.schema.json +64 -0
- package/schemas/cap/v2/manifest.schema.json +67 -0
- package/schemas/cap/v2/relation.schema.json +82 -0
- package/schemas/compute-catalog.schema.json +63 -0
- package/schemas/compute-decision.schema.json +27 -0
- package/schemas/execution-contract.schema.json +1024 -0
- package/schemas/research-card.schema.json +30 -0
- package/schemas/research-inventory-draft.schema.json +366 -0
- package/schemas/research.schema.json +1044 -0
- package/schemas/result.schema.json +173 -0
- package/schemas/verification-policy.schema.json +47 -0
- package/schemas/verified-conclusion.schema.json +58 -0
- package/schemas/workspace-summary.schema.json +24 -0
- package/scripts/build-arkgraph-view.mjs +12 -0
- package/scripts/check-execution-feasibility.mjs +24 -0
- package/scripts/check-syntax.mjs +15 -0
- package/scripts/deterministic-asset-preparation.py +438 -0
- package/scripts/package-cap.mjs +23 -0
- package/scripts/package-local-agent.mjs +23 -0
- package/scripts/preview-arkgraph.mjs +25 -0
- package/scripts/replay-research-compiler-candidate.mjs +134 -0
- package/scripts/review-compiler-sources.mjs +44 -0
- package/scripts/run-asset-preparation.sh +17 -0
- package/scripts/run-research-plan.mjs +98 -0
- package/scripts/validate-asset-preparation.py +290 -0
- package/scripts/verify-local-runtime.mjs +57 -0
- package/scripts/verify-npm-package.mjs +57 -0
- package/src/adapters/paper2agent.mjs +107 -0
- package/src/assets/cache.mjs +159 -0
- package/src/assets/compute.mjs +98 -0
- package/src/assets/executor.mjs +145 -0
- package/src/assets/lifecycle.mjs +213 -0
- package/src/assets/manifest.mjs +242 -0
- package/src/assets/opportunistic-preparation.mjs +81 -0
- package/src/assets/plan.mjs +411 -0
- package/src/assets/prompts.mjs +29 -0
- package/src/assets/public-asset-probe.mjs +525 -0
- package/src/assets/qualification.mjs +119 -0
- package/src/assets/readiness.mjs +130 -0
- package/src/assets/reproduction-admission.mjs +355 -0
- package/src/assets/requirements.mjs +152 -0
- package/src/assets/source-grounding.mjs +341 -0
- package/src/assets/source-policy.mjs +118 -0
- package/src/autodl/client.mjs +260 -0
- package/src/autodl/ssh.mjs +380 -0
- package/src/autodl/tools.mjs +129 -0
- package/src/cap/redaction.mjs +38 -0
- package/src/cap/v2/archive.mjs +152 -0
- package/src/cap/v2/attestation.mjs +204 -0
- package/src/cap/v2/canonical-json.mjs +114 -0
- package/src/cap/v2/compilation-artifact.mjs +240 -0
- package/src/cap/v2/core.mjs +282 -0
- package/src/cap/v2/measurement-assessment-records.mjs +23 -0
- package/src/cap/v2/pipeline-artifact.mjs +922 -0
- package/src/cap/v2/read.mjs +41 -0
- package/src/cap/v2/reassessment-artifact.mjs +383 -0
- package/src/cap/v2/research-artifact.mjs +231 -0
- package/src/cap/v2/research-map-records.mjs +46 -0
- package/src/cap/v2/research-object-records.mjs +163 -0
- package/src/cap/v2/research-records.mjs +187 -0
- package/src/cap/v2/verify.mjs +642 -0
- package/src/cli.mjs +1146 -0
- package/src/compute/autodl-pro-compiler.mjs +347 -0
- package/src/compute/autodl-pro-executor.mjs +459 -0
- package/src/compute/autodl-pro-job.mjs +843 -0
- package/src/compute/autodl-pro-network.mjs +295 -0
- package/src/compute/autodl-pro-remote.mjs +810 -0
- package/src/compute/autodl-pro-staging.mjs +117 -0
- package/src/compute/campaign.mjs +110 -0
- package/src/compute/catalog.mjs +123 -0
- package/src/compute/checkpoint-protocol.mjs +154 -0
- package/src/compute/codex-account-lock.mjs +111 -0
- package/src/compute/codex-account-session.mjs +107 -0
- package/src/compute/compiler-profile.mjs +38 -0
- package/src/compute/compiler-router.mjs +23 -0
- package/src/compute/coordinator-recovery.mjs +210 -0
- package/src/compute/executor-router.mjs +29 -0
- package/src/compute/gcp-batch-compiler.mjs +685 -0
- package/src/compute/gcp-batch-executor.mjs +1215 -0
- package/src/compute/gcp-batch-failure.mjs +92 -0
- package/src/compute/gcp-batch-job.mjs +527 -0
- package/src/compute/gcp-batch-lifecycle.mjs +81 -0
- package/src/compute/gcp-checkpoint-worker.mjs +1633 -0
- package/src/compute/local-codex-compiler.mjs +52 -0
- package/src/compute/measurement-hardware.mjs +128 -0
- package/src/compute/remote-attempt.mjs +226 -0
- package/src/compute/requirements.mjs +124 -0
- package/src/compute/research-phases.mjs +48 -0
- package/src/compute/scheduler.mjs +452 -0
- package/src/compute/shared-workloads.mjs +26 -0
- package/src/compute/stage-archive.mjs +79 -0
- package/src/contracts/campaign-contract.mjs +52 -0
- package/src/contracts/execution-contract.mjs +819 -0
- package/src/contracts/execution-mode.mjs +19 -0
- package/src/contracts/execution-timeouts.mjs +45 -0
- package/src/contracts/execution-workload.mjs +68 -0
- package/src/contracts/preflight-schema.mjs +25 -0
- package/src/contracts/public-contract.mjs +63 -0
- package/src/contracts/subject-tags.mjs +31 -0
- package/src/dashboard/data.mjs +898 -0
- package/src/dashboard/server.mjs +79 -0
- package/src/dashboard/static/dashboard.css +366 -0
- package/src/dashboard/static/dashboard.js +560 -0
- package/src/dashboard/static/index.html +85 -0
- package/src/deployment/community-policy.mjs +9 -0
- package/src/deployment/environment.mjs +112 -0
- package/src/deployment/guided.mjs +98 -0
- package/src/deployment/handoff.mjs +102 -0
- package/src/deployment/local-contract.mjs +31 -0
- package/src/deployment/local.mjs +100 -0
- package/src/deployment/prepare.mjs +46 -0
- package/src/deployment/recipe.mjs +108 -0
- package/src/deployment/supplement.mjs +51 -0
- package/src/deployment/terminal.mjs +43 -0
- package/src/diagnosis/renderer.mjs +75 -0
- package/src/diagnosis/target-failure.mjs +46 -0
- package/src/evidence/parser-registry.mjs +54 -0
- package/src/evidence/parsers/fasttext-classification.mjs +82 -0
- package/src/evidence/parsers/json-scalar.mjs +96 -0
- package/src/evidence/parsers/simcse-senteval.mjs +104 -0
- package/src/evidence/parsers/starspace-classification.mjs +78 -0
- package/src/evidence/registry.mjs +147 -0
- package/src/execution/runner-audit.mjs +473 -0
- package/src/gcp/auth.mjs +106 -0
- package/src/gcp/batch-client.mjs +120 -0
- package/src/gcp/resource-discovery.mjs +177 -0
- package/src/gcp/rest.mjs +82 -0
- package/src/gcp/secret-manager.mjs +34 -0
- package/src/gcp/signed-url.mjs +133 -0
- package/src/gcp/storage.mjs +220 -0
- package/src/graph/command.mjs +41 -0
- package/src/graph/execution.mjs +97 -0
- package/src/graph/model.mjs +37 -0
- package/src/graph/presentation.mjs +110 -0
- package/src/graph/query.mjs +159 -0
- package/src/graph/research-relations.mjs +69 -0
- package/src/graph/source-page.mjs +12 -0
- package/src/graph/source-preview.mjs +34 -0
- package/src/graph/validate.mjs +76 -0
- package/src/job.mjs +496 -0
- package/src/network/autodl-routing-proxy.mjs +462 -0
- package/src/network/egress-proxy.mjs +158 -0
- package/src/observability/event-contract.mjs +230 -0
- package/src/observability/pipeline-monitor.mjs +166 -0
- package/src/pipeline/orchestrator.mjs +1281 -0
- package/src/pipeline/recovery-error.mjs +11 -0
- package/src/pipeline/replay.mjs +304 -0
- package/src/pipeline/shared-execution.mjs +115 -0
- package/src/pipeline/stage-checkpoint.mjs +86 -0
- package/src/pipeline/stage-recovery.mjs +101 -0
- package/src/pipeline/targets.mjs +110 -0
- package/src/process.mjs +143 -0
- package/src/protocol.mjs +312 -0
- package/src/provider/codex-account.mjs +44 -0
- package/src/provider/codex-completion.mjs +49 -0
- package/src/provider/completion.mjs +292 -0
- package/src/provider/model-client.mjs +44 -0
- package/src/provider/model-route.mjs +29 -0
- package/src/provider/openrouter-readiness.mjs +189 -0
- package/src/provider/reader-bridge.mjs +35 -0
- package/src/provider/relay.mjs +263 -0
- package/src/provider/runtime-auth.mjs +40 -0
- package/src/public/cap.d.mts +90 -0
- package/src/public/cap.mjs +12 -0
- package/src/public/contracts.d.mts +2 -0
- package/src/public/host.mjs +171 -0
- package/src/public/operations.d.mts +11 -0
- package/src/public/presentation.d.mts +4 -0
- package/src/records/views.mjs +26 -0
- package/src/remote/command.mjs +178 -0
- package/src/remote/ssh.mjs +59 -0
- package/src/repository-origin.mjs +81 -0
- package/src/reproduction/evidence-feedback.mjs +96 -0
- package/src/reproduction/incomplete-initialization.mjs +25 -0
- package/src/reproduction/lifecycle.mjs +253 -0
- package/src/reproduction/plan.mjs +132 -0
- package/src/reproduction/prompts.mjs +70 -0
- package/src/reproduction/runner.mjs +188 -0
- package/src/reproduction/summary.mjs +130 -0
- package/src/reproduction/workspace-mode.mjs +7 -0
- package/src/research/automatic-admission.mjs +156 -0
- package/src/research/compiler-coverage.mjs +85 -0
- package/src/research/compiler-failure.mjs +24 -0
- package/src/research/compiler-normalization-guards.mjs +112 -0
- package/src/research/compiler-repair.mjs +3 -0
- package/src/research/compiler.mjs +853 -0
- package/src/research/continuation-selection.mjs +26 -0
- package/src/research/execution-graph-context.mjs +43 -0
- package/src/research/experiment-importance.mjs +15 -0
- package/src/research/inventory-handoff.mjs +104 -0
- package/src/research/inventory-revisions.mjs +32 -0
- package/src/research/mineru-local.mjs +73 -0
- package/src/research/paper-command.mjs +19 -0
- package/src/research/paper-markdown.mjs +180 -0
- package/src/research/paper-source-map.mjs +69 -0
- package/src/research/planning-policy.mjs +88 -0
- package/src/research/reference-materials.mjs +11 -0
- package/src/research/reproduction-scope.mjs +30 -0
- package/src/research/research-map.mjs +94 -0
- package/src/research/research-objects.mjs +88 -0
- package/src/research/source-discovery.mjs +646 -0
- package/src/research/source-observations.mjs +75 -0
- package/src/research/source-review-cli-mcp.mjs +26 -0
- package/src/research/source-review-input.mjs +209 -0
- package/src/research/source-review-local-codex.mjs +36 -0
- package/src/research/source-review-model.mjs +70 -0
- package/src/research/source-review.mjs +173 -0
- package/src/research/structure.mjs +3163 -0
- package/src/research-card/renderer.mjs +277 -0
- package/src/research-card/verified-conclusion.mjs +143 -0
- package/src/results/output-registry.mjs +183 -0
- package/src/runtime/claude-code.mjs +52 -0
- package/src/runtime/codex-capacity-retry.mjs +87 -0
- package/src/runtime/codex.mjs +64 -0
- package/src/runtime/config.mjs +157 -0
- package/src/runtime/final-output.mjs +40 -0
- package/src/runtime/index.mjs +21 -0
- package/src/runtime/local-codex.mjs +74 -0
- package/src/runtime/opencode.mjs +95 -0
- package/src/runtime/prompt.mjs +13 -0
- package/src/sandbox/docker.mjs +363 -0
- package/src/settings/command.mjs +297 -0
- package/src/settings/store.mjs +119 -0
- package/src/telemetry/pricing.mjs +68 -0
- package/src/telemetry/usage.mjs +265 -0
- package/src/terminal/events.mjs +97 -0
- package/src/terminal/input.mjs +40 -0
- package/src/terminal/plain.mjs +40 -0
- package/src/terminal/remote-stream.mjs +22 -0
- package/src/terminal/screen.mjs +214 -0
- package/src/terminal/transcript.mjs +69 -0
- package/src/util.mjs +107 -0
- package/src/verification/ai-assessor.mjs +534 -0
- package/src/verification/claim-evaluator.mjs +242 -0
- package/src/verification/evidence-context.mjs +165 -0
- package/src/verification/evidence-reader.mjs +95 -0
- package/src/verification/integrity.mjs +570 -0
- package/src/verification/tolerance.mjs +32 -0
- package/src/workloads/cpu-research-preparation.mjs +56 -0
- package/src/workloads/definition.mjs +74 -0
- package/src/workloads/phase-aware-reproduction.mjs +46 -0
- package/src/workloads/reproduction.mjs +85 -0
- package/src/workspace/command.mjs +242 -0
- package/src/workspace/control.mjs +49 -0
- package/src/workspace/entry.mjs +28 -0
- package/src/workspace/input.mjs +93 -0
- package/src/workspace/interactive.mjs +94 -0
- package/src/workspace/jobs.mjs +418 -0
- package/src/workspace/session.mjs +97 -0
- package/src/workspace/worker.mjs +137 -0
- package/ui/arkgraph/ambient-motion.mjs +10 -0
- package/ui/arkgraph/app.jsx +153 -0
- package/ui/arkgraph/boot.js +6 -0
- package/ui/arkgraph/camera-motion.mjs +20 -0
- package/ui/arkgraph/context-reveal.mjs +39 -0
- package/ui/arkgraph/details.css +3 -0
- package/ui/arkgraph/entry.jsx +28 -0
- package/ui/arkgraph/experiment-curves.mjs +17 -0
- package/ui/arkgraph/experiment-selection.mjs +15 -0
- package/ui/arkgraph/experiment-style.css +26 -0
- package/ui/arkgraph/experiment-ui.jsx +32 -0
- package/ui/arkgraph/frame.html +1 -0
- package/ui/arkgraph/graph-gestures.mjs +62 -0
- package/ui/arkgraph/label-layout.mjs +57 -0
- package/ui/arkgraph/locales/en.json +229 -0
- package/ui/arkgraph/locales/source-types.json +15 -0
- package/ui/arkgraph/localization-build.mjs +27 -0
- package/ui/arkgraph/material-build.mjs +23 -0
- package/ui/arkgraph/material-colors.mjs +39 -0
- package/ui/arkgraph/material-style.css +15 -0
- package/ui/arkgraph/open-graph.jsx +326 -0
- package/ui/arkgraph/outline.jsx +49 -0
- package/ui/arkgraph/package-lock.json +888 -0
- package/ui/arkgraph/package.json +17 -0
- package/ui/arkgraph/reading-layout.mjs +130 -0
- package/ui/arkgraph/reading-presentation.mjs +73 -0
- package/ui/arkgraph/record-detail.css +51 -0
- package/ui/arkgraph/record-details.jsx +29 -0
- package/ui/arkgraph/research-types.mjs +31 -0
- package/ui/arkgraph/selection-mark.jsx +6 -0
- package/ui/arkgraph/soft-spine.mjs +26 -0
- package/ui/arkgraph/steering-style.css +187 -0
- package/ui/arkgraph/style.css +272 -0
|
@@ -0,0 +1,79 @@
|
|
|
1
|
+
import assert from "node:assert/strict";
|
|
2
|
+
import { execFileSync } from "node:child_process";
|
|
3
|
+
|
|
4
|
+
import { createCodexProRelay } from "./codex-pro-relay.mjs";
|
|
5
|
+
|
|
6
|
+
const uvVersion = execFileSync("uv", ["--version"], { encoding: "utf8" }).trim();
|
|
7
|
+
const pythonVersion = execFileSync("python3.12", ["--version"], { encoding: "utf8" }).trim();
|
|
8
|
+
const codexVersion = execFileSync("codex", ["--version"], { encoding: "utf8" }).trim();
|
|
9
|
+
assert.match(uvVersion, /^uv 0\./);
|
|
10
|
+
assert.match(pythonVersion, /^Python 3\.12\./);
|
|
11
|
+
assert.match(codexVersion, /^codex-cli 0\.159\.1$/, "Scientific runtime must use the validated Codex version");
|
|
12
|
+
|
|
13
|
+
process.env.CITEARK_PROVIDER_UPSTREAM_BASE_URL = "https://provider.invalid/api";
|
|
14
|
+
process.env.CITEARK_API_KEY = "citeark-runtime-contract-check";
|
|
15
|
+
process.env.CITEARK_CODEX_REASONING_MODE = "pro";
|
|
16
|
+
delete process.env.CITEARK_CODEX_UPSTREAM_BASE_URL;
|
|
17
|
+
delete process.env.OPENAI_API_KEY;
|
|
18
|
+
|
|
19
|
+
const observed = [];
|
|
20
|
+
const relay = createCodexProRelay({
|
|
21
|
+
fetchImpl: async (url, options) => {
|
|
22
|
+
observed.push({
|
|
23
|
+
url: String(url),
|
|
24
|
+
authorization: options.headers.get("authorization"),
|
|
25
|
+
body: JSON.parse(Buffer.from(options.body).toString("utf8")),
|
|
26
|
+
});
|
|
27
|
+
return new Response('{"ok":true}', {
|
|
28
|
+
status: 200,
|
|
29
|
+
headers: { "content-type": "application/json" },
|
|
30
|
+
});
|
|
31
|
+
},
|
|
32
|
+
});
|
|
33
|
+
|
|
34
|
+
await listen(relay);
|
|
35
|
+
const address = relay.address();
|
|
36
|
+
assert.ok(address && typeof address !== "string");
|
|
37
|
+
|
|
38
|
+
try {
|
|
39
|
+
const health = await fetch(`http://127.0.0.1:${address.port}/health`);
|
|
40
|
+
assert.equal(health.status, 200);
|
|
41
|
+
|
|
42
|
+
for (const path of ["/v1/responses", "/v1/chat/completions"]) {
|
|
43
|
+
const response = await fetch(`http://127.0.0.1:${address.port}${path}`, {
|
|
44
|
+
method: "POST",
|
|
45
|
+
headers: { "content-type": "application/json" },
|
|
46
|
+
body: JSON.stringify({
|
|
47
|
+
model: "contract-check",
|
|
48
|
+
reasoning: { effort: "high" },
|
|
49
|
+
}),
|
|
50
|
+
});
|
|
51
|
+
assert.equal(response.status, 200);
|
|
52
|
+
}
|
|
53
|
+
} finally {
|
|
54
|
+
await close(relay);
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
assert.deepEqual(observed.map((request) => request.url), [
|
|
58
|
+
"https://provider.invalid/api/v1/responses",
|
|
59
|
+
"https://provider.invalid/api/v1/chat/completions",
|
|
60
|
+
]);
|
|
61
|
+
for (const request of observed) {
|
|
62
|
+
assert.equal(request.authorization, "Bearer citeark-runtime-contract-check");
|
|
63
|
+
assert.deepEqual(request.body.reasoning, { effort: "high", mode: "pro" });
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
console.log("CiteArk runtime contract check passed");
|
|
67
|
+
|
|
68
|
+
function listen(server) {
|
|
69
|
+
return new Promise((resolve, reject) => {
|
|
70
|
+
server.once("error", reject);
|
|
71
|
+
server.listen(0, "127.0.0.1", resolve);
|
|
72
|
+
});
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
function close(server) {
|
|
76
|
+
return new Promise((resolve, reject) => {
|
|
77
|
+
server.close((error) => error ? reject(error) : resolve());
|
|
78
|
+
});
|
|
79
|
+
}
|
|
@@ -0,0 +1,79 @@
|
|
|
1
|
+
# ArkGraph 研究概览与实验选择
|
|
2
|
+
|
|
3
|
+
Agent 的来源理解产物保存完整科研图,并用独立的 `reading` 元数据表达阅读层级。通用界面延续 context-34 原型:横向柔性主线、自由分支、基于 Material 色板的主线渐变与五组内容分类、世界坐标中的标签避让、缩放、缓动、稳定微动、按需显示证据和完整大纲。原型中的论文节点编号、中文文案、三条实验路线与手工分类已从通用组件移除。
|
|
4
|
+
|
|
5
|
+
## 科研内容与呈现
|
|
6
|
+
|
|
7
|
+
十二种阅读类别是问题、概念、假说/前提、主张、方法、方案、活动、观测/结果、论证/证明、评估、资源和主体。它们映射到 CAP 的五种基础类型,不要求每篇论文包含全部类别。
|
|
8
|
+
|
|
9
|
+
`researchObjects` 新增 question、concept、premise、argument、research-plan、research-activity、source-assessment、person、organization、instrument 角色;来源定位必填。仪器与硬件是 Entity 资源,人物与组织是 Agent 主体。论文叙述的活动使用 Activity,但明确为 declared、unknown 和固定来源上下文;不得伪装成本次执行。原文的评述是 source-assessment 的文本声明,没有本次复现的科学裁决。假说、独立主张和测量集合沿用既有字段。
|
|
10
|
+
|
|
11
|
+
`researchRelations` 使用显式类型化引用(如 `object:method-1`、`claim:finding-1`),记录有来源的科学关系。关系进入 CAP 时绑定端点、来源与归属的确切内容摘要。该数组不重复现有 objectIds、observationId 等字段已声明的关系,也不保存阅读顺序。
|
|
12
|
+
|
|
13
|
+
`reading` 包含 schemaVersion 1.0、overview、mainline、branches、labels。输入标签用 `target` 保存局部标识;组装 CAP 后才转换为全局 `ref`,保留的原始清单不会被误当成 CAP 引用。短标题不代替完整说明;分支成员可以不在默认画布显示。主线无固定节点数;每个展示子节点只有一个展示父节点,科学关系仍可多对多。标签、分支和主线引用必须存在,循环与重复归属明确报错。准备阶段保留来源关系与阅读信息。
|
|
14
|
+
|
|
15
|
+
在 CAP 中,来源论文的 `citeark.reading` 保存上述编辑组织并引用已组装的科学对象;坐标和动画不进入记录。`buildArkGraphPresentation` 只消费授权、验签后、范围完整且版本明确的图,保留全部对象与关系及其原始内容,另给出主线、分支、临时证据组和实验入口。缺失外部依赖保留为不可用关系;不能把截断结果描述为完整。
|
|
16
|
+
|
|
17
|
+
## 面向读者的表述
|
|
18
|
+
|
|
19
|
+
来源理解提示词要求标题、概述、说明、主张、条件、局限性和关系注释直接表达研究事实及必要的证据边界。具体缺失、原文矛盾和适用范围保留在受影响对象上;提取过程、编号管理和“没有擅自推断/没有制造结论”等自我辩护留在内部工作记录。没有具体局限性时允许字段为空,不为每个对象补通用免责声明;尚未查到的信息不能写成论文未报告。实验规划中的问题、比较、实验目标、执行说明、工作量说明、处置理由与限制沿用同一写作原则。数据完整性要求直接写成可检查的验收条件,同时保留执行所需的条件和细节。
|
|
20
|
+
|
|
21
|
+
例如,编号关系需要解释时,直接写“表格列出 980 条测试,正文描述 100 条,两者对应关系未说明”;最佳层结论受原文矛盾影响时,说明矛盾所在及结论范围。文风调整不得删除真实不确定性、改变数值与科学含义,也不得直接覆盖已有签名 CAP;新提示词用于后续生成。
|
|
22
|
+
|
|
23
|
+
## 图上的实验
|
|
24
|
+
|
|
25
|
+
实验仍是 procedure 包。规划器新增 researchIntent:问题、比较、判据、起止来源引用和上下文;终点必须属于该实验明确检验的主张。呈现器通过真实 plannedFor、requires、hasStep 关系读取目标、输入、步骤和依赖。连线是实验的交互入口,不能替代实际科研合同,也不能把普通 about 关系直接当成执行。
|
|
26
|
+
|
|
27
|
+
选择记录为 `{artifactDigest, procedures:[{ref,digest}]}`,复用网站已有确认接口格式。必要实验依赖闭合,共享步骤按稳定身份展示一次。独立预览只保存本地选择;资源准入、费用报价和执行确认属于宿主,预览没有自动执行入口。
|
|
28
|
+
|
|
29
|
+
## 本地预览与集成
|
|
30
|
+
|
|
31
|
+
```sh
|
|
32
|
+
npm run build:arkgraph
|
|
33
|
+
node src/cli.mjs graph --cap /path/plan.cap --operation presentation
|
|
34
|
+
node scripts/preview-arkgraph.mjs /path/plan.cap /path/preview /path/paper.pdf
|
|
35
|
+
```
|
|
36
|
+
|
|
37
|
+
将输出目录由本地 HTTP 服务提供。预览读取实际签名 CAP,使用 `ui/arkgraph` 中可复用的 React/Force Graph/D3 组件。轻量 CAP 包提供纯呈现读模型和类型声明,科研引擎不依赖网站源码。
|
|
38
|
+
|
|
39
|
+
原文定位支持英文 `p. 8`、`page 8` 及中文“物理页 8”“第 8 页”。明确的单页定位生成原文页预览;多页范围可跳转到起始页,但不据此生成某一页作为图表预览。
|
|
40
|
+
|
|
41
|
+
当前工作阶段:生产者、校验、CAP 组装、纯读模型及本地通用组件已实现,已完成两篇真实论文的本地读取与签名产物预览验收。网站现有生产组件、费用确认和结果回填尚未切换。真实试跑不启动实验,不代表科学结果已获独立验证。
|
|
42
|
+
|
|
43
|
+
2026-10-05 首次验收:区域引导向量论文的清单包含 6 个主张、47 条数值记录;准备阶段形成 6 个绑定真实 procedure 的候选实验,计划共 170 个对象,默认显示 24 个。联合选择两条拒答实验时,12 个必要步骤中有 6 个共享步骤,界面按身份去重,并可保存、恢复及清空本地选择。Generalization Dynamics 清单包含 23 个主张、278 条数值记录及 41 个观测集合,196 行矩阵保持为单个集合;217 个对象全部保留,9 个主线节点和直接分支默认显示。原文含混条件与冲突保留为待澄清项,未编造复现结果。
|
|
44
|
+
|
|
45
|
+
此轮开发中,先生成的真实草稿经过有记录的类型/引用转换才完成打包:仪器归入 Entity;原始阅读标签的局部 `ref` 改名为 `target`;已有显式 `basedOn` 的前提断言从不适用的 Entity 简写中移除重复引用。没有改写数值或科学陈述。生产提示与校验器已同步修正,签名清单转计划的回归覆盖该问题;原始草稿与转换记录随本地验收产物保留。
|
|
46
|
+
|
|
47
|
+
软件回归覆盖全部十二类别、120 行观测集合不拆节点、签名清单/计划归档往返、精确实验绑定、非法端点、展示循环、来源保留与准备阶段禁止覆盖。软件夹具不能替代真实论文的提取质量审查。
|
|
48
|
+
|
|
49
|
+
## 阅读布局
|
|
50
|
+
|
|
51
|
+
2026-10-05 补充:节点圆点与完整标题占位共同参与 D3 碰撞。每个主节点的阅读分支先按实际文字尺寸安排,再按内容量分配横向间距;邻组可利用上下空隙柔性避让,不为每组划一个互不相交的大矩形。分支放置也计入连线穿越文字的代价。标签优先使用节点旁已预留的位置;中英文标题按词边界均衡换行,避免末行单字。
|
|
52
|
+
|
|
53
|
+
主线保留柔性的横向趋势,节点拖动后更新休止位置而不固定坐标。默认全局重心力被移除,避免与阅读位置约束冲突。所有占位与避让都在固定图坐标中完成;平移、缩放和静止微动不重新安排阅读组。显示数量及科研内容不因本次布局优化而减少。
|
|
54
|
+
|
|
55
|
+
## 阅读控制与大纲分组
|
|
56
|
+
|
|
57
|
+
2026-10-05 界面整理:查看全图、实验选择和大纲入口合并到左上论文卡片;窄窗口查看详情时卡片保留精简控制栏。画布不再显示固定拖动提示。
|
|
58
|
+
|
|
59
|
+
底部分类和大纲对象筛选统一为问题与概念、观点与解释、方法与过程、结果与评价、材料与来源五组。分组只作用于阅读界面,原始十二类别和详情子类型保留。点击底部一组同时弱化或恢复该组全部类别。主线颜色按阅读次序从红、橙、黄、绿、青、蓝到紫过渡,按主节点数量在 Material 色板之间插值;分支沿用内容类别配色。主线编号按背景亮度选择黑白文字,保持对比度。
|
|
60
|
+
|
|
61
|
+
大纲递归保留主线下的展示分支;其余对象改为“按需查阅”,按原文图表与评述、补充概念与解释、方法与活动细节、模型/数据/材料、后续研究与实验计划、论文来源与参与者分组折叠,只显示本篇存在的分组。材料进一步区分模型与检查点、数据与提示、材料与工具;计划进一步区分研究方案、计划步骤、预期材料与产出。尚未执行的预期结果归入计划,避免与原文报告的结果混排;组内标题按含数字的自然顺序排序。全部对象和关系仍可搜索、查看及下载。
|
|
62
|
+
|
|
63
|
+
两篇本地真实论文的阅读分组核对分别完整覆盖 143/143、217/217 个对象,没有重复;原科研数据保持不变。此项调整覆盖六节点与九节点主线、分类弱化/恢复、大纲搜索和窄窗口详情控制栏。
|
|
64
|
+
|
|
65
|
+
## 新提示词的真实再生成验收
|
|
66
|
+
|
|
67
|
+
2026-10-05 使用面向读者的新提示词重新读取同一篇区域引导向量论文,再从新清单生成实验方案。新清单含 7 个主张、5 个假说和 43 条数值记录;原先核对的 47 处原文报告全部覆盖,其中图 4 与表 2 重复的四处共用记录并保留两处定位。清单为 89 个对象;计划为 143 个对象、728 条关系,6 个主线节点、24 个默认可见对象。
|
|
68
|
+
|
|
69
|
+
新计划为五个实验包:中间层拒答比较、完整 23 层拒答扫描、中文/日文抑制、中间层与深层语言对照、多轮语言自述。图 3 的同提示对话并入中间层拒答包;43 条数值均有精确实验绑定。两个拒答包联合选择时共有 9 个必要步骤,其中 5 个共享;浏览器保存、刷新恢复和清空通过。缺少原文判据的定性问题与语言计量限制仍保留。
|
|
70
|
+
|
|
71
|
+
原始清单完整保存在计划的 sourceInventory 中;主张文字、数值、条件、限制、来源对象、来源关系和阅读组织均保留。中间层语言问题通过有记录的 inventoryRevisions 增加表 2、图 4 的深层对照关联,没有新增中间层测量。生成后的限定模型文案复核修订一条清单说明及五个计划说明字段,记录前后文本后重新编译、签名和读取;不将这些修订描述为第一次草稿直接通过。
|
|
72
|
+
|
|
73
|
+
本次还修复中文物理页码定位,8 项图呈现/布局回归、源码检查和界面构建通过。原本地预览地址已更新;没有执行实验、申请资源、发布包或部署网站。初始独立来源评审仍为 deferred。
|
|
74
|
+
|
|
75
|
+
## 网站共用呈现
|
|
76
|
+
|
|
77
|
+
网站固定本次 Agent 包中的中英文界面,在同源隔离画布中呈现完整签名快照的只读投影;构建时从已校验归档提取界面字节,按包摘要使用独立地址。`presentation` 查询指定一个精确制品,拒绝截断或缺页,不把阅读连线变为科学关系。历史产物没有阅读组织时保留其主张与原始对象,不重新提取或改写签名记录。全部对象、关系及数值可在大纲搜索并下载。原文预览依赖现有来源地址与明确物理页码;本地提供匹配签名摘要的 PDF 时继续呈现已校验页面图片。
|
|
78
|
+
|
|
79
|
+
实验选择窗口的选择、可选集合和锁定状态由网站控制,画布只发出点选意图。网站沿用精确计划摘要、程序引用与摘要、依赖闭包、可用性、资源报价及最终费用确认。页面中的本地草案不会启动实验;实际执行仍通过原确认入口。网站默认英文,中文路由使用中文界面;不在展示阶段调用模型翻译科学记录。
|
|
@@ -0,0 +1,100 @@
|
|
|
1
|
+
# Models, compute and server sessions
|
|
2
|
+
|
|
3
|
+
CiteArk Agent can run on your computer, your Linux server or a CiteArk server. Each uses the same paper understanding, planning, experiment execution, evidence assessment and signed CAP workflow. The website adds accounts, scheduling, access control, billing and result pages. Independent research needs no CiteArk account.
|
|
4
|
+
|
|
5
|
+
## Model profiles
|
|
6
|
+
|
|
7
|
+
```sh
|
|
8
|
+
citeark model add research --agent opencode \
|
|
9
|
+
--api-base-url https://openrouter.ai/api/v1 \
|
|
10
|
+
--model <model-id> --api-key-env RESEARCH_API_KEY
|
|
11
|
+
citeark model use research
|
|
12
|
+
citeark model list
|
|
13
|
+
citeark model check research
|
|
14
|
+
```
|
|
15
|
+
|
|
16
|
+
Set the named environment variable in your shell or your server's service environment. Configuration stores the variable name, never the key. `model check` makes up to three small, billable requests to check tool calling and structured output; it does not run experiments. The check does not certify every capability of the provider or a model's scientific quality.
|
|
17
|
+
|
|
18
|
+
The runtime selects the protocol: `opencode` uses Chat Completions, `codex` uses Responses, and `claude-code` uses Messages. Source review and scientific assessment use these protocols too. Explicit API-mode OpenRouter reviews use its common Chat Completions endpoint. Hosted configurations with `codexAccountId` use the selected ChatGPT subscription for source review, handoff and independent assessment through the shared model client; they never resolve an API key or fall back to API billing. Subscription failures stop for recovery of the same account. Choose an endpoint and model that support the selected protocol, tool calling and structured responses. See the provider references for [tool calling](https://developers.openai.com/api/docs/guides/function-calling) and [Anthropic structured output](https://platform.claude.com/docs/en/build-with-claude/structured-outputs).
|
|
19
|
+
|
|
20
|
+
`--model-profile research` selects a profile per study. `--assessment-profile reviewer` selects a separate scientific assessor. Research workspaces snapshot model settings; changing your default profile does not change a resumed study. To deliberately change its model, use `--resume --change-model --model-profile <name>`; the change is recorded. API keys are resolved from the environment again on restart.
|
|
21
|
+
|
|
22
|
+
`--effort` sets the runtime's reasoning effort. `--model-budget` sets its per-runtime model budget; enforcement depends on the runtime. It is **not** a study-wide model or compute spending cap. Compute policy limits and hosted paper budget admission have separate roles. Standalone direct-provider calls do not currently enforce a cumulative financial cap across compilation, reviews and assessment.
|
|
23
|
+
|
|
24
|
+
## Compute profiles
|
|
25
|
+
|
|
26
|
+
```sh
|
|
27
|
+
citeark compute add laptop --type local --device cpu
|
|
28
|
+
citeark compute add gpu --type local --device cuda
|
|
29
|
+
citeark compute add cloud --type catalog --catalog ./compute.json --policy ./policy.json
|
|
30
|
+
citeark compute use laptop
|
|
31
|
+
citeark compute check laptop
|
|
32
|
+
```
|
|
33
|
+
|
|
34
|
+
Local execution uses Docker; run `citeark setup` once, or `citeark setup --gpu` on a compatible NVIDIA Linux host. Catalogs describe local Docker, GCP Batch or AutoDL capacity. Their contents and policy are copied into each workspace. `--compute <name>` selects a saved configuration. Model and compute selection are independent.
|
|
35
|
+
|
|
36
|
+
## A server you own
|
|
37
|
+
|
|
38
|
+
First install on the Linux server and prepare its runtime:
|
|
39
|
+
|
|
40
|
+
```sh
|
|
41
|
+
npm install -g https://citeark.co/cli.tgz
|
|
42
|
+
citeark setup # use --gpu for NVIDIA execution
|
|
43
|
+
citeark model add research # configure the model on this server
|
|
44
|
+
citeark doctor
|
|
45
|
+
mkdir -p /home/research/citeark
|
|
46
|
+
```
|
|
47
|
+
|
|
48
|
+
On your laptop, use an existing SSH configuration alias. Connect once with ordinary `ssh` to verify the host key, then save its profile:
|
|
49
|
+
|
|
50
|
+
```sh
|
|
51
|
+
citeark compute add lab --type ssh --host my-lab \
|
|
52
|
+
--directory /home/research/citeark
|
|
53
|
+
citeark compute check lab
|
|
54
|
+
citeark start --paper ./paper.pdf --compute lab
|
|
55
|
+
```
|
|
56
|
+
|
|
57
|
+
The remote shell must find `citeark`; use `--command /absolute/path/to/citeark` when needed. Node.js must also be available in that shell. Optional `--identity`, `--port` and `--ssh-config` select normal SSH settings. Files explicitly provided through `--paper`, `--cap`, `--repository` or `--guide` are transferred to the new workspace. Code transfers omit `.git`, `.env*`, `node_modules` and `.citeark`. URLs remain URLs. Remote signing keys and model credentials are configured on the server; laptop model keys are not forwarded. Model profile names and catalog paths in a remote command refer to the server.
|
|
58
|
+
|
|
59
|
+
The command starts a detached remote worker and attaches its transcript. `--background` starts it without attaching. The returned absolute workspace path is used for subsequent operations:
|
|
60
|
+
|
|
61
|
+
```sh
|
|
62
|
+
citeark session attach /home/research/citeark/research-... --compute lab
|
|
63
|
+
citeark session status /home/research/citeark/research-... --compute lab
|
|
64
|
+
citeark start --resume --work-dir /home/research/citeark/research-... \
|
|
65
|
+
--compute lab --experiments <experiment-id>
|
|
66
|
+
citeark session fetch /home/research/citeark/research-... --compute lab --output ./results
|
|
67
|
+
```
|
|
68
|
+
|
|
69
|
+
Planning stops for experiment selection. Fetch downloads CAP artifacts and verifies their content and signatures. It does not mirror datasets or model weights. `--dry-run --compute lab` previews the target without contacting it; `compute check lab` performs a real connection check.
|
|
70
|
+
|
|
71
|
+
CiteArk's own Linux servers use these same commands. Platform workers instead call the fixed-version `@citeark/agent/host` API and keep platform queues and leases outside this package. SSH support does not require a second research engine or a CiteArk platform login.
|
|
72
|
+
|
|
73
|
+
## Background operation and recovery
|
|
74
|
+
|
|
75
|
+
```sh
|
|
76
|
+
citeark start --paper ./paper.pdf --background --work-dir ./study
|
|
77
|
+
citeark session list
|
|
78
|
+
citeark session attach ./study
|
|
79
|
+
citeark session pause ./study
|
|
80
|
+
citeark session resume ./study
|
|
81
|
+
citeark session note ./study --message 'Check whether batch size explains the discrepancy.'
|
|
82
|
+
citeark session cancel ./study
|
|
83
|
+
citeark session status ./study
|
|
84
|
+
```
|
|
85
|
+
|
|
86
|
+
Interactive research also uses a background worker. Noninteractive foreground commands remain foreground processes; add `--background` to survive a disconnected terminal. A background worker inherits the launcher environment. Keep model credentials available for a later restart. A key entered in an interactive prompt exists only in that worker's environment; resuming after it exits requires supplying it again.
|
|
87
|
+
|
|
88
|
+
Pause takes effect at a pipeline stage boundary. Notes apply at the next runtime invocation and do not bypass selected experiment or resource constraints. Cancellation stops the current task's container or requests release of Agent-created cloud resources; existing SSH servers are never shut down. Inspect `resourceCleanup` in session status, especially failed cleanup entries. Cloud deletion can be asynchronous.
|
|
89
|
+
|
|
90
|
+
The worker saves `.citeark-job/state.json`; research and execution keep their normal recovery bundles, runtime sessions and experiment checkpoints. Reconnect to a live worker with `session attach`. Restart an interrupted worker with `start --resume --work-dir ...`. Starting a second worker in an active workspace is rejected. If a crashed coordinator left its previous Docker container running, recovery reports that container and refuses to start another Agent; inspect and stop the surviving execution before restarting the workspace. Machine restart recovery is manual; this release does not install a system service or automatically restart interrupted training. Back up research workspaces as normal project data.
|
|
91
|
+
|
|
92
|
+
## Independent AutoDL
|
|
93
|
+
|
|
94
|
+
GCP Secret Manager remains the default credential source for existing hosted configurations. An independent AutoDL catalog may explicitly set `executorConfig.credentialSource` to `task-key`. Starting that task also requires `--allow-remote-model-key`; this authorizes sending the configured model key to the selected AutoDL instance's temporary runtime environment. Without that flag, the task-key mode fails before instance acquisition. AutoDL's own API token is supplied through `AUTODL_API_TOKEN` or `--autodl-token-env`.
|
|
95
|
+
|
|
96
|
+
Existing GCP Batch and AutoDL catalog fields, images, capacity and credentials still need provider-specific setup. This release does not provision cloud accounts or guarantee GPU availability.
|
|
97
|
+
|
|
98
|
+
## Current validation boundary
|
|
99
|
+
|
|
100
|
+
The automated suite covers actual npm installation, protocol requests with synthetic model responses, detached Node workers, child-process cancellation and native SSH with a loopback test server, including signed CAP transfer. These checks are distinct from a real paper reproduction. Native provider models, production GPU capacity and machine-loss recovery need separately recorded real-environment validation. The project remains alpha.
|
|
@@ -0,0 +1,92 @@
|
|
|
1
|
+
# CiteArk 与 CiteArk Agent 的集成边界
|
|
2
|
+
|
|
3
|
+
## 所有权
|
|
4
|
+
|
|
5
|
+
PDF 解析属于 Agent:本机默认运行随包发布的 MinerU CPU 解析器,平台通过 `prepareMarkdown` 注入按需运行和私有存储运输。解析脚本、输出映射、缓存与原文回看规则在 Agent 中统一维护;宿主负责云任务身份、预算和恢复。官方 MinerU API 已从默认实现移除,详见[论文阅读](paper-reading.md)。
|
|
6
|
+
|
|
7
|
+
CiteArk Agent 维护来源理解、用户目标与实验选择、科研规划、自主执行、证据检查、评估、CAP,以及本地和远程执行器。原论文、可选代码、路径说明、历史 CAP 都可独立输入;没有平台账户也能完成研究。
|
|
8
|
+
|
|
9
|
+
CiteArk 维护网站、身份权限、任务队列、租约、费用授权、托管服务、呈现与发布。Processing Worker、派发服务、重评服务和页面生成位于网站的 `services/agent/`;Agent 无网站源码依赖。平台费用模式只在宿主解释,科研调度接收通用资源约束。
|
|
10
|
+
|
|
11
|
+
## 三个接口
|
|
12
|
+
|
|
13
|
+
- `@citeark/agent/host`:宿主启动科研流水线、调用执行器和处理检查点的受支持接口。宿主决定任务授权、预算和调度;科研证据由引擎生成。
|
|
14
|
+
- `@citeark/agent/cap` 与轻量包 `citeark-cap`:CAP 读取、验签和内容验证。网站 HTTP 进程安装轻量包,不加载科研引擎、云端执行器或页面生成器。协议包在本仓库构建,不新增仓库。
|
|
15
|
+
- 网站 `/api/v1/artifacts`:显式上传签名研究清单、研究计划与实验结果;默认私有,可明确公开并关联论文。`reproduction` 先上传其 `reproduces` 指向的计划,或引用网站已经验证的论文计划。结果保留自己的实验,不要求已有平台执行交付。
|
|
16
|
+
|
|
17
|
+
论文页 `/api/v1/repositories/{id}/deployment` 提供 `researchInput`(论文与代码来源)以及可选历史交付。`available` 表示可读取原论文,`replayAvailable` 表示存在历史复跑交付;二者独立。`start` 正常经过理解与规划,`deploy --replay` 是显式旧路线复跑。
|
|
18
|
+
|
|
19
|
+
## 工作区与 CAP
|
|
20
|
+
|
|
21
|
+
GCP Batch 的 GPU 工作区在分配 GPU 前,用同一科研 Agent 在 CPU 准备资料、任务依赖与可执行代码,再携带完整工作区、模型会话和累计预算进入 GPU。单项和整组选择共用这条路径;GPU 发现大量下载或其他 CPU 准备工作时,沿用已有检查点返回 CPU,准备后继续原实验。已经完成的恢复结果跳过准备,纯 CPU 任务在自己的工作区准备,不增加独立 GPU 阶段。设备交接不拆分科研目标或重置时间与费用授权。
|
|
22
|
+
|
|
23
|
+
工作区保存可继续研究的状态、用户选择和运行证据。CAP 保存某时刻的不可变科研快照。研究计划携带签名的 `research-plan-input`,研究清单使用 `research-inventory`;后续运行可复用来源理解,按新目标与算力重新规划。历史计划、代码和说明作为参考,不代替本次实验或授权。
|
|
24
|
+
|
|
25
|
+
验签证明字节与签名者,不自动授予平台运行者身份,也不证明结论为真。独立上传保留来源、计划关联与社区标记,不覆盖平台官方评估。数据集、模型权重和独立外部 Blob 不进入网站上传包。
|
|
26
|
+
|
|
27
|
+
网站托管服务当前以依赖名 `citeark-agent` 安装固定归档,保留原有导入路径;新独立集成可以使用 npm 包名 `@citeark/agent`。两者导出同一组接口。
|
|
28
|
+
|
|
29
|
+
## 发布
|
|
30
|
+
|
|
31
|
+
本仓库独立安装、检查、测试,使用 `npm pack` 生成 `@citeark/agent`,同一字节产物复制为网站 `cli.tgz`;另生成手动安装包和轻量 CAP 包。npm 发布归本仓库的 `publish.yml` 管理,详见 [npm 发布](npm-release.md)。网站的版本清单记录 Agent 提交、`cli.tgz`、`citeark-agent.tgz`、`cap.tgz` 摘要。网站与托管服务各自安装固定依赖、维护锁文件、检查兼容性,再分别从各自的提交构建。科研基础运行时镜像仍在 Agent 仓库维护。
|
|
32
|
+
|
|
33
|
+
公开 GitHub 前仍须检查拟公开源码与提交历史;更新代码不会自动改变仓库可见性。生产操作遵守 CiteArk 的部署文档,不用真实论文调度来代替无副作用验收。
|
|
34
|
+
|
|
35
|
+
执行阶段的计划快照通过 `derivedFrom` 关联用户选择时看到的签名计划,两份快照均保留在工作区和上传顺序中;结果的 `reproduces` 指向执行阶段计划。
|
|
36
|
+
|
|
37
|
+
## ArkGraph 切换
|
|
38
|
+
|
|
39
|
+
CiteArk Agent 与轻量 `citeark-cap@2.0.0-alpha.2` 输出及接收 CAP 2。宿主必须同步更新公开函数名(取消 V1 后缀)、五类 Record 与语义 role、图查询接口。`graphRecords` 是规范对象,`records` 是按 role 派生的科研视图。网站和宿主须固定三个新包后协调发布,不将新旧协议混用。原始工作区和执行契约仍供科研引擎使用,不是旧 CAP 兼容路径。
|
|
40
|
+
|
|
41
|
+
## 本地终端呈现
|
|
42
|
+
|
|
43
|
+
CiteArk Agent 的终端以五阶段导航和完整会话记录呈现现有科研流程。输出通过调用上下文中的本地订阅接入,本机 Docker、Codex 与远程执行器共用事件解码;订阅不依赖网站源码。原始 AI 发言和工具结果保存在用户工作区,网站不因安装包更新而获得这些记录。GCP 私有暂存轨迹与 AutoDL SSH 轨迹只在 CLI 订阅存在时开启;托管 Worker 默认行为保持原有接口。控制与边界见 [终端说明](terminal.md)。
|
|
44
|
+
|
|
45
|
+
## 独立服务器与托管平台
|
|
46
|
+
|
|
47
|
+
通用 SSH 接入把用户材料传到已配置的 Linux 服务器,由同一 CLI 启动持久工作进程;用户服务器和 CiteArk 自有服务器没有特殊协议差异。模型命名配置、后台会话和运行时控制属于 Agent。网站托管 Worker 继续直接调用固定版本的 host API,账号、派发、预算授权与租约属于网站;不会读取本地用户配置或被 CLI 的取消上下文影响。
|
|
48
|
+
|
|
49
|
+
原生 Responses、Messages 及 Chat Completions 来源审查与科学评估共用协议适配器;OpenRouter 托管调用保留共同接口和预算包装。原始供应商续接数据只存在内存,不写入用户会话或 CAP。独立 AutoDL 的任务密钥传递必须同时设置目录选项和显式命令授权;现有托管默认密钥模式不变。
|
|
50
|
+
|
|
51
|
+
## 运行完成与检查点
|
|
52
|
+
|
|
53
|
+
远端 Codex 容量错误由随任务暂存的启动器在同一环境内恢复:默认等待 10 秒后续接实际会话,最多 30 次;不改变模型、账号锁、已完成文件、费用或不可变执行时限。`agent.capacityRetryDelaySeconds` 可设为 1–60,`agent.capacityMaxRetries` 可设为 0–30(0 禁用);仅明确的 `turn.failed` 容量错误可触发。额度、鉴权、输入错误、取消和超时仍立即返回。恢复耗尽记录 `codex.capacity_retry_exhausted`,保留现场并停止外层自动重跑。启动器保存在当前调用的 `/tmp` 中,旧检查点不能覆盖它。
|
|
54
|
+
|
|
55
|
+
GCP 运输任务 `SUCCEEDED` 只证明归档已回传;编译状态为 `invalid_result` 且原 Agent 进程失败,或本次是显式修复时,后续恢复从检查点建立新一代任务,不能再次读取同一个已拒绝的终态结果。原任务与费用记录保留;真正完成的编译结果、独立来源审查的恢复和仍运行的任务继续复用。
|
|
56
|
+
|
|
57
|
+
远端运行器等待 Agent 进程成功退出,再验证和发布输出。输出 JSON 暂时没有变化只代表一个工作区快照,不能据此终止 Agent 或接收未完成计划。恢复时仅复用进程正常完成后生成的 `agent-completed` 标记;旧 `stable` 标记不构成完成证据。时限、预算和停滞检测继续独立生效。
|
|
58
|
+
|
|
59
|
+
0.3.4 修复了真实托管验收中发现的草稿被提前接收问题;不改变科研范围或 CAP 协议。
|
|
60
|
+
|
|
61
|
+
## 细化科研对象(0.3.5)
|
|
62
|
+
|
|
63
|
+
编译、计划和执行产物现在保留显式数据集、划分、模型、检查点及细分步骤。新产物协议为 CAP `2.0.0-alpha.2`;轻量读取器继续验签读取既有 CAP 2 alpha.1。网站应同步固定安装包,读取新增 `hasStep`、`expects`、`about` 关系,并严格区分待生成规格、论文声明、实际观测和单项科学判断。计划步骤不形成额外调度授权,单项判断不覆盖整体主张判断;完整语义见 `protocol/CAP.md`。
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
## 统一模型路由(0.3.7)
|
|
67
|
+
|
|
68
|
+
宿主从 `host` 导入 `createModelClient`、`createScientificAssessor`、`checkModelReadiness` 与 `openCodexAccount`。模型传输与凭据解析集中在 `provider/`;来源审查、交接后处理和科学评估不再自行解析 API 密钥。原生编译与执行共用 `modelRoute` 的订阅选择规则,显式 API 模式仍使用对应协议和论文预算准入。
|
|
69
|
+
|
|
70
|
+
托管配置包含 `codexAccountId` 时,全部科研模型阶段遵循该订阅和选定模型。协调器为每次来源审查、交接或独立评估开启新的 Codex 会话,只提供当前阶段的只读证据工具;不继承实验会话的上下文。账号通过同一 GCS 锁和 Secret Manager 登录协议使用,刷新后保存并清理临时登录目录。密钥、云端凭据和其他模型配置不进入子进程环境。订阅失败向上报错并保留科研检查点,禁止读取 API 密钥、使用备用模型或签发无法判断结果来掩盖登录故障。订阅费用标记为 `subscription`,过去的 API 支出继续保留。
|
|
71
|
+
|
|
72
|
+
无模型回归覆盖三阶段路由、独立会话、只读工具、账号清理、故障停止与 API 禁止回退;这不等同于真实模型能力或论文复现验收。宿主协调器及科学容器都应使用支持所选原生模型及 `exec --output-schema` 的 Codex CLI,不能只升级协调器。当前构建固定 `0.159.1`,镜像契约检查同时验证该版本。订阅服务明确拒绝模型时以 `codex.subscription_model_unavailable` 结束,不重新派发相同失败结果;恢复前检查实际执行端版本与账号模型目录。
|
|
73
|
+
|
|
74
|
+
0.3.8 将同一路由规则延伸到 GCP 参数解析:订阅任务完全不要求或继承 API Secret 引用,显式 API 任务继续验证引用格式。
|
|
75
|
+
来源审查的订阅错误经过所有编译执行器共享的终止边界,不作为计划内容错误触发重编译;原始 Provider 故障保留在模型观测与错误因果中。回归同时覆盖初次编译和已完成草稿的恢复。
|
|
76
|
+
|
|
77
|
+
## 大型 CAP 读取(0.3.10)
|
|
78
|
+
|
|
79
|
+
通过归档、签名和逐项摘要校验后,读取器按清单顺序每批最多打开 32 个 Record 文件,批次全部完成后才进入下一批。读取失败等待本批结束并返回原始错误,不返回缺失记录的图。该限制不裁剪科研对象、不改变归档字节或 CAP 协议版本;宿主仍需按解压后大小与图规模分配导入资源。
|
|
80
|
+
|
|
81
|
+
## 科学提取粒度(0.3.14)
|
|
82
|
+
|
|
83
|
+
来源理解按科学命题组织主张,表格、矩阵和曲线族用显式观测集合保留;不限制主张总数,也不要求一图一主张。数值仍位于 `reportedMeasurements` 的稳定测量目录,通过可选 `observationId` 指向来源观测对象。CAP 生成端把这些行写入同一个观测实体,保留全部数值、条件、维度和来源,供现有执行契约和逐项评估继续引用。机制假说独立为 `hypotheses` 和对应的假说断言,不进入实验目标目录。
|
|
84
|
+
|
|
85
|
+
准备交接保留观测归属、来源对象和假说;来源审查可以看见这些对象并分别审查假说的来源忠实度。计划步骤不再继承实验包主张,只有显式 `claimIds` 形成直接主张关系,包级目标与共享依赖不变。新安装包同时更新提示、结构模式、交接与 CAP 生成;旧产物不迁移、不按展示层规则合并。签名往返及无模型回归验证数据保留和图语义,不代表真实模型已经重新提取论文。
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
## 平台域名迁移(2026-10-04)
|
|
89
|
+
|
|
90
|
+
Agent 0.3.18 的默认平台地址为 `https://citeark.co`,用于启动、显式历史复跑、上传及平台健康预检;用户显式指定的 `--platform` 或 `baseUrl` 仍优先。网站为旧 `.com` API 与下载地址保留直接兼容,不要求旧客户端通过跨域跳转传递凭据。网站、登录、邮箱和支付后台的切换按 CiteArk 的 `docs/project/domain-migration.md` 执行;源码或安装包分支完成不代表生产域名已切换。
|
|
91
|
+
|
|
92
|
+
CAP 类型、模式、扩展规范和签名谓词的 `.com` URI 保持原值,它们是协议身份;不会重写既有产物、证据或签名。本次基于已发布 0.3.16,只调整平台地址与文档,避开另一个未发布任务的 0.3.17。
|
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
# 独立与托管运行完善计划
|
|
2
|
+
|
|
3
|
+
确认范围:用户本机、用户远程服务器、CiteArk 自有服务器(直接运行及平台托管)使用同一科研引擎。版本保持 0.3.x,普通发布只增加补丁号。科研流程不由历史 CAP 锁定。
|
|
4
|
+
|
|
5
|
+
## 实施顺序
|
|
6
|
+
|
|
7
|
+
1. 统一模型配置、协议和凭据解析;支持命名配置、默认选择及研究配置快照。科学评估和来源审查沿用所选协议。消除 AutoDL 对 GCP 密钥服务的强制依赖。
|
|
8
|
+
2. 增加可独立部署的任务工作进程及通用 SSH 接入,复用现有工作区和事件协议。完善查看、连接、停止、检查点恢复和产物取回。自有服务器和用户服务器使用同一入口。
|
|
9
|
+
3. 配置算力目录、模型连接检查、英文交互、阶段边界的后续指令与暂停。平台继续负责账号、队列、租约和收费,Agent 负责研究和执行。
|
|
10
|
+
4. 验证真实安装包、三类协议、后台进程、SSH、断开重连、取消及产物;更新两仓库集成说明和固定包,完成网站发布。托管服务变更按活动任务门禁发布。
|
|
11
|
+
|
|
12
|
+
## 验收与限制
|
|
13
|
+
|
|
14
|
+
- 自动测试验证软件行为,合成模型响应不能证明真实科研成功。
|
|
15
|
+
- 真实验收从一篇小型 CPU 论文开始,再验证独立远程 GPU、GCP 与 AutoDL。分别记录配置、模型、实际执行、费用、结果和未验证范围。
|
|
16
|
+
- 付费模型及云计算验收先列出具体任务和费用上限;未经本次明确授权,不运行正式科研调度器。
|
|
17
|
+
- 暂停在可保存的阶段边界生效;训练恢复以实际检查点为准。停止必须报告资源释放情况。
|
|
18
|
+
- 已有服务器只停止本次任务;Agent 创建的云实例按任务生命周期释放。用户退出终端不隐式终止远程研究。
|
|
19
|
+
- 本轮版本 0.3.3 仍属于 alpha;完成配置和自动测试后仍须如实列出未完成的实机验收。
|
|
20
|
+
|
|
21
|
+
## 0.3.3 实施记录
|
|
22
|
+
|
|
23
|
+
已实现命名模型/算力配置与研究快照、三类供应商协议、通用 SSH、后台工作进程、重连与记录浏览、阶段边界暂停、后续说明、取消及清理结果、独立 AutoDL 显式凭据模式和本地恢复包。交互终端与非交互后台启动使用同一工作进程。
|
|
24
|
+
|
|
25
|
+
本机完整回归 515 项通过;配置快照追加验收 18 项通过。实际 npm 全局安装及两个命令入口通过;SSH 测试使用系统 SSH 客户端与本机隔离服务,实际启动独立工作进程、续跑、拒绝重复工作区并取回验签 CAP。模型协议测试使用合成响应。首次准备 CPU 容器与生产固定包发布另行记录,不能由以上数字推定完成。
|
|
26
|
+
|
|
27
|
+
尚未确认:原生供应商实际模型响应、生产 GPU、真实论文科学效果、机器故障后的实际训练检查点恢复,以及独立研究跨所有模型阶段的统一费用上限。当前模型预算参数不是整项研究总费用承诺。继续使用 alpha 标记。
|
|
@@ -0,0 +1,44 @@
|
|
|
1
|
+
# npm 发布
|
|
2
|
+
|
|
3
|
+
## 状态与安装
|
|
4
|
+
|
|
5
|
+
包名:`@citeark/agent`;主命令:`citeark`;兼容命令:`citeark-agent`。当前准备首次发布 `0.3.4`,npm 账号及 `@citeark` 组织权限尚待配置。仓库仍私有。
|
|
6
|
+
|
|
7
|
+
发布后安装与升级均使用 `npm install -g @citeark/agent`;固定研究环境可使用 `npm install -g @citeark/agent@0.3.4`。`citeark` 在交互终端提供论文、平台短编号、CAP 和恢复入口;没有终端时打印帮助。论文理解、目标规划、实验选择与执行沿用同一研究流程。
|
|
8
|
+
|
|
9
|
+
## 版本约定
|
|
10
|
+
|
|
11
|
+
- 产品保持 `0.x`。日常功能、界面、重构与修复发布只递增末位补丁号,例如 `0.3.3 → 0.3.4`;代码提交本身不要求加版本,发布变更后的安装包才更新版本。
|
|
12
|
+
- 不因新增功能、架构调整、分仓或开发阶段自行提升次版本或主版本。需要改变这一约定时,先说明原因,由用户明确决定;不兼容的接口或数据变化须在发布说明中说明。
|
|
13
|
+
- Agent 软件、CAP 协议和图查询规则各自编号,不为对齐产品版本而更改协议格式或历史产物。
|
|
14
|
+
|
|
15
|
+
2026-09-29 按用户要求将产品从 `3.2.0` 重新编号为 `0.3.2`,功能不回退;历史发行记录保留原始编号。已安装用户重新执行 `npm install -g https://citeark.com/cli.tgz` 即可切换。
|
|
16
|
+
|
|
17
|
+
## 本地验证与首次发布
|
|
18
|
+
|
|
19
|
+
```bash
|
|
20
|
+
npm ci
|
|
21
|
+
npm run check
|
|
22
|
+
npm test
|
|
23
|
+
npm run test:package
|
|
24
|
+
COPYFILE_DISABLE=1 npm run package:local
|
|
25
|
+
npm publish ./dist/citeark-agent-0.3.4.tgz --dry-run
|
|
26
|
+
```
|
|
27
|
+
|
|
28
|
+
`test:package` 将真实 npm 压缩包全局安装到临时目录,验证两个命令、帮助、示例、独立输入准备及公开接口,不调用模型或启动实验。`files` 指定安装所需源码、提示词、模式、运行环境及示例;不打包仓库历史、环境文件、依赖缓存和测试。
|
|
29
|
+
|
|
30
|
+
首次正式发布需要具备 `@citeark` 写入权限的 npm 账号,完成邮箱验证与双重验证,在终端运行 `npm login --registry=https://registry.npmjs.org/`。核对上述产物后执行 `npm publish ./dist/citeark-agent-0.3.4.tgz --access public`。公开 npm 包包含源码,但不会公开 Git 历史或改变 GitHub 可见性。
|
|
31
|
+
|
|
32
|
+
## 后续自动发布
|
|
33
|
+
|
|
34
|
+
在 npm 包设置的 Trusted Publisher 中配置:
|
|
35
|
+
|
|
36
|
+
- GitHub 账号:`ganwumeng`
|
|
37
|
+
- 仓库:`CiteArk-Agent`
|
|
38
|
+
- 工作流文件名:`publish.yml`
|
|
39
|
+
|
|
40
|
+
更新 `package.json` 与锁文件版本后,推送匹配的 `v<版本>` 标签。GitHub Actions 检查、测试、打包并通过短期身份凭证发布,无需在仓库保存 npm 发布令牌。普通代码 push 只检查并构建安装包,不发布 npm。私有仓库不生成 provenance;仓库公开后工作流自动启用。
|
|
41
|
+
|
|
42
|
+
网站只消费通过检查的固定安装包,记录源码提交与摘要。npm 首次发布验证成功前,网站继续显示可用的固定归档安装命令;成功后再切换为短 npm 包名。
|
|
43
|
+
|
|
44
|
+
参考:[npm 可信发布](https://docs.npmjs.com/trusted-publishers/)、[package.json 文件清单与命令入口](https://docs.npmjs.com/cli/v11/configuring-npm/package-json/)。
|
|
@@ -0,0 +1,40 @@
|
|
|
1
|
+
# 本机 CPU 论文解析
|
|
2
|
+
|
|
3
|
+
PDF 阅读副本由独立 CiteArk Agent 生成。默认使用本机 MinerU 4.0.10、ONNX 和 llama.cpp CPU;无需 CiteArk 账号、Google Cloud、GPU 或 MinerU 官方接口密钥。解析器不接受远端推理地址,也不回退到官方 API。科研编译和评估所需的模型仍按用户自己的配置运行。
|
|
4
|
+
|
|
5
|
+
## 安装与使用
|
|
6
|
+
|
|
7
|
+
建议 Python 3.12、4 核 CPU 和 16 GiB 可用内存。Linux 需要 `libgl1`、`libglib2.0-0`、`libgomp1`、`libvulkan1`;读取归档需要 `unzip`。macOS 使用本机 Python;Windows 当前建议在 WSL2 中运行,原生 Windows 尚未完成端到端验收。
|
|
8
|
+
|
|
9
|
+
```bash
|
|
10
|
+
# 首次安装依赖并下载公开模型;保存到 ~/.citeark/mineru-4.0.10
|
|
11
|
+
citeark paper setup --python python3.12
|
|
12
|
+
# 网络环境需要时可显式改为 ModelScope 下载公开模型
|
|
13
|
+
citeark paper setup --python python3.12 --source modelscope
|
|
14
|
+
|
|
15
|
+
# 单独解析,无需配置科研模型或启动实验
|
|
16
|
+
citeark paper parse --input ./paper.pdf --output ./paper-reading
|
|
17
|
+
|
|
18
|
+
# 正常研究流程也自动使用同一个本机解析器
|
|
19
|
+
citeark start --paper ./paper.pdf --repository ./code --plan-only
|
|
20
|
+
```
|
|
21
|
+
|
|
22
|
+
安装使用独立 Python 虚拟环境,不把大型模型加入 npm 安装包。首次下载需要联网;成功后使用本地模型离线解析。缺少运行环境或模型时给出失败,不上传论文或静默跳过阅读副本。Markdown 等非 PDF 输入继续直接读取。
|
|
23
|
+
|
|
24
|
+
`CITEARK_MINERU_HOME` 修改运行环境与模型目录;`CITEARK_MINERU_PYTHON` 可指定已经安装固定依赖的 Python 解释器绝对路径;`CITEARK_MINERU_THREADS` 修改 CPU 线程数(默认 4,1–64)。其他 MinerU 配置文件、远端推理设置及官方 API 密钥不参与解析。用户仍须保留原 PDF,转换文字不构成独立核验的原文证据。
|
|
25
|
+
|
|
26
|
+
## 阅读产物与复用
|
|
27
|
+
|
|
28
|
+
输出包括 `paper.md`、`source-map.json` 和 `images/`。每个块、图注和脚注保留从 1 开始的 PDF 物理页码;坐标采用 MinerU 4 的 0–1 归一化范围,映射格式版本为 2。表格 HTML、公式 LaTeX、公式原图和图像均保留。遇到文字、数值、图注关联或上下标疑问时回看标记对应的原 PDF 页。
|
|
29
|
+
|
|
30
|
+
缓存键包含 PDF 字节摘要、MinerU/llama.cpp 版本、CPU 解析配置和阅读格式版本。命中时逐文件校验摘要,后续规划与执行复用同一阅读副本。旧官方 API 缓存有不同配置摘要,不被冒充为本次本机解析结果。失败保留诊断,不自动重新提交解析。
|
|
31
|
+
|
|
32
|
+
损坏的缓存会在下一次显式解析时清除并重建。解析沿用研究任务的取消信号;取消时终止本机解析进程组、清理临时 PDF,并保留取消原因,不把取消转换成普通解析失败。
|
|
33
|
+
|
|
34
|
+
2026-10-02 的 Linux 4 核、16 GiB 测试中,24 页 Fourier 论文首次解析 301 秒,同进程第二遍 260 秒,整容器峰值约 7.8 GiB;这是单篇样本,不代表所有论文。另在 macOS Apple Silicon 上用同一 CLI 实际解析原公式页与表格页组成的两页样本,CPU 解析约 58 秒,再次读取命中缓存;不将两页结果外推为全文速度。表 1 的 28 个数值和抽查公式匹配,但附录仍有上下标及切图关联问题。
|
|
35
|
+
|
|
36
|
+
## 平台集成
|
|
37
|
+
|
|
38
|
+
宿主可以通过 `preparePaperMarkdown({ convert })` 与编译入口的 `prepareMarkdown` 注入自己的运输方式,输出必须是同版本解析归档。平台按需 CPU 任务执行 npm 包内同一个 `runtime/mineru/parse.py`;Cloud Run、对象存储、费用预留和任务身份仅存在于网站宿主。Agent 本机默认路径不导入平台代码、不要求其云资源。
|
|
39
|
+
|
|
40
|
+
解析完成不授权科研实验。任务预算、实验选择、原文核验与 CAP 科学判断的边界保持独立。
|
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
# 研究计划的科研粒度
|
|
2
|
+
|
|
3
|
+
来源理解、科研规划与实际执行各有不同的身份粒度。来源清单保留论文明确指认的对象、科学命题、观测集合、逐项测量和机制假说;规划描述科研方法阶段与完整结果集合;执行记录实际运行的配置、独立科学状态、产出和原始证据。完整性不要求提前枚举每个运行组合,也不允许通过减少参数或删除测量来缩小图。
|
|
4
|
+
|
|
5
|
+
## 何时另建节点
|
|
6
|
+
|
|
7
|
+
同一评估方法遍历检查点、数据集和随机种子,可以使用一个评估步骤,产出一个带完整维度和行键的分数表。计算相关矩阵或逐层预测指标时,层、指标和数据集对属于结果集合维度,不是每个单元格的独立未来观测。预期训练状态族使用完整配方和种子轴作为待生成规格;这不合并实际模型字节,已明确命名的来源检查点仍保持各自身份。
|
|
8
|
+
|
|
9
|
+
当科研操作、输入科学状态、解释或决策规则、可复用的消费范围实质不同,才拆方法步骤或产物。数据集或模型变体本身不必成为新实验包;需要独立选择的科研问题、方法、硬件比较或成本类别可以拆包。共享图号、标题或名字不建立对象同一性。步骤输入和产物来源声明直接依赖,后续分析引用分数集合,不反复列出每个传递祖先。
|
|
10
|
+
|
|
11
|
+
完整参数轴、来源选择规则、成员身份、对照、重复数、聚合和行键写入 `protocol.objective/conditions`、步骤描述及未来对象 `conditions`;`reportedMeasurementId` 继续逐项绑定来源测量。没有可靠数值目标的发现,用实验包的 `observationTargets: [{claimId, observationId, comparison, limitation: {kind: "decision_rule", evidence}}]` 保留来源观察比较及方法步骤。观察身份必须显式关联到该来源主张,不能填进数值 `reportedMeasurementId` 或配上标量解析器。只有观察目标的提案使用空 `measurements`,保留在目录但禁止自动数值执行;即使显式选择,也不会产生空的执行合同。混合包保留有效数值目标,定性语境仍不能视为已验证。错误观察身份或观察冒充标量是结构错误,不能静默删掉方法。
|
|
12
|
+
|
|
13
|
+
## 批次计价与复用
|
|
14
|
+
|
|
15
|
+
`compute.workloads` 是精确的计算与复用身份,不是图节点或独立调度任务。一个条目可涵盖完整配方及参数、种子集合,时长计算必须包含每个成员。多个实验消费相同批次时使用完全相同的身份和定义。部分共享才按消费者需要划分精确共享子集及不重叠余集,例如关键检查点路线与全量扫描;全量扫描引用子集和余集,不能把重叠批次当作互不重叠的工作重复计价。
|
|
16
|
+
|
|
17
|
+
计算分区不强制对应图分区:一个科研步骤和分数集合可以关联多个计算条目。不同完整批次不能共享身份,也不能用“执行器会去重”的句子代替真实复用声明。具体循环、成员运行记录和调度由选择后的执行阶段完成,保留完整条件和证据,不把预期集合当作运行成功。
|
|
18
|
+
|
|
19
|
+
暂定成本估算可以披露未知网格大小的假设,但不能把估价包络写进不可变的 `protocol.workload` 科研完成数量。保留完整来源选择规则和未知基数,只有真实完成数量已知时才声明该字段;执行在启动科学计算前必须发现并记录完整来源范围。`environment.timeoutMinutes` 表示独立的执行超时,不能替代完整工作量估计。
|
|
20
|
+
|
|
21
|
+
## 契约与验收
|
|
22
|
+
|
|
23
|
+
规划提示、`scientific-planning-v8`、结构模式描述、覆盖修订反馈、有界修复和执行提示遵循同一粒度。CAP 仍为 `2.0.0-alpha.2`,使用既有对象条件、步骤和标准 `about` 来源观察关系,提案比较保存在实验的附加元数据中;不改写历史签名产物,也不增加节点数上限。
|
|
24
|
+
|
|
25
|
+
回归把 50 个检查点、4 个种子、32 层及 3 项指标贯穿规范化、执行合同和签名 CAP 往返,检查完整轴、运行单元数和成员身份规则保留,未来图不被强制展开。真实模型试跑必须比较原始草稿与验收计划,披露校验舍弃的绑定和实验;仅看最终图变小不足以证明生成质量改善。逐项核对数值、单位、维度、条件、来源、观测成员身份及假说不变,并核实完整工作量、对照和复用边界。
|
|
26
|
+
|
|
27
|
+
观察目标回归覆盖来源身份与主张归属校验、混合数值/定性包、禁止空目标自动执行,以及签名计划归档往返后保留方法与来源关系。
|
package/docs/terminal.md
ADDED
|
@@ -0,0 +1,49 @@
|
|
|
1
|
+
# Terminal workspace
|
|
2
|
+
|
|
3
|
+
Run `citeark` in a terminal to open the research workspace. Enter a paper URL, a local file path, or a CiteArk paper ID. `/cap` opens a CAP artifact; `/resume` continues a workspace; `/model` and `/compute` select reusable configurations; `/help` lists entry commands.
|
|
4
|
+
|
|
5
|
+
The interface uses English. The Agent is instructed to communicate in English while preserving source quotations, paths, code, and user-provided text. External tools and research material retain their original contents.
|
|
6
|
+
|
|
7
|
+
## Conversation and stages
|
|
8
|
+
|
|
9
|
+
The five stages are **Get paper**, **Understand research**, **Prepare materials**, **Run experiments**, and **Assess & deliver**. These reflect actual pipeline events. They provide orientation, not a percentage complete or a fixed experiment plan. The active stage can move back when the workflow revisits earlier work.
|
|
10
|
+
|
|
11
|
+
The conversation displays the runtime's user-facing messages, tool calls, commands, results, and errors in order. Assistant messages are shown in full. Long tool results can be expanded with `Ctrl+O`; their complete contents remain in the transcript. Internal reasoning events are not displayed. Structured HTTP diagnostics are retained in the session file without filling the conversation.
|
|
12
|
+
|
|
13
|
+
Codex, Claude Code, and OpenCode output is read as events arrive. Claude Code partial text updates in place; other runtimes may emit a complete message or tool result at once. Display latency depends on what the runtime emits. Local Docker and local Codex forward process output; AutoDL forwards trace events over SSH; GCP reads incremental bytes from a private staging object updated by the worker about every two seconds. This transport is enabled for CLI subscribers only.
|
|
14
|
+
|
|
15
|
+
## Controls
|
|
16
|
+
|
|
17
|
+
| Control | Action |
|
|
18
|
+
| --- | --- |
|
|
19
|
+
| Mouse wheel / PgUp / PgDn | Scroll the conversation |
|
|
20
|
+
| End | Return to live output while viewing history |
|
|
21
|
+
| Ctrl+O | Expand or collapse long tool results |
|
|
22
|
+
| Up / Down | Select an option in a menu |
|
|
23
|
+
| Enter | Submit the current prompt |
|
|
24
|
+
| Ctrl+U | Clear input before the cursor |
|
|
25
|
+
| Ctrl+C | Cancel a prompt, or disconnect the running CLI |
|
|
26
|
+
|
|
27
|
+
Scrolling up holds the view while new output arrives. A counter shows new events. Interactive research uses a separate worker process. Ctrl+C detaches the terminal; the worker continues. Reconnect with `citeark session attach <directory>`. `session status` reports the worker state and any resource cleanup results.
|
|
28
|
+
|
|
29
|
+
While attached, enter a message to include it in the next Agent invocation. The current model turn is not interrupted. `/pause` pauses at the next pipeline stage boundary; `/resume` releases that pause; `/cancel` requests cancellation and resource cleanup. Pausing a phase does not suspend a training process halfway through a command. Resource cleanup may be reported as requested, released, pooled, or failed; a cloud deletion request is not proof that deletion has finished.
|
|
30
|
+
|
|
31
|
+
`citeark start --resume --work-dir <directory>` continues a stopped workspace. `citeark session resume <directory>` releases a live worker's pause. Neither means that training automatically resumes at the exact interrupted instruction: that depends on the experiment's saved checkpoints. A completed workspace is opened without starting new compute.
|
|
32
|
+
|
|
33
|
+
## History and ordinary terminal output
|
|
34
|
+
|
|
35
|
+
`session.events.jsonl` in the workspace retains messages, updates, tool results, stage events, and structured diagnostics. Hidden credential prompts are recorded only as `[hidden]`. Treat the transcript like your local research files: tool output can include source contents and commands.
|
|
36
|
+
|
|
37
|
+
```sh
|
|
38
|
+
citeark view ./my-research
|
|
39
|
+
citeark start --resume --work-dir ./my-research
|
|
40
|
+
citeark start --paper ./paper.pdf --plain
|
|
41
|
+
```
|
|
42
|
+
|
|
43
|
+
`view` only reads the saved conversation. It does not resume research or contact a model. `--plain`, redirected output, and `TERM=dumb` use line-oriented output with full tool results. `--non-interactive` disables questions and requires explicit options. `--dry-run` prepares inputs without models or compute.
|
|
44
|
+
|
|
45
|
+
Named configurations are stored in `~/.citeark/config.json`; existing `local-model.json` preferences remain readable. See [configuration and servers](configuration.md). Override them through command options or the documented environment variables. API credentials are not saved by onboarding; set `CITEARK_MODEL_API_KEY` or use `--api-key-env` to avoid entering a key each time.
|
|
46
|
+
|
|
47
|
+
## Integration boundary
|
|
48
|
+
|
|
49
|
+
The event subscription is local to the invoking CLI. The host API, CAP protocol, scientific prompts, experiment selection, resource constraints, evidence checks, and optional upload flow keep their existing roles. The CLI adds a communication instruction for user-facing messages. Hosted runs without a CLI subscriber do not publish an additional raw trace object or request partial message output.
|
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
{
|
|
2
|
+
"schemaVersion": "0.1",
|
|
3
|
+
"name": "compile-toy-checkpoint-evaluation",
|
|
4
|
+
"paper": {
|
|
5
|
+
"title": "A Toy Accuracy Claim",
|
|
6
|
+
"path": "./paper.md"
|
|
7
|
+
},
|
|
8
|
+
"repository": {
|
|
9
|
+
"path": "./repository"
|
|
10
|
+
},
|
|
11
|
+
"agent": {
|
|
12
|
+
"runtime": "claude-code",
|
|
13
|
+
"effort": "max",
|
|
14
|
+
"maxTurns": 40,
|
|
15
|
+
"maxBudgetUsd": 3,
|
|
16
|
+
"validationRetries": 1
|
|
17
|
+
},
|
|
18
|
+
"environment": {
|
|
19
|
+
"image": "citeark-agent/runtime:0.2.0",
|
|
20
|
+
"timeoutMinutes": 15,
|
|
21
|
+
"cpus": 2,
|
|
22
|
+
"memoryGb": 4,
|
|
23
|
+
"shmGb": 1,
|
|
24
|
+
"gpu": "none",
|
|
25
|
+
"networkAccess": true
|
|
26
|
+
}
|
|
27
|
+
}
|
|
@@ -0,0 +1,9 @@
|
|
|
1
|
+
# Toy evaluation repository
|
|
2
|
+
|
|
3
|
+
Run the released checkpoint evaluation with:
|
|
4
|
+
|
|
5
|
+
```bash
|
|
6
|
+
python3 evaluate.py --checkpoint checkpoint.json
|
|
7
|
+
```
|
|
8
|
+
|
|
9
|
+
The command prints a JSON metric object. Save the actual command output as evidence for any reproduction conclusion.
|