@citeark/agent 0.3.20
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +202 -0
- package/README.md +128 -0
- package/data/dataset-source-registry.v1.json +300 -0
- package/dist/arkgraph/boot.js +6 -0
- package/dist/arkgraph/index.html +1 -0
- package/dist/arkgraph/viewer.css +1 -0
- package/dist/arkgraph/viewer.en.css +1 -0
- package/dist/arkgraph/viewer.en.js +49 -0
- package/dist/arkgraph/viewer.en.js.LEGAL.txt +56 -0
- package/dist/arkgraph/viewer.js +49 -0
- package/dist/arkgraph/viewer.js.LEGAL.txt +56 -0
- package/docker/claude-code/Dockerfile +97 -0
- package/docker/claude-code/codex-pro-relay.mjs +466 -0
- package/docker/claude-code/runtime-contract-check.mjs +79 -0
- package/docs/arkgraph-reading.md +79 -0
- package/docs/configuration.md +100 -0
- package/docs/integration.md +92 -0
- package/docs/maturity-plan.md +27 -0
- package/docs/npm-release.md +44 -0
- package/docs/paper-reading.md +40 -0
- package/docs/research-plan-granularity.md +27 -0
- package/docs/terminal.md +49 -0
- package/examples/toy-evaluation/compile-task.json +27 -0
- package/examples/toy-evaluation/paper.md +5 -0
- package/examples/toy-evaluation/repository/README.md +9 -0
- package/examples/toy-evaluation/repository/checkpoint.json +4 -0
- package/examples/toy-evaluation/repository/evaluate.py +17 -0
- package/examples/toy-evaluation/task.json +81 -0
- package/package.json +59 -0
- package/prompts/compile-research.md +58 -0
- package/prompts/execute-contract.md +72 -0
- package/prompts/execute-workspace-simple.md +51 -0
- package/prompts/execute-workspace.md +34 -0
- package/prompts/prepare-reproduction.md +82 -0
- package/prompts/repair-research.md +45 -0
- package/protocol/CAP.md +129 -0
- package/protocol/LICENSE +12 -0
- package/protocol/MAPPINGS.md +72 -0
- package/protocol/README.md +38 -0
- package/protocol/conformance-v2.0-alpha.1.json +36 -0
- package/protocol/examples/arkgraph/checkpoint-evaluation.json +309 -0
- package/protocol/examples/arkgraph/fixtures.mjs +49 -0
- package/protocol/examples/arkgraph/paper-free.json +291 -0
- package/protocol/examples/arkgraph/partial-failure.json +344 -0
- package/protocol/examples/arkgraph/training-evaluation.json +443 -0
- package/protocol/profiles/agent-trace.md +16 -0
- package/protocol/profiles/computational-run.md +16 -0
- package/protocol/profiles/core.md +15 -0
- package/protocol/profiles/public-bundle.md +18 -0
- package/protocol/profiles/reproduction.md +29 -0
- package/protocol/profiles/research-compilation.md +44 -0
- package/protocol/profiles/research-plan.md +39 -0
- package/protocol/profiles/restricted-evidence.md +15 -0
- package/runtime/bootstrap-autodl-runtime.sh +314 -0
- package/runtime/create-runtime-venv.sh +41 -0
- package/runtime/install-local-cpu-runtime.sh +23 -0
- package/runtime/install-scientific-runtime.sh +153 -0
- package/runtime/mineru/parse.py +62 -0
- package/runtime/mineru/requirements.txt +4 -0
- package/runtime/requirements-baseline.txt +38 -0
- package/schemas/cap/v2/activity.schema.json +47 -0
- package/schemas/cap/v2/agent.schema.json +32 -0
- package/schemas/cap/v2/assertion.schema.json +110 -0
- package/schemas/cap/v2/descriptor.schema.json +243 -0
- package/schemas/cap/v2/entity.schema.json +64 -0
- package/schemas/cap/v2/manifest.schema.json +67 -0
- package/schemas/cap/v2/relation.schema.json +82 -0
- package/schemas/compute-catalog.schema.json +63 -0
- package/schemas/compute-decision.schema.json +27 -0
- package/schemas/execution-contract.schema.json +1024 -0
- package/schemas/research-card.schema.json +30 -0
- package/schemas/research-inventory-draft.schema.json +366 -0
- package/schemas/research.schema.json +1044 -0
- package/schemas/result.schema.json +173 -0
- package/schemas/verification-policy.schema.json +47 -0
- package/schemas/verified-conclusion.schema.json +58 -0
- package/schemas/workspace-summary.schema.json +24 -0
- package/scripts/build-arkgraph-view.mjs +12 -0
- package/scripts/check-execution-feasibility.mjs +24 -0
- package/scripts/check-syntax.mjs +15 -0
- package/scripts/deterministic-asset-preparation.py +438 -0
- package/scripts/package-cap.mjs +23 -0
- package/scripts/package-local-agent.mjs +23 -0
- package/scripts/preview-arkgraph.mjs +25 -0
- package/scripts/replay-research-compiler-candidate.mjs +134 -0
- package/scripts/review-compiler-sources.mjs +44 -0
- package/scripts/run-asset-preparation.sh +17 -0
- package/scripts/run-research-plan.mjs +98 -0
- package/scripts/validate-asset-preparation.py +290 -0
- package/scripts/verify-local-runtime.mjs +57 -0
- package/scripts/verify-npm-package.mjs +57 -0
- package/src/adapters/paper2agent.mjs +107 -0
- package/src/assets/cache.mjs +159 -0
- package/src/assets/compute.mjs +98 -0
- package/src/assets/executor.mjs +145 -0
- package/src/assets/lifecycle.mjs +213 -0
- package/src/assets/manifest.mjs +242 -0
- package/src/assets/opportunistic-preparation.mjs +81 -0
- package/src/assets/plan.mjs +411 -0
- package/src/assets/prompts.mjs +29 -0
- package/src/assets/public-asset-probe.mjs +525 -0
- package/src/assets/qualification.mjs +119 -0
- package/src/assets/readiness.mjs +130 -0
- package/src/assets/reproduction-admission.mjs +355 -0
- package/src/assets/requirements.mjs +152 -0
- package/src/assets/source-grounding.mjs +341 -0
- package/src/assets/source-policy.mjs +118 -0
- package/src/autodl/client.mjs +260 -0
- package/src/autodl/ssh.mjs +380 -0
- package/src/autodl/tools.mjs +129 -0
- package/src/cap/redaction.mjs +38 -0
- package/src/cap/v2/archive.mjs +152 -0
- package/src/cap/v2/attestation.mjs +204 -0
- package/src/cap/v2/canonical-json.mjs +114 -0
- package/src/cap/v2/compilation-artifact.mjs +240 -0
- package/src/cap/v2/core.mjs +282 -0
- package/src/cap/v2/measurement-assessment-records.mjs +23 -0
- package/src/cap/v2/pipeline-artifact.mjs +922 -0
- package/src/cap/v2/read.mjs +41 -0
- package/src/cap/v2/reassessment-artifact.mjs +383 -0
- package/src/cap/v2/research-artifact.mjs +231 -0
- package/src/cap/v2/research-map-records.mjs +46 -0
- package/src/cap/v2/research-object-records.mjs +163 -0
- package/src/cap/v2/research-records.mjs +187 -0
- package/src/cap/v2/verify.mjs +642 -0
- package/src/cli.mjs +1146 -0
- package/src/compute/autodl-pro-compiler.mjs +347 -0
- package/src/compute/autodl-pro-executor.mjs +459 -0
- package/src/compute/autodl-pro-job.mjs +843 -0
- package/src/compute/autodl-pro-network.mjs +295 -0
- package/src/compute/autodl-pro-remote.mjs +810 -0
- package/src/compute/autodl-pro-staging.mjs +117 -0
- package/src/compute/campaign.mjs +110 -0
- package/src/compute/catalog.mjs +123 -0
- package/src/compute/checkpoint-protocol.mjs +154 -0
- package/src/compute/codex-account-lock.mjs +111 -0
- package/src/compute/codex-account-session.mjs +107 -0
- package/src/compute/compiler-profile.mjs +38 -0
- package/src/compute/compiler-router.mjs +23 -0
- package/src/compute/coordinator-recovery.mjs +210 -0
- package/src/compute/executor-router.mjs +29 -0
- package/src/compute/gcp-batch-compiler.mjs +685 -0
- package/src/compute/gcp-batch-executor.mjs +1215 -0
- package/src/compute/gcp-batch-failure.mjs +92 -0
- package/src/compute/gcp-batch-job.mjs +527 -0
- package/src/compute/gcp-batch-lifecycle.mjs +81 -0
- package/src/compute/gcp-checkpoint-worker.mjs +1633 -0
- package/src/compute/local-codex-compiler.mjs +52 -0
- package/src/compute/measurement-hardware.mjs +128 -0
- package/src/compute/remote-attempt.mjs +226 -0
- package/src/compute/requirements.mjs +124 -0
- package/src/compute/research-phases.mjs +48 -0
- package/src/compute/scheduler.mjs +452 -0
- package/src/compute/shared-workloads.mjs +26 -0
- package/src/compute/stage-archive.mjs +79 -0
- package/src/contracts/campaign-contract.mjs +52 -0
- package/src/contracts/execution-contract.mjs +819 -0
- package/src/contracts/execution-mode.mjs +19 -0
- package/src/contracts/execution-timeouts.mjs +45 -0
- package/src/contracts/execution-workload.mjs +68 -0
- package/src/contracts/preflight-schema.mjs +25 -0
- package/src/contracts/public-contract.mjs +63 -0
- package/src/contracts/subject-tags.mjs +31 -0
- package/src/dashboard/data.mjs +898 -0
- package/src/dashboard/server.mjs +79 -0
- package/src/dashboard/static/dashboard.css +366 -0
- package/src/dashboard/static/dashboard.js +560 -0
- package/src/dashboard/static/index.html +85 -0
- package/src/deployment/community-policy.mjs +9 -0
- package/src/deployment/environment.mjs +112 -0
- package/src/deployment/guided.mjs +98 -0
- package/src/deployment/handoff.mjs +102 -0
- package/src/deployment/local-contract.mjs +31 -0
- package/src/deployment/local.mjs +100 -0
- package/src/deployment/prepare.mjs +46 -0
- package/src/deployment/recipe.mjs +108 -0
- package/src/deployment/supplement.mjs +51 -0
- package/src/deployment/terminal.mjs +43 -0
- package/src/diagnosis/renderer.mjs +75 -0
- package/src/diagnosis/target-failure.mjs +46 -0
- package/src/evidence/parser-registry.mjs +54 -0
- package/src/evidence/parsers/fasttext-classification.mjs +82 -0
- package/src/evidence/parsers/json-scalar.mjs +96 -0
- package/src/evidence/parsers/simcse-senteval.mjs +104 -0
- package/src/evidence/parsers/starspace-classification.mjs +78 -0
- package/src/evidence/registry.mjs +147 -0
- package/src/execution/runner-audit.mjs +473 -0
- package/src/gcp/auth.mjs +106 -0
- package/src/gcp/batch-client.mjs +120 -0
- package/src/gcp/resource-discovery.mjs +177 -0
- package/src/gcp/rest.mjs +82 -0
- package/src/gcp/secret-manager.mjs +34 -0
- package/src/gcp/signed-url.mjs +133 -0
- package/src/gcp/storage.mjs +220 -0
- package/src/graph/command.mjs +41 -0
- package/src/graph/execution.mjs +97 -0
- package/src/graph/model.mjs +37 -0
- package/src/graph/presentation.mjs +110 -0
- package/src/graph/query.mjs +159 -0
- package/src/graph/research-relations.mjs +69 -0
- package/src/graph/source-page.mjs +12 -0
- package/src/graph/source-preview.mjs +34 -0
- package/src/graph/validate.mjs +76 -0
- package/src/job.mjs +496 -0
- package/src/network/autodl-routing-proxy.mjs +462 -0
- package/src/network/egress-proxy.mjs +158 -0
- package/src/observability/event-contract.mjs +230 -0
- package/src/observability/pipeline-monitor.mjs +166 -0
- package/src/pipeline/orchestrator.mjs +1281 -0
- package/src/pipeline/recovery-error.mjs +11 -0
- package/src/pipeline/replay.mjs +304 -0
- package/src/pipeline/shared-execution.mjs +115 -0
- package/src/pipeline/stage-checkpoint.mjs +86 -0
- package/src/pipeline/stage-recovery.mjs +101 -0
- package/src/pipeline/targets.mjs +110 -0
- package/src/process.mjs +143 -0
- package/src/protocol.mjs +312 -0
- package/src/provider/codex-account.mjs +44 -0
- package/src/provider/codex-completion.mjs +49 -0
- package/src/provider/completion.mjs +292 -0
- package/src/provider/model-client.mjs +44 -0
- package/src/provider/model-route.mjs +29 -0
- package/src/provider/openrouter-readiness.mjs +189 -0
- package/src/provider/reader-bridge.mjs +35 -0
- package/src/provider/relay.mjs +263 -0
- package/src/provider/runtime-auth.mjs +40 -0
- package/src/public/cap.d.mts +90 -0
- package/src/public/cap.mjs +12 -0
- package/src/public/contracts.d.mts +2 -0
- package/src/public/host.mjs +171 -0
- package/src/public/operations.d.mts +11 -0
- package/src/public/presentation.d.mts +4 -0
- package/src/records/views.mjs +26 -0
- package/src/remote/command.mjs +178 -0
- package/src/remote/ssh.mjs +59 -0
- package/src/repository-origin.mjs +81 -0
- package/src/reproduction/evidence-feedback.mjs +96 -0
- package/src/reproduction/incomplete-initialization.mjs +25 -0
- package/src/reproduction/lifecycle.mjs +253 -0
- package/src/reproduction/plan.mjs +132 -0
- package/src/reproduction/prompts.mjs +70 -0
- package/src/reproduction/runner.mjs +188 -0
- package/src/reproduction/summary.mjs +130 -0
- package/src/reproduction/workspace-mode.mjs +7 -0
- package/src/research/automatic-admission.mjs +156 -0
- package/src/research/compiler-coverage.mjs +85 -0
- package/src/research/compiler-failure.mjs +24 -0
- package/src/research/compiler-normalization-guards.mjs +112 -0
- package/src/research/compiler-repair.mjs +3 -0
- package/src/research/compiler.mjs +853 -0
- package/src/research/continuation-selection.mjs +26 -0
- package/src/research/execution-graph-context.mjs +43 -0
- package/src/research/experiment-importance.mjs +15 -0
- package/src/research/inventory-handoff.mjs +104 -0
- package/src/research/inventory-revisions.mjs +32 -0
- package/src/research/mineru-local.mjs +73 -0
- package/src/research/paper-command.mjs +19 -0
- package/src/research/paper-markdown.mjs +180 -0
- package/src/research/paper-source-map.mjs +69 -0
- package/src/research/planning-policy.mjs +88 -0
- package/src/research/reference-materials.mjs +11 -0
- package/src/research/reproduction-scope.mjs +30 -0
- package/src/research/research-map.mjs +94 -0
- package/src/research/research-objects.mjs +88 -0
- package/src/research/source-discovery.mjs +646 -0
- package/src/research/source-observations.mjs +75 -0
- package/src/research/source-review-cli-mcp.mjs +26 -0
- package/src/research/source-review-input.mjs +209 -0
- package/src/research/source-review-local-codex.mjs +36 -0
- package/src/research/source-review-model.mjs +70 -0
- package/src/research/source-review.mjs +173 -0
- package/src/research/structure.mjs +3163 -0
- package/src/research-card/renderer.mjs +277 -0
- package/src/research-card/verified-conclusion.mjs +143 -0
- package/src/results/output-registry.mjs +183 -0
- package/src/runtime/claude-code.mjs +52 -0
- package/src/runtime/codex-capacity-retry.mjs +87 -0
- package/src/runtime/codex.mjs +64 -0
- package/src/runtime/config.mjs +157 -0
- package/src/runtime/final-output.mjs +40 -0
- package/src/runtime/index.mjs +21 -0
- package/src/runtime/local-codex.mjs +74 -0
- package/src/runtime/opencode.mjs +95 -0
- package/src/runtime/prompt.mjs +13 -0
- package/src/sandbox/docker.mjs +363 -0
- package/src/settings/command.mjs +297 -0
- package/src/settings/store.mjs +119 -0
- package/src/telemetry/pricing.mjs +68 -0
- package/src/telemetry/usage.mjs +265 -0
- package/src/terminal/events.mjs +97 -0
- package/src/terminal/input.mjs +40 -0
- package/src/terminal/plain.mjs +40 -0
- package/src/terminal/remote-stream.mjs +22 -0
- package/src/terminal/screen.mjs +214 -0
- package/src/terminal/transcript.mjs +69 -0
- package/src/util.mjs +107 -0
- package/src/verification/ai-assessor.mjs +534 -0
- package/src/verification/claim-evaluator.mjs +242 -0
- package/src/verification/evidence-context.mjs +165 -0
- package/src/verification/evidence-reader.mjs +95 -0
- package/src/verification/integrity.mjs +570 -0
- package/src/verification/tolerance.mjs +32 -0
- package/src/workloads/cpu-research-preparation.mjs +56 -0
- package/src/workloads/definition.mjs +74 -0
- package/src/workloads/phase-aware-reproduction.mjs +46 -0
- package/src/workloads/reproduction.mjs +85 -0
- package/src/workspace/command.mjs +242 -0
- package/src/workspace/control.mjs +49 -0
- package/src/workspace/entry.mjs +28 -0
- package/src/workspace/input.mjs +93 -0
- package/src/workspace/interactive.mjs +94 -0
- package/src/workspace/jobs.mjs +418 -0
- package/src/workspace/session.mjs +97 -0
- package/src/workspace/worker.mjs +137 -0
- package/ui/arkgraph/ambient-motion.mjs +10 -0
- package/ui/arkgraph/app.jsx +153 -0
- package/ui/arkgraph/boot.js +6 -0
- package/ui/arkgraph/camera-motion.mjs +20 -0
- package/ui/arkgraph/context-reveal.mjs +39 -0
- package/ui/arkgraph/details.css +3 -0
- package/ui/arkgraph/entry.jsx +28 -0
- package/ui/arkgraph/experiment-curves.mjs +17 -0
- package/ui/arkgraph/experiment-selection.mjs +15 -0
- package/ui/arkgraph/experiment-style.css +26 -0
- package/ui/arkgraph/experiment-ui.jsx +32 -0
- package/ui/arkgraph/frame.html +1 -0
- package/ui/arkgraph/graph-gestures.mjs +62 -0
- package/ui/arkgraph/label-layout.mjs +57 -0
- package/ui/arkgraph/locales/en.json +229 -0
- package/ui/arkgraph/locales/source-types.json +15 -0
- package/ui/arkgraph/localization-build.mjs +27 -0
- package/ui/arkgraph/material-build.mjs +23 -0
- package/ui/arkgraph/material-colors.mjs +39 -0
- package/ui/arkgraph/material-style.css +15 -0
- package/ui/arkgraph/open-graph.jsx +326 -0
- package/ui/arkgraph/outline.jsx +49 -0
- package/ui/arkgraph/package-lock.json +888 -0
- package/ui/arkgraph/package.json +17 -0
- package/ui/arkgraph/reading-layout.mjs +130 -0
- package/ui/arkgraph/reading-presentation.mjs +73 -0
- package/ui/arkgraph/record-detail.css +51 -0
- package/ui/arkgraph/record-details.jsx +29 -0
- package/ui/arkgraph/research-types.mjs +31 -0
- package/ui/arkgraph/selection-mark.jsx +6 -0
- package/ui/arkgraph/soft-spine.mjs +26 -0
- package/ui/arkgraph/steering-style.css +187 -0
- package/ui/arkgraph/style.css +272 -0
|
@@ -0,0 +1,570 @@
|
|
|
1
|
+
import { lstat, readFile, realpath } from "node:fs/promises";
|
|
2
|
+
import path from "node:path";
|
|
3
|
+
|
|
4
|
+
import { executionContractMeasurements } from "../contracts/execution-contract.mjs";
|
|
5
|
+
import { agentPlansExecution } from "../contracts/execution-mode.mjs";
|
|
6
|
+
import { loadAndVerifyRunnerAudit } from "../execution/runner-audit.mjs";
|
|
7
|
+
import { pathExists, readJson, sha256File, sha256Value } from "../util.mjs";
|
|
8
|
+
|
|
9
|
+
export async function checkExecutionIntegrity({
|
|
10
|
+
research,
|
|
11
|
+
contract,
|
|
12
|
+
runDirectory,
|
|
13
|
+
result,
|
|
14
|
+
metrics,
|
|
15
|
+
outputs,
|
|
16
|
+
sharedExecution = null,
|
|
17
|
+
}) {
|
|
18
|
+
const checks = [];
|
|
19
|
+
const runPath = path.join(runDirectory, "run.json");
|
|
20
|
+
const run = (await pathExists(runPath)) ? await readJson(runPath) : null;
|
|
21
|
+
check(checks, "run_record_present", Boolean(run), "Out-of-sandbox run record exists");
|
|
22
|
+
const experiment = research.experiments.find((item) => item.versionId === contract.research.experimentVersionId);
|
|
23
|
+
const claim = research.claims.find((item) => item.versionId === contract.research.claimVersionId);
|
|
24
|
+
const executionStatus = result?.execution?.status ?? "unknown";
|
|
25
|
+
const scientificExecutionSucceeded = executionStatus === "succeeded";
|
|
26
|
+
const agentPlanned = agentPlansExecution(contract?.protocol);
|
|
27
|
+
check(checks, "claim_binding", Boolean(claim), "contract claimVersionId exists in research.json");
|
|
28
|
+
check(checks, "experiment_binding", Boolean(experiment), "contract experimentVersionId exists in research.json");
|
|
29
|
+
if (sharedExecution) {
|
|
30
|
+
if (sharedExecution.campaignTargetContractDigest) check(checks, "campaign_target_commitment",
|
|
31
|
+
sharedExecution.campaignTargetContractDigest === contract.contractDigest,
|
|
32
|
+
"The original shared execution committed to this target contract before execution");
|
|
33
|
+
const sourceMeasurementIds = new Set(sharedExecution.sourceMeasurementIds ?? []);
|
|
34
|
+
check(
|
|
35
|
+
checks,
|
|
36
|
+
"shared_execution_experiment_binding",
|
|
37
|
+
sharedExecution.experimentVersionId === contract.research.experimentVersionId,
|
|
38
|
+
"Shared execution and claim projection reference the same immutable experiment version",
|
|
39
|
+
sharedExecution,
|
|
40
|
+
);
|
|
41
|
+
check(
|
|
42
|
+
checks,
|
|
43
|
+
"shared_execution_measurement_scope",
|
|
44
|
+
(sharedExecution.projectedMeasurementIds ?? []).every((id) =>
|
|
45
|
+
sourceMeasurementIds.has(id)),
|
|
46
|
+
"Every projected claim measurement was assigned to the original shared execution",
|
|
47
|
+
sharedExecution,
|
|
48
|
+
);
|
|
49
|
+
}
|
|
50
|
+
check(checks, "contract_digest", metrics.contractDigest === contract.contractDigest, "metrics are bound to the contract digest");
|
|
51
|
+
check(checks, "output_registry_contract", outputs?.contractDigest === contract.contractDigest, "typed outputs are bound to the contract digest");
|
|
52
|
+
check(checks, "output_registry_run", outputs?.runId === run?.runId, "typed outputs are bound to the execution run");
|
|
53
|
+
check(checks, "execution_status_recognized", new Set(["succeeded", "failed", "partial"]).has(executionStatus), "Execution report uses a recognized status", { observed: executionStatus });
|
|
54
|
+
check(checks, "scientific_execution_succeeded", scientificExecutionSucceeded, "The scientific execution produced a successful outcome", { observed: executionStatus }, false);
|
|
55
|
+
|
|
56
|
+
const runnerAudit = await loadAndVerifyRunnerAudit(runDirectory);
|
|
57
|
+
const isolatedCapture = new Set([
|
|
58
|
+
"host-runner-outside-agent-mounts",
|
|
59
|
+
"gcp-batch-supervisor-separate-uid",
|
|
60
|
+
"autodl-pro-supervisor-separate-uid",
|
|
61
|
+
]).has(runnerAudit.audit?.captureBoundary);
|
|
62
|
+
check(checks, "runner_audit_present", Boolean(runnerAudit.audit), "Runner-owned execution audit exists outside the Agent result envelope");
|
|
63
|
+
check(checks, "runner_audit_integrity", runnerAudit.issues.length === 0, "Runner audit digest and companion file hashes verify", { issues: runnerAudit.issues });
|
|
64
|
+
check(checks, "runner_capture_isolated", isolatedCapture, "The Agent could not write the runner audit directory while commands were captured", { captureBoundary: runnerAudit.audit?.captureBoundary ?? null });
|
|
65
|
+
check(checks, "runner_raw_stream_digest", Boolean(runnerAudit.audit?.rawStreams?.stdout?.sha256), "The runner retained the raw runtime stdout digest", { rawStreams: runnerAudit.audit?.rawStreams ?? null });
|
|
66
|
+
check(checks, "runner_command_records", runnerAudit.commandRecords.length > 0, "Runner-captured runtime tool-command records exist", { count: runnerAudit.commandRecords.length }, scientificExecutionSucceeded);
|
|
67
|
+
if (contract.executionPlan) {
|
|
68
|
+
check(checks, "research_plan_binding", metrics.executionPlanDigest === contract.executionPlan.digest,
|
|
69
|
+
"Parsed evidence identifies the exact research plan revision");
|
|
70
|
+
const captured = runnerAudit.commandRecords.some((record) =>
|
|
71
|
+
(record.stdout?.text ?? "").split("\n").some((line) => {
|
|
72
|
+
try {
|
|
73
|
+
const event = JSON.parse(line);
|
|
74
|
+
return event.event === "research_plan_execution"
|
|
75
|
+
&& event.planDigest === contract.executionPlan.digest
|
|
76
|
+
&& `sha256:${sha256Value(event.plan)}` === contract.executionPlan.digest;
|
|
77
|
+
} catch { return false; }
|
|
78
|
+
}));
|
|
79
|
+
check(checks, "research_plan_captured", captured,
|
|
80
|
+
"The runner captured the final plan bytes; the assessor must still verify their relationship to actual computation", {}, scientificExecutionSucceeded);
|
|
81
|
+
}
|
|
82
|
+
|
|
83
|
+
const contractMeasurements = executionContractMeasurements(contract);
|
|
84
|
+
const parsedMeasurements = Array.isArray(metrics.measurements)
|
|
85
|
+
? metrics.measurements
|
|
86
|
+
: [{ ...metrics, measurementId: contractMeasurements[0]?.measurementId }];
|
|
87
|
+
for (const [index, binding] of contractMeasurements.entries()) {
|
|
88
|
+
const parsed = parsedMeasurements.find((item) => item.measurementId === binding.measurementId);
|
|
89
|
+
const label = `${index + 1}:${binding.measurementId}`;
|
|
90
|
+
check(checks, `measurement_record:${label}`, Boolean(parsed), "A parsed measurement record exists for every contract measurement");
|
|
91
|
+
if (!parsed) continue;
|
|
92
|
+
if (parsed.sourceEvidence) {
|
|
93
|
+
const evidencePath = path.join(runDirectory, "output", parsed.sourceEvidence.path);
|
|
94
|
+
const evidenceDigest = (await pathExists(evidencePath)) ? await sha256File(evidencePath) : null;
|
|
95
|
+
check(checks, `evidence_digest:${label}`, evidenceDigest === parsed.sourceEvidence.sha256, "Raw evidence hash matches metrics.json", { observed: evidenceDigest, expected: parsed.sourceEvidence.sha256 });
|
|
96
|
+
} else {
|
|
97
|
+
check(checks, `metric_source_absence_explained:${label}`, !scientificExecutionSucceeded && parsed.availability?.status === "unavailable", "A missing metric source is allowed only for an explicitly unsuccessful execution", { availability: parsed.availability ?? null });
|
|
98
|
+
}
|
|
99
|
+
if (parsed.primary) {
|
|
100
|
+
check(checks, `metric_binding:${label}`, parsed.primary.metric === binding.metric && parsed.primary.unit === binding.unit, "Parsed metric matches the contract");
|
|
101
|
+
} else {
|
|
102
|
+
check(checks, `metric_absence_explained:${label}`, !scientificExecutionSucceeded && parsed.availability?.status === "unavailable", "A missing primary metric is allowed only for an explicitly unsuccessful execution", { availability: parsed.availability ?? null });
|
|
103
|
+
}
|
|
104
|
+
}
|
|
105
|
+
|
|
106
|
+
for (const [index, output] of (outputs?.outputs ?? []).entries()) {
|
|
107
|
+
const label = `${index + 1}:${output.id}`;
|
|
108
|
+
const outputPath = path.join(runDirectory, "output", output.sourcePath);
|
|
109
|
+
const present = await pathExists(outputPath);
|
|
110
|
+
check(checks, `output_present:${label}`, present, "Declared scientific output exists as a regular runner file");
|
|
111
|
+
if (!present) continue;
|
|
112
|
+
const observedDigest = `sha256:${await sha256File(outputPath)}`;
|
|
113
|
+
check(checks, `output_digest:${label}`, observedDigest === output.digest, "Scientific output hash matches results/outputs.json", {
|
|
114
|
+
observed: observedDigest,
|
|
115
|
+
expected: output.digest,
|
|
116
|
+
mediaType: output.mediaType,
|
|
117
|
+
outputKind: output.kind,
|
|
118
|
+
});
|
|
119
|
+
}
|
|
120
|
+
|
|
121
|
+
const observedCommit = run?.repository?.commit ?? null;
|
|
122
|
+
check(checks, "repository_commit", observedCommit === contract.repository.commit, "Observed repository commit matches the contract-pinned value", { observed: observedCommit, expected: contract.repository.commit });
|
|
123
|
+
const scientificChangePolicy = evaluateScientificChangePolicy({ contract, experiment, result });
|
|
124
|
+
check(
|
|
125
|
+
checks,
|
|
126
|
+
"scientific_change_policy",
|
|
127
|
+
scientificChangePolicy.passed,
|
|
128
|
+
"Scientific-semantic changes are absent or limited to paths pre-declared by a CiteArk reconstruction contract",
|
|
129
|
+
scientificChangePolicy,
|
|
130
|
+
);
|
|
131
|
+
|
|
132
|
+
const patchPath = path.join(runDirectory, "execution", "repository.patch");
|
|
133
|
+
const patchPresent = await pathExists(patchPath);
|
|
134
|
+
const patch = patchPresent ? await readFile(patchPath, "utf8") : "";
|
|
135
|
+
const nonGitSnapshot = run?.repository?.sourceType === "pipeline-snapshot";
|
|
136
|
+
const changedTrackedPaths = [...patch.matchAll(/^diff --git a\/(.+?) b\//gm)].map((match) => match[1]);
|
|
137
|
+
const workspaceMutationPolicy = evaluateWorkspaceMutationPolicy({
|
|
138
|
+
experiment,
|
|
139
|
+
result,
|
|
140
|
+
allowScientificChanges: Boolean(contract.executionPlan),
|
|
141
|
+
changedPaths: changedTrackedPaths,
|
|
142
|
+
});
|
|
143
|
+
check(
|
|
144
|
+
checks,
|
|
145
|
+
"source_patch_accounted",
|
|
146
|
+
(patchPresent || nonGitSnapshot) &&
|
|
147
|
+
(patch.trim() === "" || workspaceMutationPolicy.passed),
|
|
148
|
+
"Git source patch collected; non-Git snapshots are covered by before/after file-tree records; source repairs must be declared by the Agent or pre-authorized by the contract",
|
|
149
|
+
{
|
|
150
|
+
patchSha256: patch ? `sha256:${sha256Value(patch)}` : null,
|
|
151
|
+
nonGitSnapshot,
|
|
152
|
+
changedTrackedPaths,
|
|
153
|
+
...workspaceMutationPolicy,
|
|
154
|
+
},
|
|
155
|
+
);
|
|
156
|
+
for (const filename of ["invocation.json", "workspace-before.json", "workspace-after.json", "workspace-changes.json"]) {
|
|
157
|
+
check(checks, `execution_record:${filename}`, await pathExists(path.join(runDirectory, "execution", filename)), `Out-of-sandbox record exists: ${filename}`);
|
|
158
|
+
}
|
|
159
|
+
const changesPath = path.join(runDirectory, "execution", "workspace-changes.json");
|
|
160
|
+
let workspaceChanges = null;
|
|
161
|
+
if (await pathExists(changesPath)) {
|
|
162
|
+
workspaceChanges = await readJson(changesPath);
|
|
163
|
+
const changedPaths = [...(workspaceChanges.modified ?? []), ...(workspaceChanges.deleted ?? [])]
|
|
164
|
+
.map((item) => item.path);
|
|
165
|
+
const mutationPolicy = evaluateWorkspaceMutationPolicy({ experiment, result, changedPaths,
|
|
166
|
+
allowScientificChanges: Boolean(contract.executionPlan) });
|
|
167
|
+
check(
|
|
168
|
+
checks,
|
|
169
|
+
"workspace_mutations_allowed",
|
|
170
|
+
mutationPolicy.passed,
|
|
171
|
+
"Source modifications and deletions are declared as execution/environment repairs or pre-authorized by the scientific contract",
|
|
172
|
+
mutationPolicy,
|
|
173
|
+
);
|
|
174
|
+
}
|
|
175
|
+
|
|
176
|
+
const commands = runnerAudit.commandRecords
|
|
177
|
+
.map((record) => record?.command?.text)
|
|
178
|
+
.filter((command) => typeof command === "string" && command.length > 0);
|
|
179
|
+
const pinnedEntrypoint = await inspectPinnedEntrypointEvidence({
|
|
180
|
+
runDirectory,
|
|
181
|
+
entrypoint: experiment?.protocol?.entrypoint,
|
|
182
|
+
commandRecords: runnerAudit.commandRecords,
|
|
183
|
+
observedCommit,
|
|
184
|
+
expectedCommit: contract.repository.commit,
|
|
185
|
+
workspaceChanges,
|
|
186
|
+
scientificExecutionSucceeded,
|
|
187
|
+
});
|
|
188
|
+
for (const fragment of contract.protocol?.requiredCommandFragments ?? []) {
|
|
189
|
+
const evidence = protocolFragmentEvidence({
|
|
190
|
+
commands,
|
|
191
|
+
fragment,
|
|
192
|
+
pinnedEntrypoint: agentPlanned ? null : pinnedEntrypoint,
|
|
193
|
+
});
|
|
194
|
+
check(
|
|
195
|
+
checks,
|
|
196
|
+
`protocol_fragment:${fragment}`,
|
|
197
|
+
evidence.passed,
|
|
198
|
+
`Command anchor ${fragment} is a lexical review hint, not proof of scientific execution`,
|
|
199
|
+
{ ...evidence.detail, evidentiaryRole: "advisory-only; assessor must trace actual computation" },
|
|
200
|
+
false,
|
|
201
|
+
);
|
|
202
|
+
}
|
|
203
|
+
for (const [index, measurement] of contractMeasurements.entries()) {
|
|
204
|
+
const evidenceDeclared = Array.isArray(result?.evidence) && result.evidence.some((item) => item?.path === measurement.parser.evidencePath);
|
|
205
|
+
check(checks, `parser_evidence_declared:${index + 1}:${measurement.measurementId}`, evidenceDeclared, "Agent result declares the raw evidence used by the parser", {}, scientificExecutionSucceeded);
|
|
206
|
+
}
|
|
207
|
+
|
|
208
|
+
const requiredChecks = checks.filter((item) => item.required !== false);
|
|
209
|
+
const status = requiredChecks.every((item) => item.status === "passed") ? "passed" : "failed";
|
|
210
|
+
const agentStatus = normalizeAgentStatus(run?.status);
|
|
211
|
+
const packagingRecovered = await pathExists(path.join(runDirectory, "output", "packaging-recovery.txt"));
|
|
212
|
+
return {
|
|
213
|
+
schemaVersion: "0.1",
|
|
214
|
+
contractDigest: contract.contractDigest,
|
|
215
|
+
status,
|
|
216
|
+
checks,
|
|
217
|
+
states: {
|
|
218
|
+
experimentStatus: result?.execution?.status ?? "unknown",
|
|
219
|
+
agentStatus,
|
|
220
|
+
packagingStatus: packagingRecovered ? "recovered" : result ? "completed" : "missing",
|
|
221
|
+
verificationStatus: "pending",
|
|
222
|
+
publicationStatus: "draft",
|
|
223
|
+
commandProvenance: runnerAudit.audit?.status === "complete" && isolatedCapture
|
|
224
|
+
? "runner-captured"
|
|
225
|
+
: "degraded",
|
|
226
|
+
},
|
|
227
|
+
executionAudit: runnerAudit.audit ? {
|
|
228
|
+
auditDigest: runnerAudit.audit.auditDigest,
|
|
229
|
+
captureBoundary: runnerAudit.audit.captureBoundary,
|
|
230
|
+
commandRecordCount: runnerAudit.audit.commandRecordCount,
|
|
231
|
+
} : null,
|
|
232
|
+
...(sharedExecution ? { sharedExecution } : {}),
|
|
233
|
+
checkedAt: run?.finishedAt ?? new Date().toISOString(),
|
|
234
|
+
};
|
|
235
|
+
}
|
|
236
|
+
|
|
237
|
+
export async function inspectPinnedEntrypointEvidence({
|
|
238
|
+
runDirectory,
|
|
239
|
+
entrypoint,
|
|
240
|
+
commandRecords,
|
|
241
|
+
observedCommit,
|
|
242
|
+
expectedCommit,
|
|
243
|
+
workspaceChanges,
|
|
244
|
+
scientificExecutionSucceeded,
|
|
245
|
+
}) {
|
|
246
|
+
const normalizedEntrypoint = extractEntrypointPath(entrypoint);
|
|
247
|
+
const detail = {
|
|
248
|
+
eligible: false,
|
|
249
|
+
entrypoint: normalizedEntrypoint,
|
|
250
|
+
entrypointSha256: null,
|
|
251
|
+
invocationRecordIds: [],
|
|
252
|
+
reason: null,
|
|
253
|
+
};
|
|
254
|
+
if (!scientificExecutionSucceeded) return { ...detail, reason: "scientific execution did not succeed", sourceText: null };
|
|
255
|
+
if (!normalizedEntrypoint) return { ...detail, reason: "entrypoint is not a safe relative path", sourceText: null };
|
|
256
|
+
if (!observedCommit || observedCommit !== expectedCommit) {
|
|
257
|
+
return { ...detail, reason: "observed repository commit does not match the contract", sourceText: null };
|
|
258
|
+
}
|
|
259
|
+
|
|
260
|
+
const successfulInvocations = (Array.isArray(commandRecords) ? commandRecords : []).filter(
|
|
261
|
+
(record) => commandRecordSuccessfullyInvokesEntrypoint(record, normalizedEntrypoint),
|
|
262
|
+
);
|
|
263
|
+
detail.invocationRecordIds = successfulInvocations.map((record) => record.recordId).filter(Boolean);
|
|
264
|
+
if (!successfulInvocations.length) {
|
|
265
|
+
return { ...detail, reason: "runner did not capture a successful entrypoint invocation", sourceText: null };
|
|
266
|
+
}
|
|
267
|
+
|
|
268
|
+
const changedEntrypoint = [...(workspaceChanges?.modified ?? []), ...(workspaceChanges?.deleted ?? [])]
|
|
269
|
+
.some((item) => item?.path === normalizedEntrypoint);
|
|
270
|
+
if (changedEntrypoint) {
|
|
271
|
+
return { ...detail, reason: "entrypoint was modified or deleted during execution", sourceText: null };
|
|
272
|
+
}
|
|
273
|
+
|
|
274
|
+
const repositoryRoot = path.join(runDirectory, "workspace", "repository");
|
|
275
|
+
const entrypointPath = path.resolve(repositoryRoot, normalizedEntrypoint);
|
|
276
|
+
const rootPrefix = `${path.resolve(repositoryRoot)}${path.sep}`;
|
|
277
|
+
if (!entrypointPath.startsWith(rootPrefix)) {
|
|
278
|
+
return { ...detail, reason: "entrypoint resolves outside the repository", sourceText: null };
|
|
279
|
+
}
|
|
280
|
+
const [resolvedRoot, resolvedEntrypoint] = await Promise.all([
|
|
281
|
+
realpath(repositoryRoot).catch(() => null),
|
|
282
|
+
realpath(entrypointPath).catch(() => null),
|
|
283
|
+
]);
|
|
284
|
+
if (!resolvedRoot || !resolvedEntrypoint || !resolvedEntrypoint.startsWith(`${resolvedRoot}${path.sep}`)) {
|
|
285
|
+
return { ...detail, reason: "entrypoint source is missing or escapes the repository", sourceText: null };
|
|
286
|
+
}
|
|
287
|
+
const stat = await lstat(resolvedEntrypoint).catch(() => null);
|
|
288
|
+
if (!stat?.isFile() || stat.size > 8 * 1024 * 1024) {
|
|
289
|
+
return { ...detail, reason: "entrypoint source is not a bounded regular file", sourceText: null };
|
|
290
|
+
}
|
|
291
|
+
|
|
292
|
+
const beforePath = path.join(runDirectory, "execution", "workspace-before.json");
|
|
293
|
+
const before = (await pathExists(beforePath)) ? await readJson(beforePath) : null;
|
|
294
|
+
const beforeEntry = before?.entries?.find((item) => item?.path === normalizedEntrypoint && item?.type === "file");
|
|
295
|
+
const entrypointSha256 = await sha256File(resolvedEntrypoint);
|
|
296
|
+
detail.entrypointSha256 = entrypointSha256;
|
|
297
|
+
if (!beforeEntry || beforeEntry.sha256 !== entrypointSha256) {
|
|
298
|
+
return { ...detail, reason: "entrypoint source does not match the runner-owned pre-execution snapshot", sourceText: null };
|
|
299
|
+
}
|
|
300
|
+
|
|
301
|
+
return {
|
|
302
|
+
...detail,
|
|
303
|
+
eligible: true,
|
|
304
|
+
reason: null,
|
|
305
|
+
sourceText: await readFile(resolvedEntrypoint, "utf8"),
|
|
306
|
+
};
|
|
307
|
+
}
|
|
308
|
+
|
|
309
|
+
export function protocolFragmentEvidence({ commands, fragment, pinnedEntrypoint }) {
|
|
310
|
+
if (commandMatchesFragment(commands, fragment)) {
|
|
311
|
+
return {
|
|
312
|
+
passed: true,
|
|
313
|
+
detail: { provenance: "runner-captured-runtime-json-stream" },
|
|
314
|
+
};
|
|
315
|
+
}
|
|
316
|
+
if (pinnedEntrypoint?.eligible && pinnedEntrypoint.sourceText?.includes(fragment)) {
|
|
317
|
+
return {
|
|
318
|
+
passed: true,
|
|
319
|
+
detail: {
|
|
320
|
+
provenance: "pinned-entrypoint-source-plus-runner-invocation",
|
|
321
|
+
entrypoint: pinnedEntrypoint.entrypoint,
|
|
322
|
+
entrypointSha256: pinnedEntrypoint.entrypointSha256,
|
|
323
|
+
invocationRecordIds: pinnedEntrypoint.invocationRecordIds,
|
|
324
|
+
},
|
|
325
|
+
};
|
|
326
|
+
}
|
|
327
|
+
return {
|
|
328
|
+
passed: false,
|
|
329
|
+
detail: {
|
|
330
|
+
provenance: "unverified",
|
|
331
|
+
entrypoint: pinnedEntrypoint?.entrypoint ?? null,
|
|
332
|
+
entrypointSha256: pinnedEntrypoint?.entrypointSha256 ?? null,
|
|
333
|
+
entrypointEvidenceReason: pinnedEntrypoint?.reason ?? "no pinned entrypoint evidence is available",
|
|
334
|
+
},
|
|
335
|
+
};
|
|
336
|
+
}
|
|
337
|
+
|
|
338
|
+
export function evaluateScientificChangePolicy({ contract, experiment, result }) {
|
|
339
|
+
const scientificChanges = (Array.isArray(result?.codeChanges) ? result.codeChanges : [])
|
|
340
|
+
.filter((change) => change?.classification === "scientific");
|
|
341
|
+
const scientificChangePaths = [...new Set(scientificChanges
|
|
342
|
+
.map((change) => normalizeRelativePath(change?.path))
|
|
343
|
+
.filter(Boolean))];
|
|
344
|
+
const invalidScientificChangePaths = scientificChanges
|
|
345
|
+
.filter((change) => !normalizeRelativePath(change?.path))
|
|
346
|
+
.map((change) => typeof change?.path === "string" ? change.path : null);
|
|
347
|
+
const reconstructionAuthorized = contract?.repository?.implementationOrigin === "citeark_reconstruction"
|
|
348
|
+
&& new Set(["exact", "faithful", "approximate", "proxy"]).has(contract?.reconstructionFidelity);
|
|
349
|
+
const entrypoint = extractEntrypointPath(experiment?.protocol?.entrypoint);
|
|
350
|
+
const allowedChangePaths = [...new Set(
|
|
351
|
+
(experiment?.protocol?.allowedChangePaths ?? []).map(normalizeRelativePath).filter(Boolean),
|
|
352
|
+
)];
|
|
353
|
+
const allowedScientificChangePaths = [...new Set([entrypoint, ...allowedChangePaths].filter(Boolean))];
|
|
354
|
+
const unauthorizedScientificChangePaths = contract.executionPlan ? [] : scientificChangePaths.filter(
|
|
355
|
+
(changedPath) => !reconstructionAuthorized
|
|
356
|
+
|| !(
|
|
357
|
+
changePathMatchesEntrypoint(changedPath, entrypoint)
|
|
358
|
+
|| allowedChangePaths.some((allowedPath) => pathAllowed(changedPath, allowedPath))
|
|
359
|
+
),
|
|
360
|
+
);
|
|
361
|
+
return {
|
|
362
|
+
passed: invalidScientificChangePaths.length === 0 && unauthorizedScientificChangePaths.length === 0,
|
|
363
|
+
reconstructionAuthorized,
|
|
364
|
+
scientificChangePaths,
|
|
365
|
+
invalidScientificChangePaths,
|
|
366
|
+
allowedScientificChangePaths,
|
|
367
|
+
unauthorizedScientificChangePaths,
|
|
368
|
+
};
|
|
369
|
+
}
|
|
370
|
+
|
|
371
|
+
export function evaluateWorkspaceMutationPolicy({ experiment, result, changedPaths, allowScientificChanges = false }) {
|
|
372
|
+
const observedChangePaths = [...new Set(
|
|
373
|
+
(Array.isArray(changedPaths) ? changedPaths : [])
|
|
374
|
+
.map(normalizeWorkspaceChangePath)
|
|
375
|
+
.filter(Boolean),
|
|
376
|
+
)];
|
|
377
|
+
const invalidObservedChangePaths = (Array.isArray(changedPaths) ? changedPaths : [])
|
|
378
|
+
.filter((changedPath) => !normalizeWorkspaceChangePath(changedPath));
|
|
379
|
+
const allowedChangePaths = [...new Set(
|
|
380
|
+
(experiment?.protocol?.allowedChangePaths ?? [])
|
|
381
|
+
.map(normalizeWorkspaceChangePath)
|
|
382
|
+
.filter(Boolean),
|
|
383
|
+
)];
|
|
384
|
+
const declaredRepairPaths = [...new Set(
|
|
385
|
+
(Array.isArray(result?.codeChanges) ? result.codeChanges : [])
|
|
386
|
+
.filter((change) => new Set(allowScientificChanges ? ["environment", "execution", "scientific"] : ["environment", "execution"]).has(change?.classification))
|
|
387
|
+
.map((change) => normalizeWorkspaceChangePath(change?.path))
|
|
388
|
+
.filter(Boolean),
|
|
389
|
+
)];
|
|
390
|
+
const unaccountedChangePaths = observedChangePaths.filter((changedPath) =>
|
|
391
|
+
!declaredRepairPaths.includes(changedPath)
|
|
392
|
+
&& !allowedChangePaths.some((allowedPath) => pathAllowed(changedPath, allowedPath)),
|
|
393
|
+
);
|
|
394
|
+
return {
|
|
395
|
+
passed: invalidObservedChangePaths.length === 0 && unaccountedChangePaths.length === 0,
|
|
396
|
+
observedChangePaths,
|
|
397
|
+
invalidObservedChangePaths,
|
|
398
|
+
declaredRepairPaths,
|
|
399
|
+
allowedChangePaths,
|
|
400
|
+
unaccountedChangePaths,
|
|
401
|
+
};
|
|
402
|
+
}
|
|
403
|
+
|
|
404
|
+
function changePathMatchesEntrypoint(changedPath, entrypoint) {
|
|
405
|
+
if (!changedPath || !entrypoint) return false;
|
|
406
|
+
return changedPath === entrypoint
|
|
407
|
+
|| changedPath === `output/${entrypoint}`
|
|
408
|
+
|| changedPath === `repository/${entrypoint}`;
|
|
409
|
+
}
|
|
410
|
+
|
|
411
|
+
/**
|
|
412
|
+
* Research plans historically used protocol.entrypoint for either a path or a
|
|
413
|
+
* full invocation. Integrity policy is path-based, so never compare a changed
|
|
414
|
+
* file with the entire command string.
|
|
415
|
+
*/
|
|
416
|
+
export function extractEntrypointPath(value) {
|
|
417
|
+
if (typeof value !== "string" || !value.trim()) return null;
|
|
418
|
+
const trimmed = value.trim();
|
|
419
|
+
if (!/[\s;&|<>]/.test(trimmed)) return normalizeEntrypointLogicalPath(stripShellQuotes(trimmed));
|
|
420
|
+
const script = shellWords(trimmed).find((token) =>
|
|
421
|
+
!token.startsWith("-")
|
|
422
|
+
&& /\.(?:py|sh|js|mjs|cjs|rb|pl|jl|r)$/i.test(token),
|
|
423
|
+
);
|
|
424
|
+
return normalizeEntrypointLogicalPath(script);
|
|
425
|
+
}
|
|
426
|
+
|
|
427
|
+
export function commandMatchesFragment(commands, fragment) {
|
|
428
|
+
if (!Array.isArray(commands) || typeof fragment !== "string" || fragment.length === 0) return false;
|
|
429
|
+
const requiredTokens = anchorWords(fragment);
|
|
430
|
+
if (requiredTokens.length && commands.some((command) => {
|
|
431
|
+
if (typeof command !== "string") return false;
|
|
432
|
+
const tokens = anchorWords(command);
|
|
433
|
+
return tokens.some((_, index) => requiredTokens.every((token, offset) => tokens[index + offset] === token));
|
|
434
|
+
})) return true;
|
|
435
|
+
|
|
436
|
+
const required = fragment.match(/^bash\s+([^\s;&|]+)$/);
|
|
437
|
+
if (!required) return false;
|
|
438
|
+
const requiredScript = path.posix.normalize(required[1]);
|
|
439
|
+
const requiredDirectory = path.posix.dirname(requiredScript);
|
|
440
|
+
const requiredBasename = path.posix.basename(requiredScript);
|
|
441
|
+
|
|
442
|
+
return commands.some((command) => {
|
|
443
|
+
if (typeof command !== "string") return false;
|
|
444
|
+
let workingDirectory = null;
|
|
445
|
+
for (const segment of command.split(/&&|;/).map((item) => item.trim())) {
|
|
446
|
+
const cd = segment.match(/^cd\s+([^\s]+)$/);
|
|
447
|
+
if (cd) {
|
|
448
|
+
workingDirectory = stripShellQuotes(cd[1]);
|
|
449
|
+
continue;
|
|
450
|
+
}
|
|
451
|
+
const bash = segment.match(/^bash\s+([^\s]+)(?:\s|$)/);
|
|
452
|
+
if (!bash || !workingDirectory) continue;
|
|
453
|
+
const invokedScript = stripShellQuotes(bash[1]);
|
|
454
|
+
if (path.posix.basename(invokedScript) !== requiredBasename) continue;
|
|
455
|
+
const resolvedDirectory = path.posix.dirname(path.posix.resolve(workingDirectory, invokedScript));
|
|
456
|
+
if (resolvedDirectory === requiredDirectory || resolvedDirectory.endsWith(`/${requiredDirectory}`)) return true;
|
|
457
|
+
}
|
|
458
|
+
return false;
|
|
459
|
+
});
|
|
460
|
+
}
|
|
461
|
+
|
|
462
|
+
// Used only for advisory matching, never to authenticate execution.
|
|
463
|
+
function anchorWords(value) {
|
|
464
|
+
const words = [];
|
|
465
|
+
let word = "", quote = null;
|
|
466
|
+
const flush = () => {
|
|
467
|
+
if (word) words.push(...(word.startsWith("--") && word.includes("=")
|
|
468
|
+
? [word.slice(0, word.indexOf("=")), word.slice(word.indexOf("=") + 1)] : [word]));
|
|
469
|
+
word = "";
|
|
470
|
+
};
|
|
471
|
+
for (let index = 0; index < value.length; index++) {
|
|
472
|
+
const char = value[index];
|
|
473
|
+
if (quote) {
|
|
474
|
+
if (char === quote) quote = null;
|
|
475
|
+
else word += char;
|
|
476
|
+
} else if (char === "#" && !word) {
|
|
477
|
+
while (index < value.length && value[index] !== "\n") index++;
|
|
478
|
+
flush();
|
|
479
|
+
} else if (char === "'" || char === '"') quote = char;
|
|
480
|
+
else if (/\s/.test(char)) flush();
|
|
481
|
+
else word += char;
|
|
482
|
+
}
|
|
483
|
+
flush();
|
|
484
|
+
return words;
|
|
485
|
+
}
|
|
486
|
+
|
|
487
|
+
export function commandInvokesEntrypoint(command, entrypoint) {
|
|
488
|
+
const normalizedEntrypoint = normalizeRelativePath(entrypoint);
|
|
489
|
+
if (typeof command !== "string" || !normalizedEntrypoint) return false;
|
|
490
|
+
const interpreters = new Set(["bash", "sh", "dash", "zsh", "python", "python3", "perl", "ruby", "node"]);
|
|
491
|
+
for (const segment of command.split(/&&|\|\||;|\|/).map((item) => item.trim()).filter(Boolean)) {
|
|
492
|
+
const tokens = shellWords(segment);
|
|
493
|
+
while (tokens.length && /^[A-Za-z_][A-Za-z0-9_]*=/.test(tokens[0])) tokens.shift();
|
|
494
|
+
if (!tokens.length) continue;
|
|
495
|
+
if (tokenMatchesEntrypoint(tokens[0], normalizedEntrypoint)) return true;
|
|
496
|
+
const executable = path.posix.basename(tokens[0]);
|
|
497
|
+
if (!interpreters.has(executable)) continue;
|
|
498
|
+
const script = tokens.slice(1).find((token) => !token.startsWith("-"));
|
|
499
|
+
if (script && tokenMatchesEntrypoint(script, normalizedEntrypoint)) return true;
|
|
500
|
+
}
|
|
501
|
+
return false;
|
|
502
|
+
}
|
|
503
|
+
|
|
504
|
+
function commandRecordSuccessfullyInvokesEntrypoint(record, entrypoint) {
|
|
505
|
+
const command = record?.command?.text;
|
|
506
|
+
const completed = record?.exitCode === 0 && Boolean(record?.finishedAt);
|
|
507
|
+
return completed && commandInvokesEntrypoint(command, entrypoint);
|
|
508
|
+
}
|
|
509
|
+
|
|
510
|
+
function shellWords(value) {
|
|
511
|
+
const words = [];
|
|
512
|
+
for (const match of value.matchAll(/"([^"]*)"|'([^']*)'|([^\s<>]+)/g)) {
|
|
513
|
+
words.push(match[1] ?? match[2] ?? match[3]);
|
|
514
|
+
}
|
|
515
|
+
return words;
|
|
516
|
+
}
|
|
517
|
+
|
|
518
|
+
function tokenMatchesEntrypoint(token, entrypoint) {
|
|
519
|
+
const normalizedToken = path.posix.normalize(stripShellQuotes(token).replaceAll("\\", "/"));
|
|
520
|
+
if (normalizedToken === entrypoint || normalizedToken === `./${entrypoint}`) return true;
|
|
521
|
+
return normalizedToken.startsWith("/") && (
|
|
522
|
+
normalizedToken.endsWith(`/repository/${entrypoint}`)
|
|
523
|
+
|| normalizedToken === `/job/output/${entrypoint}`
|
|
524
|
+
);
|
|
525
|
+
}
|
|
526
|
+
|
|
527
|
+
function stripShellQuotes(value) {
|
|
528
|
+
if ((value.startsWith('"') && value.endsWith('"')) || (value.startsWith("'") && value.endsWith("'"))) return value.slice(1, -1);
|
|
529
|
+
return value;
|
|
530
|
+
}
|
|
531
|
+
|
|
532
|
+
function normalizeRelativePath(value) {
|
|
533
|
+
if (typeof value !== "string" || !value.trim()) return null;
|
|
534
|
+
const normalized = path.posix.normalize(value.trim().replaceAll("\\", "/"));
|
|
535
|
+
if (normalized === "." || normalized.startsWith("/") || normalized === ".." || normalized.startsWith("../")) return null;
|
|
536
|
+
return normalized;
|
|
537
|
+
}
|
|
538
|
+
|
|
539
|
+
function normalizeWorkspaceChangePath(value) {
|
|
540
|
+
const normalized = normalizeRelativePath(value);
|
|
541
|
+
if (!normalized) return null;
|
|
542
|
+
for (const prefix of ["workspace/repository/", "repository/"]) {
|
|
543
|
+
if (normalized.startsWith(prefix)) return normalizeRelativePath(normalized.slice(prefix.length));
|
|
544
|
+
}
|
|
545
|
+
return normalized;
|
|
546
|
+
}
|
|
547
|
+
|
|
548
|
+
function normalizeEntrypointLogicalPath(value) {
|
|
549
|
+
if (typeof value !== "string" || !value.trim()) return null;
|
|
550
|
+
const normalized = path.posix.normalize(value.trim().replaceAll("\\", "/"));
|
|
551
|
+
for (const prefix of ["/job/workspace/repository/", "/job/output/"]) {
|
|
552
|
+
if (normalized.startsWith(prefix)) return normalizeRelativePath(normalized.slice(prefix.length));
|
|
553
|
+
}
|
|
554
|
+
return normalizeRelativePath(normalized);
|
|
555
|
+
}
|
|
556
|
+
|
|
557
|
+
function pathAllowed(changedPath, allowedPath) {
|
|
558
|
+
return changedPath === allowedPath || changedPath.startsWith(`${allowedPath}/`);
|
|
559
|
+
}
|
|
560
|
+
|
|
561
|
+
function check(output, name, passed, description, detail = {}, required = true) {
|
|
562
|
+
output.push({ name, status: passed ? "passed" : "failed", required, description, ...detail });
|
|
563
|
+
}
|
|
564
|
+
|
|
565
|
+
function normalizeAgentStatus(status) {
|
|
566
|
+
if (status === "timed_out") return "timed_out";
|
|
567
|
+
if (status === "completed") return "completed";
|
|
568
|
+
if (status === "invalid_result") return "failed";
|
|
569
|
+
return status ?? "unknown";
|
|
570
|
+
}
|
|
@@ -0,0 +1,32 @@
|
|
|
1
|
+
import { CiteArkError } from "../util.mjs";
|
|
2
|
+
|
|
3
|
+
const UNIT_DEFAULTS = new Map([
|
|
4
|
+
["percentage_points", 0.2],
|
|
5
|
+
["fraction", 0.002],
|
|
6
|
+
]);
|
|
7
|
+
|
|
8
|
+
export function resolveVerificationTolerance({ unit, tolerance }) {
|
|
9
|
+
if (tolerance !== undefined && tolerance !== null && String(tolerance).trim() !== "") {
|
|
10
|
+
const numeric = Number(tolerance);
|
|
11
|
+
if (!Number.isFinite(numeric) || numeric < 0) throw new CiteArkError("tolerance 必须是非负数");
|
|
12
|
+
return {
|
|
13
|
+
value: numeric,
|
|
14
|
+
source: { kind: "explicit", unit },
|
|
15
|
+
};
|
|
16
|
+
}
|
|
17
|
+
const unitDefault = UNIT_DEFAULTS.get(unit);
|
|
18
|
+
if (unitDefault === undefined) {
|
|
19
|
+
return {
|
|
20
|
+
value: 0,
|
|
21
|
+
source: {
|
|
22
|
+
kind: "unconfigured",
|
|
23
|
+
unit,
|
|
24
|
+
reason: `No safe default tolerance exists for unit ${unit}`,
|
|
25
|
+
},
|
|
26
|
+
};
|
|
27
|
+
}
|
|
28
|
+
return {
|
|
29
|
+
value: unitDefault,
|
|
30
|
+
source: { kind: "unit-default", unit },
|
|
31
|
+
};
|
|
32
|
+
}
|
|
@@ -0,0 +1,56 @@
|
|
|
1
|
+
import path from "node:path";
|
|
2
|
+
import { readAgentInstructions, updateRunState } from "../job.mjs";
|
|
3
|
+
import { resolveResearchPlan } from "../reproduction/plan.mjs";
|
|
4
|
+
import { readAgentUsage } from "../telemetry/usage.mjs";
|
|
5
|
+
import { CiteArkError, readJson } from "../util.mjs";
|
|
6
|
+
import { createAutonomousReproductionWorkload } from "./reproduction.mjs";
|
|
7
|
+
import { defineAgentWorkload } from "./definition.mjs";
|
|
8
|
+
|
|
9
|
+
const prompt = `Continue the same research workspace on CPU before requesting the allocated GPU. Read the fixed paper (Markdown first when available), the initial inventory, current research-plan.json, progress and any CPU handoff in output/result.json. input/compute-decision.json describes this phase's actual CPU allocation; the input task retains the scientific resource ceiling. The initial interpretation is provisional: reread sources and correct the plan as needed. Obtain the actual public inputs, resume downloads, install task-local dependencies and prepare the executable scripts. Use CPU import/loading checks where useful; a CUDA-only check belongs on GPU and need not block this handoff. Do not run the expensive scientific workload here. Do not treat a stale compiler asset checklist as exhaustive or authoritative.
|
|
10
|
+
Save the current plan and progress under /job/output and working code under /job/workspace/repository. Put downloaded model weights and datasets under /job/assets, package/download caches under /job/runtime-cache, and dependencies under /job/runtime-venv. These directories survive machine transitions. Repository changes are transferred through a bounded patch: untracked files larger than 8 MiB are excluded, so repository/artifacts is NOT a durable location for weights or datasets. If existing inputs are there, move or copy the verified bytes into /job/assets and update the loader to use that location before declaring ready. Verify the actual file and its source identity at the retained location; a manifest or successful probe against a file that will be excluded does not establish a ready handoff. Retain source-grounded reasons, asset identities, unresolved GPU-specific checks and the concrete next command. No repeated inventory extraction or detailed manifest is required. Write /job/output/cpu-preparation.json as {"schemaVersion":"1.0","status":"ready","summary":"What is prepared and what the GPU must check next"}. If no justified GPU experiment remains possible within the authorized scope and resources, use status "blocked" and explain the observed constraint. This declaration is a handoff, not proof of feasibility or scientific support. Cumulative time and monetary budgets continue across machines; never reset them.`;
|
|
11
|
+
|
|
12
|
+
/** One research actor moves machines with its complete workspace and session. */
|
|
13
|
+
export function createCpuResearchPreparationWorkload({ runtimeBudgetMinutes } = {}) {
|
|
14
|
+
const actor = createAutonomousReproductionWorkload();
|
|
15
|
+
return defineAgentWorkload({
|
|
16
|
+
...actor,
|
|
17
|
+
workload: "research-preparation",
|
|
18
|
+
runtimeBudgetMinutes,
|
|
19
|
+
validateScientificComputeDecision: false,
|
|
20
|
+
completionRelativePath: "output/cpu-preparation.json",
|
|
21
|
+
completionHasEvidence: false,
|
|
22
|
+
completionSuccessJq: '.schemaVersion == "1.0" and (.status == "ready" or .status == "blocked")',
|
|
23
|
+
preserveRebuildableCache: true,
|
|
24
|
+
finalCheckpointKind: "full",
|
|
25
|
+
finalCheckpointReason: "research-preparation",
|
|
26
|
+
recoverFinalMessage: false,
|
|
27
|
+
recover: undefined,
|
|
28
|
+
instructions: (bundle) => readAgentInstructions(bundle.runtimeTask),
|
|
29
|
+
prompt: () => prompt,
|
|
30
|
+
resumePrompt: () => `Resume existing CPU preparation. Inspect saved processes, outputs and progress before doing anything again.\n${prompt}`,
|
|
31
|
+
async validate({ bundle, completionPath }) {
|
|
32
|
+
try {
|
|
33
|
+
const result = await readJson(completionPath);
|
|
34
|
+
const issues = [];
|
|
35
|
+
if (result.schemaVersion !== "1.0" || !["ready", "blocked"].includes(result.status)
|
|
36
|
+
|| typeof result.summary !== "string" || !result.summary.trim()) issues.push("CPU preparation needs a status and concrete handoff summary");
|
|
37
|
+
await resolveResearchPlan(bundle.runtimeTask, bundle.directories.output);
|
|
38
|
+
return { result, issues, protocolVerification: null };
|
|
39
|
+
} catch (error) { return { result: null, issues: [error.message], protocolVerification: null }; }
|
|
40
|
+
},
|
|
41
|
+
isSucceeded: (result, issues) => issues.length === 0 && ["ready", "blocked"].includes(result?.status),
|
|
42
|
+
async finalize({ bundle, validation, attempts, timedOut, cloudExecution, checkpoint }) {
|
|
43
|
+
const usage = await readAgentUsage(path.join(bundle.directories.execution, "trace.jsonl"), {
|
|
44
|
+
runtimeHome: bundle.directories.runtimeHome, model: bundle.runtimeAgent.model,
|
|
45
|
+
});
|
|
46
|
+
await updateRunState(bundle, { status: validation.issues.length || timedOut ? "invalid_result" : "completed",
|
|
47
|
+
finishedAt: new Date().toISOString(), usage, attempts, cloudExecution, preparation: validation.result });
|
|
48
|
+
if (validation.issues.length || timedOut || checkpoint?.kind !== "full") {
|
|
49
|
+
throw new CiteArkError(`CPU 研究准备尚无完整交接:${validation.issues.join(";") || "缺少完整检查点或时间已耗尽"}`, {
|
|
50
|
+
failureCode: "research_preparation.handoff_incomplete", retryable: false,
|
|
51
|
+
});
|
|
52
|
+
}
|
|
53
|
+
return { bundle, checkpoint, preparation: validation.result, preparationMode: "research-workspace", cacheHit: false };
|
|
54
|
+
},
|
|
55
|
+
});
|
|
56
|
+
}
|