@citeark/agent 0.3.20
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +202 -0
- package/README.md +128 -0
- package/data/dataset-source-registry.v1.json +300 -0
- package/dist/arkgraph/boot.js +6 -0
- package/dist/arkgraph/index.html +1 -0
- package/dist/arkgraph/viewer.css +1 -0
- package/dist/arkgraph/viewer.en.css +1 -0
- package/dist/arkgraph/viewer.en.js +49 -0
- package/dist/arkgraph/viewer.en.js.LEGAL.txt +56 -0
- package/dist/arkgraph/viewer.js +49 -0
- package/dist/arkgraph/viewer.js.LEGAL.txt +56 -0
- package/docker/claude-code/Dockerfile +97 -0
- package/docker/claude-code/codex-pro-relay.mjs +466 -0
- package/docker/claude-code/runtime-contract-check.mjs +79 -0
- package/docs/arkgraph-reading.md +79 -0
- package/docs/configuration.md +100 -0
- package/docs/integration.md +92 -0
- package/docs/maturity-plan.md +27 -0
- package/docs/npm-release.md +44 -0
- package/docs/paper-reading.md +40 -0
- package/docs/research-plan-granularity.md +27 -0
- package/docs/terminal.md +49 -0
- package/examples/toy-evaluation/compile-task.json +27 -0
- package/examples/toy-evaluation/paper.md +5 -0
- package/examples/toy-evaluation/repository/README.md +9 -0
- package/examples/toy-evaluation/repository/checkpoint.json +4 -0
- package/examples/toy-evaluation/repository/evaluate.py +17 -0
- package/examples/toy-evaluation/task.json +81 -0
- package/package.json +59 -0
- package/prompts/compile-research.md +58 -0
- package/prompts/execute-contract.md +72 -0
- package/prompts/execute-workspace-simple.md +51 -0
- package/prompts/execute-workspace.md +34 -0
- package/prompts/prepare-reproduction.md +82 -0
- package/prompts/repair-research.md +45 -0
- package/protocol/CAP.md +129 -0
- package/protocol/LICENSE +12 -0
- package/protocol/MAPPINGS.md +72 -0
- package/protocol/README.md +38 -0
- package/protocol/conformance-v2.0-alpha.1.json +36 -0
- package/protocol/examples/arkgraph/checkpoint-evaluation.json +309 -0
- package/protocol/examples/arkgraph/fixtures.mjs +49 -0
- package/protocol/examples/arkgraph/paper-free.json +291 -0
- package/protocol/examples/arkgraph/partial-failure.json +344 -0
- package/protocol/examples/arkgraph/training-evaluation.json +443 -0
- package/protocol/profiles/agent-trace.md +16 -0
- package/protocol/profiles/computational-run.md +16 -0
- package/protocol/profiles/core.md +15 -0
- package/protocol/profiles/public-bundle.md +18 -0
- package/protocol/profiles/reproduction.md +29 -0
- package/protocol/profiles/research-compilation.md +44 -0
- package/protocol/profiles/research-plan.md +39 -0
- package/protocol/profiles/restricted-evidence.md +15 -0
- package/runtime/bootstrap-autodl-runtime.sh +314 -0
- package/runtime/create-runtime-venv.sh +41 -0
- package/runtime/install-local-cpu-runtime.sh +23 -0
- package/runtime/install-scientific-runtime.sh +153 -0
- package/runtime/mineru/parse.py +62 -0
- package/runtime/mineru/requirements.txt +4 -0
- package/runtime/requirements-baseline.txt +38 -0
- package/schemas/cap/v2/activity.schema.json +47 -0
- package/schemas/cap/v2/agent.schema.json +32 -0
- package/schemas/cap/v2/assertion.schema.json +110 -0
- package/schemas/cap/v2/descriptor.schema.json +243 -0
- package/schemas/cap/v2/entity.schema.json +64 -0
- package/schemas/cap/v2/manifest.schema.json +67 -0
- package/schemas/cap/v2/relation.schema.json +82 -0
- package/schemas/compute-catalog.schema.json +63 -0
- package/schemas/compute-decision.schema.json +27 -0
- package/schemas/execution-contract.schema.json +1024 -0
- package/schemas/research-card.schema.json +30 -0
- package/schemas/research-inventory-draft.schema.json +366 -0
- package/schemas/research.schema.json +1044 -0
- package/schemas/result.schema.json +173 -0
- package/schemas/verification-policy.schema.json +47 -0
- package/schemas/verified-conclusion.schema.json +58 -0
- package/schemas/workspace-summary.schema.json +24 -0
- package/scripts/build-arkgraph-view.mjs +12 -0
- package/scripts/check-execution-feasibility.mjs +24 -0
- package/scripts/check-syntax.mjs +15 -0
- package/scripts/deterministic-asset-preparation.py +438 -0
- package/scripts/package-cap.mjs +23 -0
- package/scripts/package-local-agent.mjs +23 -0
- package/scripts/preview-arkgraph.mjs +25 -0
- package/scripts/replay-research-compiler-candidate.mjs +134 -0
- package/scripts/review-compiler-sources.mjs +44 -0
- package/scripts/run-asset-preparation.sh +17 -0
- package/scripts/run-research-plan.mjs +98 -0
- package/scripts/validate-asset-preparation.py +290 -0
- package/scripts/verify-local-runtime.mjs +57 -0
- package/scripts/verify-npm-package.mjs +57 -0
- package/src/adapters/paper2agent.mjs +107 -0
- package/src/assets/cache.mjs +159 -0
- package/src/assets/compute.mjs +98 -0
- package/src/assets/executor.mjs +145 -0
- package/src/assets/lifecycle.mjs +213 -0
- package/src/assets/manifest.mjs +242 -0
- package/src/assets/opportunistic-preparation.mjs +81 -0
- package/src/assets/plan.mjs +411 -0
- package/src/assets/prompts.mjs +29 -0
- package/src/assets/public-asset-probe.mjs +525 -0
- package/src/assets/qualification.mjs +119 -0
- package/src/assets/readiness.mjs +130 -0
- package/src/assets/reproduction-admission.mjs +355 -0
- package/src/assets/requirements.mjs +152 -0
- package/src/assets/source-grounding.mjs +341 -0
- package/src/assets/source-policy.mjs +118 -0
- package/src/autodl/client.mjs +260 -0
- package/src/autodl/ssh.mjs +380 -0
- package/src/autodl/tools.mjs +129 -0
- package/src/cap/redaction.mjs +38 -0
- package/src/cap/v2/archive.mjs +152 -0
- package/src/cap/v2/attestation.mjs +204 -0
- package/src/cap/v2/canonical-json.mjs +114 -0
- package/src/cap/v2/compilation-artifact.mjs +240 -0
- package/src/cap/v2/core.mjs +282 -0
- package/src/cap/v2/measurement-assessment-records.mjs +23 -0
- package/src/cap/v2/pipeline-artifact.mjs +922 -0
- package/src/cap/v2/read.mjs +41 -0
- package/src/cap/v2/reassessment-artifact.mjs +383 -0
- package/src/cap/v2/research-artifact.mjs +231 -0
- package/src/cap/v2/research-map-records.mjs +46 -0
- package/src/cap/v2/research-object-records.mjs +163 -0
- package/src/cap/v2/research-records.mjs +187 -0
- package/src/cap/v2/verify.mjs +642 -0
- package/src/cli.mjs +1146 -0
- package/src/compute/autodl-pro-compiler.mjs +347 -0
- package/src/compute/autodl-pro-executor.mjs +459 -0
- package/src/compute/autodl-pro-job.mjs +843 -0
- package/src/compute/autodl-pro-network.mjs +295 -0
- package/src/compute/autodl-pro-remote.mjs +810 -0
- package/src/compute/autodl-pro-staging.mjs +117 -0
- package/src/compute/campaign.mjs +110 -0
- package/src/compute/catalog.mjs +123 -0
- package/src/compute/checkpoint-protocol.mjs +154 -0
- package/src/compute/codex-account-lock.mjs +111 -0
- package/src/compute/codex-account-session.mjs +107 -0
- package/src/compute/compiler-profile.mjs +38 -0
- package/src/compute/compiler-router.mjs +23 -0
- package/src/compute/coordinator-recovery.mjs +210 -0
- package/src/compute/executor-router.mjs +29 -0
- package/src/compute/gcp-batch-compiler.mjs +685 -0
- package/src/compute/gcp-batch-executor.mjs +1215 -0
- package/src/compute/gcp-batch-failure.mjs +92 -0
- package/src/compute/gcp-batch-job.mjs +527 -0
- package/src/compute/gcp-batch-lifecycle.mjs +81 -0
- package/src/compute/gcp-checkpoint-worker.mjs +1633 -0
- package/src/compute/local-codex-compiler.mjs +52 -0
- package/src/compute/measurement-hardware.mjs +128 -0
- package/src/compute/remote-attempt.mjs +226 -0
- package/src/compute/requirements.mjs +124 -0
- package/src/compute/research-phases.mjs +48 -0
- package/src/compute/scheduler.mjs +452 -0
- package/src/compute/shared-workloads.mjs +26 -0
- package/src/compute/stage-archive.mjs +79 -0
- package/src/contracts/campaign-contract.mjs +52 -0
- package/src/contracts/execution-contract.mjs +819 -0
- package/src/contracts/execution-mode.mjs +19 -0
- package/src/contracts/execution-timeouts.mjs +45 -0
- package/src/contracts/execution-workload.mjs +68 -0
- package/src/contracts/preflight-schema.mjs +25 -0
- package/src/contracts/public-contract.mjs +63 -0
- package/src/contracts/subject-tags.mjs +31 -0
- package/src/dashboard/data.mjs +898 -0
- package/src/dashboard/server.mjs +79 -0
- package/src/dashboard/static/dashboard.css +366 -0
- package/src/dashboard/static/dashboard.js +560 -0
- package/src/dashboard/static/index.html +85 -0
- package/src/deployment/community-policy.mjs +9 -0
- package/src/deployment/environment.mjs +112 -0
- package/src/deployment/guided.mjs +98 -0
- package/src/deployment/handoff.mjs +102 -0
- package/src/deployment/local-contract.mjs +31 -0
- package/src/deployment/local.mjs +100 -0
- package/src/deployment/prepare.mjs +46 -0
- package/src/deployment/recipe.mjs +108 -0
- package/src/deployment/supplement.mjs +51 -0
- package/src/deployment/terminal.mjs +43 -0
- package/src/diagnosis/renderer.mjs +75 -0
- package/src/diagnosis/target-failure.mjs +46 -0
- package/src/evidence/parser-registry.mjs +54 -0
- package/src/evidence/parsers/fasttext-classification.mjs +82 -0
- package/src/evidence/parsers/json-scalar.mjs +96 -0
- package/src/evidence/parsers/simcse-senteval.mjs +104 -0
- package/src/evidence/parsers/starspace-classification.mjs +78 -0
- package/src/evidence/registry.mjs +147 -0
- package/src/execution/runner-audit.mjs +473 -0
- package/src/gcp/auth.mjs +106 -0
- package/src/gcp/batch-client.mjs +120 -0
- package/src/gcp/resource-discovery.mjs +177 -0
- package/src/gcp/rest.mjs +82 -0
- package/src/gcp/secret-manager.mjs +34 -0
- package/src/gcp/signed-url.mjs +133 -0
- package/src/gcp/storage.mjs +220 -0
- package/src/graph/command.mjs +41 -0
- package/src/graph/execution.mjs +97 -0
- package/src/graph/model.mjs +37 -0
- package/src/graph/presentation.mjs +110 -0
- package/src/graph/query.mjs +159 -0
- package/src/graph/research-relations.mjs +69 -0
- package/src/graph/source-page.mjs +12 -0
- package/src/graph/source-preview.mjs +34 -0
- package/src/graph/validate.mjs +76 -0
- package/src/job.mjs +496 -0
- package/src/network/autodl-routing-proxy.mjs +462 -0
- package/src/network/egress-proxy.mjs +158 -0
- package/src/observability/event-contract.mjs +230 -0
- package/src/observability/pipeline-monitor.mjs +166 -0
- package/src/pipeline/orchestrator.mjs +1281 -0
- package/src/pipeline/recovery-error.mjs +11 -0
- package/src/pipeline/replay.mjs +304 -0
- package/src/pipeline/shared-execution.mjs +115 -0
- package/src/pipeline/stage-checkpoint.mjs +86 -0
- package/src/pipeline/stage-recovery.mjs +101 -0
- package/src/pipeline/targets.mjs +110 -0
- package/src/process.mjs +143 -0
- package/src/protocol.mjs +312 -0
- package/src/provider/codex-account.mjs +44 -0
- package/src/provider/codex-completion.mjs +49 -0
- package/src/provider/completion.mjs +292 -0
- package/src/provider/model-client.mjs +44 -0
- package/src/provider/model-route.mjs +29 -0
- package/src/provider/openrouter-readiness.mjs +189 -0
- package/src/provider/reader-bridge.mjs +35 -0
- package/src/provider/relay.mjs +263 -0
- package/src/provider/runtime-auth.mjs +40 -0
- package/src/public/cap.d.mts +90 -0
- package/src/public/cap.mjs +12 -0
- package/src/public/contracts.d.mts +2 -0
- package/src/public/host.mjs +171 -0
- package/src/public/operations.d.mts +11 -0
- package/src/public/presentation.d.mts +4 -0
- package/src/records/views.mjs +26 -0
- package/src/remote/command.mjs +178 -0
- package/src/remote/ssh.mjs +59 -0
- package/src/repository-origin.mjs +81 -0
- package/src/reproduction/evidence-feedback.mjs +96 -0
- package/src/reproduction/incomplete-initialization.mjs +25 -0
- package/src/reproduction/lifecycle.mjs +253 -0
- package/src/reproduction/plan.mjs +132 -0
- package/src/reproduction/prompts.mjs +70 -0
- package/src/reproduction/runner.mjs +188 -0
- package/src/reproduction/summary.mjs +130 -0
- package/src/reproduction/workspace-mode.mjs +7 -0
- package/src/research/automatic-admission.mjs +156 -0
- package/src/research/compiler-coverage.mjs +85 -0
- package/src/research/compiler-failure.mjs +24 -0
- package/src/research/compiler-normalization-guards.mjs +112 -0
- package/src/research/compiler-repair.mjs +3 -0
- package/src/research/compiler.mjs +853 -0
- package/src/research/continuation-selection.mjs +26 -0
- package/src/research/execution-graph-context.mjs +43 -0
- package/src/research/experiment-importance.mjs +15 -0
- package/src/research/inventory-handoff.mjs +104 -0
- package/src/research/inventory-revisions.mjs +32 -0
- package/src/research/mineru-local.mjs +73 -0
- package/src/research/paper-command.mjs +19 -0
- package/src/research/paper-markdown.mjs +180 -0
- package/src/research/paper-source-map.mjs +69 -0
- package/src/research/planning-policy.mjs +88 -0
- package/src/research/reference-materials.mjs +11 -0
- package/src/research/reproduction-scope.mjs +30 -0
- package/src/research/research-map.mjs +94 -0
- package/src/research/research-objects.mjs +88 -0
- package/src/research/source-discovery.mjs +646 -0
- package/src/research/source-observations.mjs +75 -0
- package/src/research/source-review-cli-mcp.mjs +26 -0
- package/src/research/source-review-input.mjs +209 -0
- package/src/research/source-review-local-codex.mjs +36 -0
- package/src/research/source-review-model.mjs +70 -0
- package/src/research/source-review.mjs +173 -0
- package/src/research/structure.mjs +3163 -0
- package/src/research-card/renderer.mjs +277 -0
- package/src/research-card/verified-conclusion.mjs +143 -0
- package/src/results/output-registry.mjs +183 -0
- package/src/runtime/claude-code.mjs +52 -0
- package/src/runtime/codex-capacity-retry.mjs +87 -0
- package/src/runtime/codex.mjs +64 -0
- package/src/runtime/config.mjs +157 -0
- package/src/runtime/final-output.mjs +40 -0
- package/src/runtime/index.mjs +21 -0
- package/src/runtime/local-codex.mjs +74 -0
- package/src/runtime/opencode.mjs +95 -0
- package/src/runtime/prompt.mjs +13 -0
- package/src/sandbox/docker.mjs +363 -0
- package/src/settings/command.mjs +297 -0
- package/src/settings/store.mjs +119 -0
- package/src/telemetry/pricing.mjs +68 -0
- package/src/telemetry/usage.mjs +265 -0
- package/src/terminal/events.mjs +97 -0
- package/src/terminal/input.mjs +40 -0
- package/src/terminal/plain.mjs +40 -0
- package/src/terminal/remote-stream.mjs +22 -0
- package/src/terminal/screen.mjs +214 -0
- package/src/terminal/transcript.mjs +69 -0
- package/src/util.mjs +107 -0
- package/src/verification/ai-assessor.mjs +534 -0
- package/src/verification/claim-evaluator.mjs +242 -0
- package/src/verification/evidence-context.mjs +165 -0
- package/src/verification/evidence-reader.mjs +95 -0
- package/src/verification/integrity.mjs +570 -0
- package/src/verification/tolerance.mjs +32 -0
- package/src/workloads/cpu-research-preparation.mjs +56 -0
- package/src/workloads/definition.mjs +74 -0
- package/src/workloads/phase-aware-reproduction.mjs +46 -0
- package/src/workloads/reproduction.mjs +85 -0
- package/src/workspace/command.mjs +242 -0
- package/src/workspace/control.mjs +49 -0
- package/src/workspace/entry.mjs +28 -0
- package/src/workspace/input.mjs +93 -0
- package/src/workspace/interactive.mjs +94 -0
- package/src/workspace/jobs.mjs +418 -0
- package/src/workspace/session.mjs +97 -0
- package/src/workspace/worker.mjs +137 -0
- package/ui/arkgraph/ambient-motion.mjs +10 -0
- package/ui/arkgraph/app.jsx +153 -0
- package/ui/arkgraph/boot.js +6 -0
- package/ui/arkgraph/camera-motion.mjs +20 -0
- package/ui/arkgraph/context-reveal.mjs +39 -0
- package/ui/arkgraph/details.css +3 -0
- package/ui/arkgraph/entry.jsx +28 -0
- package/ui/arkgraph/experiment-curves.mjs +17 -0
- package/ui/arkgraph/experiment-selection.mjs +15 -0
- package/ui/arkgraph/experiment-style.css +26 -0
- package/ui/arkgraph/experiment-ui.jsx +32 -0
- package/ui/arkgraph/frame.html +1 -0
- package/ui/arkgraph/graph-gestures.mjs +62 -0
- package/ui/arkgraph/label-layout.mjs +57 -0
- package/ui/arkgraph/locales/en.json +229 -0
- package/ui/arkgraph/locales/source-types.json +15 -0
- package/ui/arkgraph/localization-build.mjs +27 -0
- package/ui/arkgraph/material-build.mjs +23 -0
- package/ui/arkgraph/material-colors.mjs +39 -0
- package/ui/arkgraph/material-style.css +15 -0
- package/ui/arkgraph/open-graph.jsx +326 -0
- package/ui/arkgraph/outline.jsx +49 -0
- package/ui/arkgraph/package-lock.json +888 -0
- package/ui/arkgraph/package.json +17 -0
- package/ui/arkgraph/reading-layout.mjs +130 -0
- package/ui/arkgraph/reading-presentation.mjs +73 -0
- package/ui/arkgraph/record-detail.css +51 -0
- package/ui/arkgraph/record-details.jsx +29 -0
- package/ui/arkgraph/research-types.mjs +31 -0
- package/ui/arkgraph/selection-mark.jsx +6 -0
- package/ui/arkgraph/soft-spine.mjs +26 -0
- package/ui/arkgraph/steering-style.css +187 -0
- package/ui/arkgraph/style.css +272 -0
|
@@ -0,0 +1,242 @@
|
|
|
1
|
+
import {
|
|
2
|
+
verificationPolicyTargets,
|
|
3
|
+
} from "../contracts/execution-contract.mjs";
|
|
4
|
+
import { sha256Value, writeJson } from "../util.mjs";
|
|
5
|
+
import { SCIENTIFIC_ASSESSMENT_RUBRIC_VERSION } from "./ai-assessor.mjs";
|
|
6
|
+
|
|
7
|
+
export async function evaluateClaim({
|
|
8
|
+
contract,
|
|
9
|
+
policy,
|
|
10
|
+
metrics,
|
|
11
|
+
outputs,
|
|
12
|
+
integrity,
|
|
13
|
+
result,
|
|
14
|
+
claim,
|
|
15
|
+
scientificAssessor,
|
|
16
|
+
runnerAudit,
|
|
17
|
+
evidenceContext,
|
|
18
|
+
assessmentPath,
|
|
19
|
+
assessedAt,
|
|
20
|
+
}) {
|
|
21
|
+
const executionStatus = result?.execution?.status ?? integrity.states?.experimentStatus ?? "unknown";
|
|
22
|
+
const reconstructionFidelity = contract.reconstructionFidelity ?? "faithful";
|
|
23
|
+
const strictlyComparable = new Set(["exact", "faithful"]).has(reconstructionFidelity);
|
|
24
|
+
const parsedMeasurements = Array.isArray(metrics.measurements) ? metrics.measurements : [];
|
|
25
|
+
const deterministicAssessments = verificationPolicyTargets(policy).map((target) => {
|
|
26
|
+
const parsed = parsedMeasurements.find((item) => item.measurementId === target.measurementId);
|
|
27
|
+
const assessment = evaluateMeasurement({
|
|
28
|
+
target,
|
|
29
|
+
metrics: parsed,
|
|
30
|
+
integrity,
|
|
31
|
+
result,
|
|
32
|
+
executionStatus,
|
|
33
|
+
});
|
|
34
|
+
if (strictlyComparable || assessment.verificationStatus === "inconclusive") return assessment;
|
|
35
|
+
return {
|
|
36
|
+
...assessment,
|
|
37
|
+
verdict: "inconclusive",
|
|
38
|
+
verificationStatus: "inconclusive",
|
|
39
|
+
reason: `The observed value is recorded as directional evidence, but ${reconstructionFidelity} reconstruction fidelity cannot establish a strict Match against the paper`,
|
|
40
|
+
comparison: {
|
|
41
|
+
...assessment.comparison,
|
|
42
|
+
directionalStatus: assessment.verificationStatus,
|
|
43
|
+
},
|
|
44
|
+
};
|
|
45
|
+
});
|
|
46
|
+
let measurementAssessments = deterministicAssessments;
|
|
47
|
+
let verificationStatus = measurementAssessments.length > 0
|
|
48
|
+
&& measurementAssessments.every((item) => item.verificationStatus === "reproduced")
|
|
49
|
+
? "reproduced"
|
|
50
|
+
: measurementAssessments.some((item) => item.verificationStatus === "not_reproduced")
|
|
51
|
+
? "not_reproduced"
|
|
52
|
+
: "inconclusive";
|
|
53
|
+
let verdict = verificationStatus === "reproduced"
|
|
54
|
+
? "supports"
|
|
55
|
+
: verificationStatus === "not_reproduced"
|
|
56
|
+
? "contradicts"
|
|
57
|
+
: "inconclusive";
|
|
58
|
+
let reason = verificationStatus === "reproduced"
|
|
59
|
+
? "Every protocol-bound measurement is within its pre-declared tolerance"
|
|
60
|
+
: verificationStatus === "not_reproduced"
|
|
61
|
+
? "At least one protocol-bound measurement is outside its pre-declared tolerance"
|
|
62
|
+
: measurementAssessments
|
|
63
|
+
.filter((item) => item.verificationStatus === "inconclusive")
|
|
64
|
+
.map((item) => item.reason)
|
|
65
|
+
.filter((value, index, values) => values.indexOf(value) === index)
|
|
66
|
+
.join("; ") || "At least one protocol-bound measurement lacks sufficient trustworthy evidence";
|
|
67
|
+
let assessor = { kind: "service", name: "citeark-verification-engine", version: "0.1.0" };
|
|
68
|
+
let confidence = null;
|
|
69
|
+
let protocolComparability = strictlyComparable ? "exact" : "material_difference";
|
|
70
|
+
let assessmentLimitations = [];
|
|
71
|
+
let assessorFailure = null;
|
|
72
|
+
let assessmentUsage = null;
|
|
73
|
+
let executionEvidence = null;
|
|
74
|
+
let evidenceReads = [];
|
|
75
|
+
|
|
76
|
+
const completeTrustworthyEvidence = integrity.status === "passed"
|
|
77
|
+
&& measurementAssessments.length > 0
|
|
78
|
+
&& (contract.executionPlan
|
|
79
|
+
? measurementAssessments.some((item) => item.comparison?.observedValue != null)
|
|
80
|
+
: measurementAssessments.every((item) => item.comparison?.observedValue !== null))
|
|
81
|
+
&& new Set(contract.executionPlan ? ["succeeded", "completed", "partial"] : ["succeeded", "completed"]).has(executionStatus);
|
|
82
|
+
if (typeof scientificAssessor === "function" && completeTrustworthyEvidence) {
|
|
83
|
+
const modelResult = await scientificAssessor({
|
|
84
|
+
claim,
|
|
85
|
+
contract,
|
|
86
|
+
comparisons: contract.executionPlan
|
|
87
|
+
? measurementAssessments.filter((item) => item.comparison?.observedValue != null)
|
|
88
|
+
: measurementAssessments,
|
|
89
|
+
result,
|
|
90
|
+
integrity,
|
|
91
|
+
runnerAudit,
|
|
92
|
+
evidenceContext,
|
|
93
|
+
});
|
|
94
|
+
if (modelResult?.status === "completed") {
|
|
95
|
+
const byMeasurementId = new Map(modelResult.assessment.measurementAssessments.map((item) => [item.measurementId, item]));
|
|
96
|
+
measurementAssessments = measurementAssessments.map((item) => ({
|
|
97
|
+
...(byMeasurementId.get(item.measurementId) ?? item),
|
|
98
|
+
comparison: item.comparison,
|
|
99
|
+
}));
|
|
100
|
+
verificationStatus = modelResult.assessment.verificationStatus;
|
|
101
|
+
verdict = modelResult.assessment.conclusion;
|
|
102
|
+
reason = modelResult.assessment.reason;
|
|
103
|
+
confidence = modelResult.assessment.confidence;
|
|
104
|
+
protocolComparability = modelResult.assessment.protocolComparability;
|
|
105
|
+
assessmentLimitations = modelResult.assessment.limitations;
|
|
106
|
+
assessor = modelResult.assessor;
|
|
107
|
+
assessmentUsage = modelResult.usage;
|
|
108
|
+
evidenceReads = modelResult.evidenceReads ?? [];
|
|
109
|
+
executionEvidence = modelResult.assessment.executionEvidence ?? null;
|
|
110
|
+
} else {
|
|
111
|
+
verificationStatus = "inconclusive";
|
|
112
|
+
verdict = "inconclusive";
|
|
113
|
+
reason = "The independent AI scientific assessor is unavailable. Numeric evidence is preserved, and CiteArk does not fall back to a fixed-threshold verdict.";
|
|
114
|
+
measurementAssessments = measurementAssessments.map((item) => ({
|
|
115
|
+
...item,
|
|
116
|
+
verdict: "inconclusive",
|
|
117
|
+
verificationStatus: "inconclusive",
|
|
118
|
+
reason,
|
|
119
|
+
}));
|
|
120
|
+
assessorFailure = modelResult?.error ?? { message: "unknown assessor failure" };
|
|
121
|
+
assessor = { kind: "model", name: "unavailable", rubricVersion: SCIENTIFIC_ASSESSMENT_RUBRIC_VERSION };
|
|
122
|
+
assessmentLimitations = [reason];
|
|
123
|
+
protocolComparability = "unknown";
|
|
124
|
+
}
|
|
125
|
+
}
|
|
126
|
+
const advisoryProvenance = integrity.checks?.some((item) =>
|
|
127
|
+
item.evidentiaryRole === "advisory-only; assessor must trace actual computation");
|
|
128
|
+
if (((contract.executionPlan || advisoryProvenance) && typeof scientificAssessor !== "function")
|
|
129
|
+
|| (contract.executionPlan && measurementAssessments.some((item) => item.comparison?.observedValue == null))) {
|
|
130
|
+
verificationStatus = "inconclusive";
|
|
131
|
+
verdict = "inconclusive";
|
|
132
|
+
reason = typeof scientificAssessor !== "function"
|
|
133
|
+
? "Scientific execution provenance requires independent assessment; numeric proximity and advisory command matches are insufficient."
|
|
134
|
+
: [reason, "Some original claim measurements remain without evidence; available measurements are assessed individually."].filter(Boolean).join(" ");
|
|
135
|
+
if (typeof scientificAssessor !== "function") {
|
|
136
|
+
measurementAssessments = measurementAssessments.map((item) => ({ ...item,
|
|
137
|
+
verdict: "inconclusive", verificationStatus: "inconclusive", reason }));
|
|
138
|
+
}
|
|
139
|
+
}
|
|
140
|
+
const body = {
|
|
141
|
+
schemaVersion: "1.0",
|
|
142
|
+
kind: "citeark.claim-assessment",
|
|
143
|
+
contractDigest: contract.contractDigest,
|
|
144
|
+
...(contract.executionPlan ? { executionPlanDigest: contract.executionPlan.digest } : {}),
|
|
145
|
+
policyDigest: policy.policyDigest,
|
|
146
|
+
claimVersionId: contract.research.claimVersionId,
|
|
147
|
+
experimentVersionId: contract.research.experimentVersionId,
|
|
148
|
+
reproductionLevel: contract.reproductionLevel,
|
|
149
|
+
reconstructionFidelity,
|
|
150
|
+
verdict,
|
|
151
|
+
verificationStatus,
|
|
152
|
+
reason,
|
|
153
|
+
...(confidence ? { confidence } : {}),
|
|
154
|
+
protocolComparability,
|
|
155
|
+
...(executionEvidence ? { executionEvidence } : {}),
|
|
156
|
+
limitations: assessmentLimitations,
|
|
157
|
+
evidenceReads,
|
|
158
|
+
...(assessorFailure ? { assessorFailure } : {}),
|
|
159
|
+
...(assessmentUsage ? { usage: assessmentUsage } : {}),
|
|
160
|
+
measurementAssessments,
|
|
161
|
+
outputAssessments: (outputs?.outputs ?? []).map((output, index) => {
|
|
162
|
+
const label = `${index + 1}:${output.id}`;
|
|
163
|
+
const captured = [
|
|
164
|
+
`output_present:${label}`,
|
|
165
|
+
`output_digest:${label}`,
|
|
166
|
+
].every((name) => integrity.checks?.some((check) => check.name === name && check.status === "passed"));
|
|
167
|
+
return {
|
|
168
|
+
outputId: output.id,
|
|
169
|
+
kind: output.kind,
|
|
170
|
+
role: output.role,
|
|
171
|
+
status: captured ? "captured" : "invalid",
|
|
172
|
+
reason: captured
|
|
173
|
+
? "The output is preserved with a runner-verified content identity"
|
|
174
|
+
: "The output file or its declared content identity did not pass integrity verification",
|
|
175
|
+
digest: output.digest,
|
|
176
|
+
byteSize: output.byteSize,
|
|
177
|
+
mediaType: output.mediaType,
|
|
178
|
+
};
|
|
179
|
+
}),
|
|
180
|
+
integrityStatus: integrity.status,
|
|
181
|
+
assessedAt: assessedAt ?? integrity.checkedAt ?? new Date().toISOString(),
|
|
182
|
+
assessor,
|
|
183
|
+
};
|
|
184
|
+
const assessment = { ...body, assessmentDigest: `sha256:${sha256Value(body)}` };
|
|
185
|
+
if (assessmentPath) await writeJson(assessmentPath, assessment);
|
|
186
|
+
return assessment;
|
|
187
|
+
}
|
|
188
|
+
|
|
189
|
+
function evaluateMeasurement({ target, metrics, integrity, result, executionStatus }) {
|
|
190
|
+
let verificationStatus = "inconclusive";
|
|
191
|
+
let verdict = "inconclusive";
|
|
192
|
+
let reason = "Execution integrity checks did not pass, so this measurement cannot be evaluated";
|
|
193
|
+
let absoluteDifference = null;
|
|
194
|
+
const observedValue = typeof metrics?.primary?.value === "number" && Number.isFinite(metrics.primary.value)
|
|
195
|
+
? metrics.primary.value
|
|
196
|
+
: null;
|
|
197
|
+
if (integrity.status === "passed" && observedValue === null) {
|
|
198
|
+
const evidenceReason = metrics?.availability?.reason ?? "The execution produced no independently parseable metric";
|
|
199
|
+
reason = result?.execution?.failure
|
|
200
|
+
? `${failureReason(result.execution.failure, executionStatus, evidenceReason)} ${evidenceReason}`
|
|
201
|
+
: evidenceReason;
|
|
202
|
+
} else if (integrity.status === "passed") {
|
|
203
|
+
absoluteDifference = Number(Math.abs(observedValue - target.reportedValue).toPrecision(12));
|
|
204
|
+
if (target.toleranceSource?.kind === "unconfigured") {
|
|
205
|
+
reason = "The independently parsed observation is preserved, but no metric-specific tolerance was pre-declared; the measurement remains inconclusive";
|
|
206
|
+
} else {
|
|
207
|
+
const reproduced = absoluteDifference <= target.tolerance;
|
|
208
|
+
verificationStatus = reproduced ? "reproduced" : "not_reproduced";
|
|
209
|
+
verdict = reproduced ? "supports" : "contradicts";
|
|
210
|
+
reason = reproduced
|
|
211
|
+
? "The independently parsed observed result is within the pre-declared tolerance"
|
|
212
|
+
: "The independently parsed observed result is outside the pre-declared tolerance";
|
|
213
|
+
}
|
|
214
|
+
}
|
|
215
|
+
return {
|
|
216
|
+
measurementId: target.measurementId,
|
|
217
|
+
verdict,
|
|
218
|
+
verificationStatus,
|
|
219
|
+
reason,
|
|
220
|
+
comparison: {
|
|
221
|
+
metric: target.metric,
|
|
222
|
+
unit: target.unit,
|
|
223
|
+
reportedValue: target.reportedValue,
|
|
224
|
+
observedValue,
|
|
225
|
+
tolerance: target.tolerance,
|
|
226
|
+
...(target.toleranceSource ? { toleranceSource: target.toleranceSource } : {}),
|
|
227
|
+
...(target.evidenceBasis ? { evidenceBasis: target.evidenceBasis } : {}),
|
|
228
|
+
...(target.extractionUncertainty !== undefined
|
|
229
|
+
? { extractionUncertainty: target.extractionUncertainty }
|
|
230
|
+
: {}),
|
|
231
|
+
absoluteDifference,
|
|
232
|
+
},
|
|
233
|
+
};
|
|
234
|
+
}
|
|
235
|
+
|
|
236
|
+
function failureReason(failure, executionStatus, fallback) {
|
|
237
|
+
if (!failure || typeof failure !== "object") {
|
|
238
|
+
return fallback ?? `The execution ended with status ${executionStatus} before a trustworthy comparison metric was available`;
|
|
239
|
+
}
|
|
240
|
+
const retryability = failure.retryable ? "potentially retryable" : "not retryable without changing the available inputs or conditions";
|
|
241
|
+
return `The execution was ${executionStatus} during ${failure.stage ?? "an unknown stage"} because of ${failure.category ?? "an unclassified failure"}: ${failure.reason ?? fallback ?? "no further detail was recorded"}. This outcome is ${retryability}`;
|
|
242
|
+
}
|
|
@@ -0,0 +1,165 @@
|
|
|
1
|
+
import path from "node:path";
|
|
2
|
+
import { execFile } from "node:child_process";
|
|
3
|
+
import { promisify } from "node:util";
|
|
4
|
+
import { createHash } from "node:crypto";
|
|
5
|
+
import { lstat, readFile, realpath, readdir } from "node:fs/promises";
|
|
6
|
+
import { redactString } from "../cap/redaction.mjs";
|
|
7
|
+
import { safeRelativePath, sha256Value } from "../util.mjs";
|
|
8
|
+
|
|
9
|
+
const execute = promisify(execFile);
|
|
10
|
+
const MAX_FILE_BYTES = 64 * 1024 * 1024; // Memory safety per retained text file; never a prompt limit.
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
/** Host-selected, bounded source material for independent assessment. Selection
|
|
14
|
+
* and hashes establish which bytes were read, never scientific correctness.
|
|
15
|
+
* No arbitrary file reads or commands supplied by the execution Agent are run.
|
|
16
|
+
*/
|
|
17
|
+
export async function buildAssessmentEvidenceContext({ runDirectory, contract, claim, runnerAudit, paperDigest }) {
|
|
18
|
+
const root = await realpath(runDirectory);
|
|
19
|
+
const files = [];
|
|
20
|
+
const omissions = [];
|
|
21
|
+
|
|
22
|
+
const task = await readJson("input/task.json");
|
|
23
|
+
const after = await readJson("execution/workspace-after.json");
|
|
24
|
+
const entries = new Map((after?.entries ?? []).map((entry) => [entry.path, entry]));
|
|
25
|
+
|
|
26
|
+
const addText = async (relative, role, expectedDigest) => {
|
|
27
|
+
if (files.some((item) => item.path === relative)) return;
|
|
28
|
+
|
|
29
|
+
try {
|
|
30
|
+
const { bytes } = await readSafe(relative, MAX_FILE_BYTES);
|
|
31
|
+
if (bytes.includes(0)) throw new Error("binary_file");
|
|
32
|
+
const digest = sha256Bytes(bytes);
|
|
33
|
+
if (expectedDigest && normalizeDigest(expectedDigest) !== digest) throw new Error("workspace_snapshot_digest_mismatch");
|
|
34
|
+
const text = redactString(bytes.toString("utf8"));
|
|
35
|
+
const selected = { text, truncated: false, originalCharacters: text.length };
|
|
36
|
+
files.push({ role, path: relative, digest, byteSize: bytes.length,
|
|
37
|
+
provenance: expectedDigest ? "matches_runner_workspace_snapshot" : "retained_run_file",
|
|
38
|
+
redacted: text !== bytes.toString("utf8"), ...selected });
|
|
39
|
+
|
|
40
|
+
return true;
|
|
41
|
+
} catch (error) { omissions.push({ path: relative, reason: error.code === "ENOENT" ? "not_retained" : error.message }); return false; }
|
|
42
|
+
};
|
|
43
|
+
|
|
44
|
+
// Paper excerpts are selected by the actual source locators, not by result
|
|
45
|
+
// proximity. Missing or ambiguous locators remain visible to the assessor.
|
|
46
|
+
const paper = task?.paper?.path;
|
|
47
|
+
if (typeof paper === "string" && paper.startsWith("/job/input/paper/") && path.basename(paper) !== ".") {
|
|
48
|
+
const relative = `input/paper/${path.basename(paper)}`;
|
|
49
|
+
try {
|
|
50
|
+
const { bytes, filename } = await readSafe(relative, 64 * 1024 * 1024);
|
|
51
|
+
const digest = sha256Bytes(bytes);
|
|
52
|
+
if (paperDigest && normalizeDigest(paperDigest) !== digest) throw new Error("paper_snapshot_digest_mismatch");
|
|
53
|
+
const text = path.extname(filename).toLowerCase() === ".pdf"
|
|
54
|
+
? (await execute("pdftotext", ["-layout", filename, "-"], { timeout: 20_000, maxBuffer: MAX_FILE_BYTES })).stdout
|
|
55
|
+
: bytes.toString("utf8");
|
|
56
|
+
const locators = sourceLocators(contract, claim);
|
|
57
|
+
const excerpts = selectSourceExcerpts(text, locators, 18_000);
|
|
58
|
+
files.push({ role: "paper_source", path: relative, digest, byteSize: bytes.length,
|
|
59
|
+
provenance: paperDigest ? "matches_fixed_paper_digest" : "staged_paper_snapshot",
|
|
60
|
+
requestedLocators: locators, excerpts, text: redactString(text), truncated: false, completeDocument: true });
|
|
61
|
+
|
|
62
|
+
if (!excerpts.length) omissions.push({ path: relative, reason: "no_source_locator_resolved" });
|
|
63
|
+
} catch (error) { omissions.push({ path: relative, reason: error.code === "ENOENT" ? "not_retained" : error.message }); }
|
|
64
|
+
} else omissions.push({ role: "paper_source", reason: "staged_paper_unavailable" });
|
|
65
|
+
|
|
66
|
+
const rawPaths = [...new Set((contract.measurements ?? []).map((item) => item.parser?.evidencePath).filter(safeRelativePath))];
|
|
67
|
+
for (const relative of rawPaths) await addText(`output/${relative}`, "parser_input", null);
|
|
68
|
+
// Keep the raw histories available for paged independent reads, not only the aggregate.
|
|
69
|
+
async function visitOutput(directory) {
|
|
70
|
+
for (const entry of await readdir(path.join(root, directory), { withFileTypes: true }).catch(() => [])) {
|
|
71
|
+
const relative = `${directory}/${entry.name}`;
|
|
72
|
+
// Delivery is derived after scientific assessment. Feeding it back into
|
|
73
|
+
// evidence would invalidate the same run's checkpoint on every replay.
|
|
74
|
+
if (['output/handoff-review.json', 'output/reproduction-code.json', 'output/REPRODUCE.md'].includes(relative)) continue;
|
|
75
|
+
if (entry.isDirectory()) await visitOutput(relative);
|
|
76
|
+
else if (entry.isFile() && /\.(json|jsonl|ndjson|csv|tsv|txt|log)$/.test(relative)
|
|
77
|
+
&& !/(?:credentials|secrets?|auth\.json|\.env)/i.test(relative)) await addText(relative, "raw_evidence");
|
|
78
|
+
}
|
|
79
|
+
}
|
|
80
|
+
await visitOutput("output");
|
|
81
|
+
|
|
82
|
+
// Read helpers/configuration actually named in captured commands, followed
|
|
83
|
+
// by changed source files. Only bytes matching the captured final snapshot
|
|
84
|
+
// are included; a patch is provided separately to disclose modifications.
|
|
85
|
+
const commandPaths = (runnerAudit?.commandRecords ?? []).flatMap((record) =>
|
|
86
|
+
[...(record.command?.text ?? "").matchAll(/\/job\/workspace\/repository\/([A-Za-z0-9_./-]+)/g)].map((match) => match[1]));
|
|
87
|
+
const changes = await readJson("execution/workspace-changes.json");
|
|
88
|
+
const changedPaths = [...(changes?.added ?? []), ...(changes?.modified ?? [])].map((item) => item.path);
|
|
89
|
+
const implementationPaths = (contract.protocol?.benchmark?.implementations ?? []).map((item) => item.provenance);
|
|
90
|
+
const candidates = [...new Set([...commandPaths, ...implementationPaths, ...changedPaths, ...entries.keys()])].filter((relative) =>
|
|
91
|
+
safeRelativePath(relative) && /\.(?:py|sh|mjs|js|ts|cu|c|cpp|h|json|toml|ya?ml|ipynb|r|R|rs|jl|java|cfg|ini|md|txt)$/.test(relative)
|
|
92
|
+
&& !/(?:^|\/)(?:\.env[^/]*|[^/]*(?:credentials|signing-key|private-key)[^/]*|secrets?[^/]*|auth\.json|id_rsa)$/.test(relative));
|
|
93
|
+
for (const relative of candidates) {
|
|
94
|
+
const entry = entries.get(relative);
|
|
95
|
+
if (!entry?.sha256 || entry.type !== "file") { omissions.push({ path: relative, reason: "no_runner_snapshot_binding" }); continue; }
|
|
96
|
+
const localPath = `workspace/repository/${relative}`;
|
|
97
|
+
const included = await addText(localPath, "implementation", entry.sha256);
|
|
98
|
+
// Remote publication retains added files in the checkpoint, while the
|
|
99
|
+
// coordinator checkout can still contain the original source snapshot.
|
|
100
|
+
if (!included && await addText(`execution/workspace-checkpoint/untracked/${relative}`, "implementation", entry.sha256)) {
|
|
101
|
+
const index = omissions.findIndex((item) => item.path === localPath && item.reason === "not_retained");
|
|
102
|
+
if (index >= 0) omissions.splice(index, 1);
|
|
103
|
+
}
|
|
104
|
+
}
|
|
105
|
+
|
|
106
|
+
if (changedPaths.length) await addText("execution/repository.patch", "implementation_changes", null);
|
|
107
|
+
const body = { schemaVersion: "1.1", kind: "citeark.assessment-evidence-context",
|
|
108
|
+
trust: "Untrusted scientific content selected by the host; hashes bind bytes, not truth. Missing and truncated material cannot be assumed verified.",
|
|
109
|
+
files, omissions };
|
|
110
|
+
return { ...body, digest: `sha256:${sha256Value(body)}` };
|
|
111
|
+
|
|
112
|
+
async function readSafe(relative, maximumBytes) {
|
|
113
|
+
if (!safeRelativePath(relative)) throw new Error("unsafe_path");
|
|
114
|
+
const filename = path.join(root, relative);
|
|
115
|
+
const stat = await lstat(filename);
|
|
116
|
+
if (!stat.isFile() || !(await realpath(filename)).startsWith(`${root}${path.sep}`)) throw new Error("not_a_regular_run_file");
|
|
117
|
+
if (stat.size > maximumBytes) throw new Error("file_size_budget");
|
|
118
|
+
return { bytes: await readFile(filename), filename };
|
|
119
|
+
}
|
|
120
|
+
async function readJson(relative) {
|
|
121
|
+
try { return JSON.parse((await readSafe(relative, MAX_FILE_BYTES)).bytes.toString("utf8")); }
|
|
122
|
+
catch { return null; }
|
|
123
|
+
}
|
|
124
|
+
}
|
|
125
|
+
|
|
126
|
+
function normalizeDigest(value) { return value.startsWith("sha256:") ? value : `sha256:${value}`; }
|
|
127
|
+
function sha256Bytes(bytes) { return `sha256:${createHash("sha256").update(bytes).digest("hex")}`; }
|
|
128
|
+
|
|
129
|
+
function sourceLocators(contract, claim) {
|
|
130
|
+
const sources = [claim?.sourceLocator, ...(claim?.reportedMeasurements ?? []).map((item) => item.sourceLocator),
|
|
131
|
+
...(contract.executionPlan?.plan?.sources ?? [])];
|
|
132
|
+
return [...new Set(sources.map((source) => typeof source === "string" ? source : source?.locator).filter((value) => typeof value === "string" && value.trim()))];
|
|
133
|
+
}
|
|
134
|
+
|
|
135
|
+
export function selectSourceExcerpts(text, locators, maximumChars = 18_000) {
|
|
136
|
+
const pages = text.split("\f");
|
|
137
|
+
const selected = new Map();
|
|
138
|
+
for (const locator of locators) {
|
|
139
|
+
for (const match of locator.matchAll(/\b(?:page|p\.)\s*(\d+)\b/gi)) {
|
|
140
|
+
const index = Number(match[1]) - 1;
|
|
141
|
+
if (pages[index] !== undefined && !selected.has(index)) selected.set(index, { locator, offset: 0 });
|
|
142
|
+
}
|
|
143
|
+
const terms = [...locator.matchAll(/\b(?:table|figure|section|appendix)\s+[A-Z0-9]+(?:\.\d+)*/gi)].map((match) => match[0]);
|
|
144
|
+
for (const term of terms) {
|
|
145
|
+
const escaped = term.replace(/[.*+?^${}()|[\]\\]/g, "\\$&").replace(/\s+/g, "\\s+");
|
|
146
|
+
const pattern = new RegExp(`${escaped}\\b`, "i");
|
|
147
|
+
for (let index = 0; index < pages.length; index++) {
|
|
148
|
+
const match = pattern.exec(pages[index]);
|
|
149
|
+
if (match && !selected.has(index)) selected.set(index, { locator, offset: Math.max(0, match.index - 600) });
|
|
150
|
+
}
|
|
151
|
+
}
|
|
152
|
+
}
|
|
153
|
+
const excerpts = [];
|
|
154
|
+
let remaining = maximumChars;
|
|
155
|
+
for (const [index, selection] of selected) {
|
|
156
|
+
if (remaining <= 0 || excerpts.length >= 6) break;
|
|
157
|
+
const content = redactString(pages[index]);
|
|
158
|
+
const limit = Math.min(5_000, remaining);
|
|
159
|
+
const end = Math.min(content.length, selection.offset + limit);
|
|
160
|
+
excerpts.push({ page: index + 1, locator: selection.locator, startCharacter: selection.offset,
|
|
161
|
+
text: content.slice(selection.offset, end), truncated: selection.offset > 0 || end < content.length });
|
|
162
|
+
remaining -= end - selection.offset;
|
|
163
|
+
}
|
|
164
|
+
return excerpts;
|
|
165
|
+
}
|
|
@@ -0,0 +1,95 @@
|
|
|
1
|
+
import { sha256Value } from "../util.mjs";
|
|
2
|
+
|
|
3
|
+
// These are per-response page sizes. Every retained character remains addressable.
|
|
4
|
+
const PAGE = 12_000;
|
|
5
|
+
const tool = (name, description, properties, required) => ({ type: "function", function: {
|
|
6
|
+
name, description, parameters: { type: "object", properties, required, additionalProperties: false },
|
|
7
|
+
} });
|
|
8
|
+
export function createAssessmentEvidenceReader({ evidenceContext, runnerAudit }) {
|
|
9
|
+
const documents = new Map();
|
|
10
|
+
for (const file of evidenceContext?.files ?? []) {
|
|
11
|
+
const text = file.text ?? file.excerpts?.map(e => `[page ${e.page}]\n${e.text}`).join("\n");
|
|
12
|
+
if (typeof text === "string") documents.set(file.path, { ...file, text });
|
|
13
|
+
}
|
|
14
|
+
for (const record of runnerAudit?.commandRecords ?? []) {
|
|
15
|
+
for (const field of ["command", "stdout", "stderr"]) {
|
|
16
|
+
if (typeof record[field]?.text === "string") documents.set(`audit/${record.recordId}/${field}`, {
|
|
17
|
+
path: `audit/${record.recordId}/${field}`, role: "execution_audit", recordId: record.recordId,
|
|
18
|
+
text: record[field].text, digest: `sha256:${sha256Value(record[field].text)}`,
|
|
19
|
+
status: record.status, exitCode: record.exitCode, truncated: Boolean(record[field].truncated),
|
|
20
|
+
});
|
|
21
|
+
}
|
|
22
|
+
}
|
|
23
|
+
const descriptor = ({ text, excerpts, ...file }) => ({ ...file, characters: text.length });
|
|
24
|
+
const index = [...documents.values()].map(descriptor);
|
|
25
|
+
const reads = [];
|
|
26
|
+
const definitions = [
|
|
27
|
+
tool("list_evidence", "List retained paper, full implementation, raw data and execution logs. Pagination limits one response, not the accessible corpus.", {
|
|
28
|
+
offset: { type: "integer", minimum: 0 }, prefix: { type: "string" },
|
|
29
|
+
}, ["offset", "prefix"]),
|
|
30
|
+
tool("read_evidence", "Read exact retained characters. Follow nextOffset until the relevant implementation is complete. Legacy truncation is explicitly marked and cannot be recovered by this tool.", {
|
|
31
|
+
path: { type: "string" }, offset: { type: "integer", minimum: 0 },
|
|
32
|
+
}, ["path", "offset"]),
|
|
33
|
+
tool("search_evidence", "Literal search in all retained documents; page results with offset. A hit is not verification; read the surrounding implementation.", {
|
|
34
|
+
query: { type: "string", minLength: 1 }, prefix: { type: "string" }, offset: { type: "integer", minimum: 0 },
|
|
35
|
+
}, ["query", "prefix", "offset"]),
|
|
36
|
+
tool("read_evidence_json", "Read a JSON pointer from a retained JSON file, including exact scalars, keys or slices of arrays. No execution of scientific code.", {
|
|
37
|
+
path: { type: "string" }, pointer: { type: "string" }, offset: { type: "integer", minimum: 0 },
|
|
38
|
+
}, ["path", "pointer", "offset"]),
|
|
39
|
+
];
|
|
40
|
+
function execute(name, args) {
|
|
41
|
+
let result;
|
|
42
|
+
try {
|
|
43
|
+
const offset = args.offset;
|
|
44
|
+
if (!Number.isSafeInteger(offset) || offset < 0) throw Error("offset must be a nonnegative integer");
|
|
45
|
+
if (name === "list_evidence") {
|
|
46
|
+
const selected = index.filter(d => d.path.startsWith(args.prefix));
|
|
47
|
+
result = { files: selected.slice(offset, offset + 40), total: selected.length,
|
|
48
|
+
nextOffset: offset + 40 < selected.length ? offset + 40 : null };
|
|
49
|
+
} else if (name === "read_evidence") {
|
|
50
|
+
const d = documents.get(args.path); if (!d) throw Error("Path is not in retained evidence index");
|
|
51
|
+
result = { ...descriptor(d), offset, text: d.text.slice(offset, offset + PAGE),
|
|
52
|
+
nextOffset: offset + PAGE < d.text.length ? offset + PAGE : null };
|
|
53
|
+
} else if (name === "search_evidence") {
|
|
54
|
+
if (typeof args.query !== "string" || !args.query) throw Error("Nonempty literal query required");
|
|
55
|
+
const hits = [];
|
|
56
|
+
let total = 0;
|
|
57
|
+
for (const d of documents.values()) {
|
|
58
|
+
if (!d.path.startsWith(args.prefix)) continue;
|
|
59
|
+
let at = 0;
|
|
60
|
+
while ((at = d.text.indexOf(args.query, at)) !== -1) {
|
|
61
|
+
if (total >= offset && hits.length < 20) hits.push({ path: d.path, offset: at, preview: d.text.slice(Math.max(0, at - 100), at + 200) });
|
|
62
|
+
total++;
|
|
63
|
+
at += args.query.length;
|
|
64
|
+
}
|
|
65
|
+
}
|
|
66
|
+
result = { hits, total,
|
|
67
|
+
nextOffset: offset + 20 < total ? offset + 20 : null };
|
|
68
|
+
} else if (name === "read_evidence_json") {
|
|
69
|
+
const d = documents.get(args.path); if (!d) throw Error("Path is not in retained evidence index");
|
|
70
|
+
if (d.truncated) throw Error("Legacy file is truncated; not a complete JSON source");
|
|
71
|
+
let value = JSON.parse(d.text);
|
|
72
|
+
if (args.pointer !== "") {
|
|
73
|
+
if (!args.pointer.startsWith("/")) throw Error("Expected JSON pointer");
|
|
74
|
+
for (const part of args.pointer.slice(1).split("/")) {
|
|
75
|
+
const key = part.replace(/~1/g, "/").replace(/~0/g, "~");
|
|
76
|
+
if (value === null || typeof value !== "object" || !Object.hasOwn(value, key)) throw Error("Pointer absent");
|
|
77
|
+
value = value[key];
|
|
78
|
+
}
|
|
79
|
+
}
|
|
80
|
+
if (Array.isArray(value)) result = { length: value.length, values: value.slice(offset, offset + 20), nextOffset: offset + 20 < value.length ? offset + 20 : null };
|
|
81
|
+
else if (value && typeof value === "object") {
|
|
82
|
+
const keys = Object.keys(value); result = { keys: keys.slice(offset, offset + 100), total: keys.length, nextOffset: offset + 100 < keys.length ? offset + 100 : null };
|
|
83
|
+
} else result = { value };
|
|
84
|
+
if (JSON.stringify(result).length > PAGE) result = { error: "Selected values exceed a response page; read a narrower JSON pointer or use character pages" };
|
|
85
|
+
} else throw Error("Unknown evidence tool");
|
|
86
|
+
} catch (error) { result = { error: error.message }; }
|
|
87
|
+
reads.push({ name, args, responseDigest: `sha256:${sha256Value(result)}` });
|
|
88
|
+
return result;
|
|
89
|
+
}
|
|
90
|
+
return { definitions, execute, reads, index,
|
|
91
|
+
orientation: { digest: evidenceContext?.digest, available: documents.size > 0,
|
|
92
|
+
files: index.slice(0, 40), totalFiles: index.length, nextOffset: index.length > 40 ? 40 : null,
|
|
93
|
+
omissions: evidenceContext?.omissions ?? [],
|
|
94
|
+
instruction: "Use evidence tools to inspect complete source, called implementation and raw observations. An initial index is not evidence inspection. Read core update/evaluator code, including helpers and original source, before claiming comparability. Material omitted by the platform is not missing author material. Never infer correctness from labels, hashes or scalar proximity." } };
|
|
95
|
+
}
|