@iowarp/clio-coder 0.3.2 → 0.3.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +269 -458
- package/CONTRIBUTING.md +1 -1
- package/README.md +3 -3
- package/dist/{acp-BIYHVZIM.js → acp-S5R4RR5B.js} +7 -6
- package/dist/{agents-YT6SSRIT.js → agents-P6DMMVZY.js} +24 -21
- package/dist/assets/codewiki.json +1 -1
- package/dist/{auth-5TWEIYDN.js → auth-2XCZLPKS.js} +12 -8
- package/dist/{chunk-GGXXDWE4.js → chunk-22NAGB7X.js} +2 -2
- package/dist/{chunk-WMSVI4G2.js → chunk-2LZI5CAG.js} +133 -13
- package/dist/{chunk-OAO4GE4M.js → chunk-2TZWSW76.js} +2 -2
- package/dist/{chunk-OOJYHWRB.js → chunk-34475P3I.js} +2 -2
- package/dist/{chunk-WVO7V2QY.js → chunk-35MKKU5R.js} +4 -4
- package/dist/{chunk-LBNRH5WM.js → chunk-3HZ5RWN2.js} +5 -5
- package/dist/{chunk-AGYYIBLL.js → chunk-3JLKSKD7.js} +2 -2
- package/dist/{chunk-MBS4V7ZP.js → chunk-4JUF2NNX.js} +7 -7
- package/dist/{chunk-ZDOOVTXZ.js → chunk-4OC57DA6.js} +27 -4
- package/dist/chunk-5M54SPOL.js +926 -0
- package/dist/{chunk-STBPMHSX.js → chunk-7RXG6QRZ.js} +51 -11
- package/dist/{chunk-77VKQEHF.js → chunk-A2GZF7DC.js} +5 -5
- package/dist/{chunk-A3CYT5EX.js → chunk-AD2SYQYC.js} +55 -2
- package/dist/chunk-AOCYTWAV.js +449 -0
- package/dist/chunk-BEY543CS.js +258 -0
- package/dist/{chunk-6N5PTWMY.js → chunk-BP4OYD6A.js} +32 -13
- package/dist/chunk-BPGS2WCQ.js +612 -0
- package/dist/{chunk-J5HN4RYU.js → chunk-BRXQQJFP.js} +8 -8
- package/dist/chunk-CFGTUFWB.js +67 -0
- package/dist/{chunk-CBCAPZAA.js → chunk-E25LMLRW.js} +2 -2
- package/dist/{chunk-G2DE3C7R.js → chunk-EDRHSCIE.js} +4 -4
- package/dist/{chunk-4KLWL3UC.js → chunk-EFADSJET.js} +2 -2
- package/dist/{chunk-M6SHUN7Q.js → chunk-FO5ZOVUY.js} +2 -2
- package/dist/chunk-FYYLNIL5.js +313 -0
- package/dist/{chunk-EPVUXGXG.js → chunk-HV5X7OR2.js} +14 -12
- package/dist/{chunk-TZTZS7QK.js → chunk-HXG4IURW.js} +5 -3
- package/dist/{chunk-IGLFWIYI.js → chunk-K6WL7QZT.js} +3 -3
- package/dist/chunk-K7VKOLQQ.js +15 -0
- package/dist/{chunk-BMEMKKIT.js → chunk-KOHPCX4K.js} +2 -2
- package/dist/{chunk-V4RXGQ5Q.js → chunk-KRPY7NTG.js} +10 -7
- package/dist/chunk-LL4KHSZI.js +22 -0
- package/dist/{chunk-KJ5LWLOE.js → chunk-MEQ45TQ4.js} +15 -9
- package/dist/{chunk-5UUP6MWO.js → chunk-MV3K5QF2.js} +5 -436
- package/dist/{chunk-AO4RKG4M.js → chunk-N4CZJQRK.js} +5 -5
- package/dist/{chunk-ARBGF5F7.js → chunk-NILBFAPG.js} +14 -8
- package/dist/chunk-OZNBF4L3.js +23 -0
- package/dist/{verify-G6V4D2G7.js → chunk-PCZJO5TI.js} +127 -42
- package/dist/chunk-QQK64KLB.js +1360 -0
- package/dist/{chunk-6EJV5X2W.js → chunk-QQL5RT5M.js} +979 -1619
- package/dist/{chunk-LZSJBIVT.js → chunk-QWU7ZBO7.js} +70 -720
- package/dist/{chunk-2EHAIA3X.js → chunk-RD5U66HV.js} +3 -3
- package/dist/{chunk-OKGUZO2U.js → chunk-SPULKLCF.js} +4 -3
- package/dist/{chunk-OQ33BKR3.js → chunk-TTNYS3EA.js} +3 -60
- package/dist/chunk-TW3WDMVS.js +677 -0
- package/dist/chunk-TZSKNMZG.js +434 -0
- package/dist/{chunk-7MNJORFF.js → chunk-UL3WSD3F.js} +6 -1
- package/dist/{chunk-7EYHLWU7.js → chunk-UZHIZC5S.js} +7 -7
- package/dist/{chunk-QTYWRVRA.js → chunk-VAWWTKDP.js} +8 -8
- package/dist/{chunk-X75S7HFS.js → chunk-VEZEGCGW.js} +214 -20
- package/dist/{chunk-OHHN2SO4.js → chunk-VMNQ6OZA.js} +98 -202
- package/dist/chunk-VSNATDE6.js +122 -0
- package/dist/chunk-W6GROXXM.js +69 -0
- package/dist/chunk-WPQLXFOZ.js +375 -0
- package/dist/{chunk-ORBHGJC5.js → chunk-WR67VIZY.js} +3 -3
- package/dist/{chunk-3ZXDFGR5.js → chunk-X6COSD2O.js} +5 -5
- package/dist/chunk-ZGVHUX3M.js +66 -0
- package/dist/{chunk-MAW544W2.js → chunk-ZWMF7253.js} +4 -4
- package/dist/{chunk-MQSRRFWA.js → chunk-ZYKPLLNQ.js} +563 -546
- package/dist/cli/index.js +27 -23
- package/dist/{clio-4LY5K2AC.js → clio-J5JIOIDS.js} +7 -6
- package/dist/{code-nav-7AX6FYE6.js → code-nav-AXCXSBHX.js} +5 -3
- package/dist/{config-GTLUW2PR.js → config-OEBMIN2U.js} +37 -27
- package/dist/{configure-R6A64DHX.js → configure-PUQOSIXQ.js} +16 -13
- package/dist/{context-5VKGUVJJ.js → context-EKDCKUUZ.js} +82 -7
- package/dist/{context-RW5HC47S.js → context-MGSE4Z2T.js} +33 -23
- package/dist/{context-JFZEJ7W5.js → context-URSXPBCK.js} +17 -9
- package/dist/{context-clear-6ZHBAZZT.js → context-clear-KDAJRNUK.js} +33 -23
- package/dist/context-working-set-SBKMPPI2.js +1552 -0
- package/dist/{dispatch-runner-VKBRCWQC.js → dispatch-runner-MSWN72NK.js} +43 -29
- package/dist/{doctor-KI767GSN.js → doctor-7BSE27PJ.js} +10 -10
- package/dist/{eval-XSSNATB4.js → eval-IZGDOO4H.js} +9 -8
- package/dist/{evidence-UA6AWDQQ.js → evidence-SR7WXB5B.js} +51 -23
- package/dist/{evolve-QNTFGV6Z.js → evolve-K7VE2CBX.js} +30 -20
- package/dist/{fleet-Q7UOMUSG.js → fleet-7XMJNQNF.js} +48 -38
- package/dist/{fleet-preflight-DDN536IT.js → fleet-preflight-AQNAH644.js} +3 -3
- package/dist/{init-WBB65ZHQ.js → init-JGNPAYXT.js} +41 -31
- package/dist/{memory-MD3O64RI.js → memory-4ALKDJ4Q.js} +32 -22
- package/dist/{models-BZU34YWD.js → models-ZMMLFJNN.js} +22 -19
- package/dist/{monitor-MEQA5C3I.js → monitor-2F3T5KHP.js} +55 -43
- package/dist/{orchestrator-CGFKEP27.js → orchestrator-ORHT43JB.js} +2507 -1896
- package/dist/{reset-L2FQEE3E.js → reset-NXGTYNUO.js} +4 -3
- package/dist/{run-IV4Q6RLN.js → run-RF4WJGMT.js} +51 -41
- package/dist/{share-S5BZQC5I.js → share-UT3W6E4M.js} +5 -4
- package/dist/{skills-LQEKRDTN.js → skills-PSACKC5Q.js} +2 -2
- package/dist/{skills-eval-3DC4HEWS.js → skills-eval-WJSI55RZ.js} +34 -24
- package/dist/{targets-C4SSGQOB.js → targets-PIIRAOYS.js} +23 -20
- package/dist/{terminal-lease-IT5JW2NR.js → terminal-lease-ULWXWNVY.js} +5 -4
- package/dist/{upgrade-7TT7SQ3G.js → upgrade-346TZ6AV.js} +18 -17
- package/dist/{usage-GV4PKT3M.js → usage-6KKXR32N.js} +34 -24
- package/dist/verifiers-4UUM6TEE.js +1214 -0
- package/dist/verify-X5HDROLA.js +25 -0
- package/dist/{wiki-generate-DQF6Z66B.js → wiki-generate-7STOCIFZ.js} +42 -31
- package/dist/worker/entry.js +33 -24
- package/docs/README.md +8 -7
- package/docs/acp.md +1 -1
- package/docs/alcf-provider.md +1 -1
- package/docs/architecture.md +2 -2
- package/docs/artifact-versions.md +1 -1
- package/docs/built-in-agents.md +1 -1
- package/docs/capacity-and-scheduling.md +1 -1
- package/docs/commands-and-modes.md +53 -21
- package/docs/config-knobs-audit.md +1 -2
- package/docs/configuration-and-targets.md +15 -1
- package/docs/context-engine.md +64 -12
- package/docs/context-working-set.md +194 -0
- package/docs/development-pipeline.md +1 -1
- package/docs/documentation-coverage.md +5 -5
- package/docs/documentation-guide.md +6 -5
- package/docs/environment-variables.md +2 -1
- package/docs/eval-runner.md +1 -1
- package/docs/evals-internal.md +14 -1
- package/docs/evidence-and-memory.md +74 -2
- package/docs/evolution.md +1 -1
- package/docs/exit-codes-and-output.md +1 -1
- package/docs/extensions-and-sharing.md +2 -2
- package/docs/fleet-dispatch.md +22 -7
- package/docs/glossary.md +21 -1
- package/docs/installation-and-lifecycle.md +6 -6
- package/docs/middleware-and-components.md +1 -1
- package/docs/model-catalog.md +7 -9
- package/docs/observability.md +4 -4
- package/docs/performance-methodology.md +2 -2
- package/docs/proactive-memory.md +1 -1
- package/docs/prompt-envelope-and-tools.md +4 -4
- package/docs/provider-adapter-cookbook.md +1 -1
- package/docs/release-cut-checklist.md +35 -35
- package/docs/safety-model.md +23 -4
- package/docs/scientific-validation.md +21 -3
- package/docs/session-lifecycle.md +3 -3
- package/docs/skills-marketplace.md +1 -1
- package/docs/tool-usage.md +79 -12
- package/docs/trace-store.md +1 -1
- package/docs/troubleshooting.md +1 -1
- package/docs/tui-design.md +2 -2
- package/docs/worker-dispatch-mechanics.md +11 -1
- package/package.json +8 -11
- package/skills/meta/clio-test/SKILL.md +20 -17
- package/skills/meta/clio-test/evals.md +3 -3
- package/skills/meta/clio-test/references/harness.md +35 -6
- package/skills/meta/clio-test/references/test-map.md +20 -10
- package/skills/registry.yaml +2 -2
- package/skills/skill-marketplace.json +1 -1
- package/src/cli/context-working-set.ts +513 -0
- package/src/cli/context.ts +8 -0
- package/src/cli/evidence.ts +20 -2
- package/src/cli/index.ts +4 -0
- package/src/cli/verifiers.ts +325 -0
- package/src/core/bash-exec.ts +39 -14
- package/src/core/bus-events.ts +19 -4
- package/src/core/config.ts +54 -0
- package/src/core/defaults.ts +50 -3
- package/src/core/git-commit-attribution.ts +46 -21
- package/src/core/verification-scripts.ts +6 -0
- package/src/domains/agents/builtins/verifier.md +3 -0
- package/src/domains/config/classify.ts +1 -0
- package/src/domains/config/keybindings.ts +3 -3
- package/src/domains/context/working-set/contract.ts +161 -0
- package/src/domains/context/working-set/defaults.ts +28 -0
- package/src/domains/context/working-set/engine.ts +203 -0
- package/src/domains/context/working-set/fold.ts +62 -0
- package/src/domains/context/working-set/horizon.ts +38 -0
- package/src/domains/context/working-set/marker.ts +103 -0
- package/src/domains/context/working-set/path-index.ts +436 -0
- package/src/domains/context/working-set/payload.ts +152 -0
- package/src/domains/context/working-set/policies/age-horizon.ts +55 -0
- package/src/domains/context/working-set/policies/index.ts +21 -0
- package/src/domains/context/working-set/policies/structural.ts +160 -0
- package/src/domains/context/working-set/project.ts +132 -0
- package/src/domains/context/working-set/protect.ts +109 -0
- package/src/domains/context/working-set/recall.ts +177 -0
- package/src/domains/context/working-set/replay/controls.ts +112 -0
- package/src/domains/context/working-set/replay/load-clio.ts +199 -0
- package/src/domains/context/working-set/replay/metrics.ts +185 -0
- package/src/domains/context/working-set/replay/reference-graph.ts +79 -0
- package/src/domains/context/working-set/replay/report.ts +139 -0
- package/src/domains/context/working-set/replay/runner.ts +325 -0
- package/src/domains/context/working-set/replay/synthetic.ts +422 -0
- package/src/domains/context/working-set/replay/trace.ts +21 -0
- package/src/domains/context/working-set/visible.ts +54 -0
- package/src/domains/evidence/build.ts +112 -45
- package/src/domains/evidence/eval.ts +24 -7
- package/src/domains/evidence/index.ts +53 -0
- package/src/domains/evidence/ordering.ts +12 -0
- package/src/domains/evidence/run-trust.ts +221 -0
- package/src/domains/evidence/store.ts +46 -6
- package/src/domains/evidence/trust-status.ts +854 -0
- package/src/domains/evidence/types.ts +26 -0
- package/src/domains/middleware/memory-intervention.ts +3 -0
- package/src/domains/middleware/stalled-turn.ts +165 -4
- package/src/domains/safety/autonomy.ts +1 -1
- package/src/domains/safety/default-path-policy.ts +8 -0
- package/src/domains/safety/finish-contract.ts +4 -3
- package/src/domains/safety/policy-engine.ts +48 -6
- package/src/domains/session/compaction/compact.ts +23 -1
- package/src/domains/session/compaction/cut-point.ts +2 -0
- package/src/domains/session/compaction/tokens.ts +16 -1
- package/src/domains/session/context-ledger.ts +2 -0
- package/src/domains/session/entries.ts +107 -1
- package/src/domains/session/manager.ts +9 -2
- package/src/domains/session/migrations/index.ts +22 -3
- package/src/engine/acp/server.ts +3 -0
- package/src/engine/agent.ts +18 -1
- package/src/engine/session.ts +9 -3
- package/src/entry/orchestrator.ts +16 -4
- package/src/interactive/chat-loop-messages.ts +18 -6
- package/src/interactive/chat-panel.ts +571 -244
- package/src/interactive/chat-renderer.ts +79 -39
- package/src/interactive/context-meter.ts +10 -0
- package/src/interactive/context-overlay.ts +81 -6
- package/src/interactive/context-recall-command.ts +110 -0
- package/src/interactive/editor-submit.ts +26 -1
- package/src/interactive/footer/widgets.ts +22 -20
- package/src/interactive/footer-panel.ts +6 -1
- package/src/interactive/interactive-application.ts +2 -0
- package/src/interactive/interactive-event-projection.ts +12 -0
- package/src/interactive/interactive-slash-runtime.ts +49 -8
- package/src/interactive/model-session-replay.ts +21 -0
- package/src/interactive/overlay-general-openers.ts +6 -0
- package/src/interactive/overlay-session-lifecycle.ts +8 -4
- package/src/interactive/overlays/ask-user.ts +146 -24
- package/src/interactive/renderers/tool-execution.ts +167 -56
- package/src/interactive/session-transcript.ts +2 -2
- package/src/interactive/slash-commands.ts +29 -2
- package/src/interactive/status/index.ts +12 -1
- package/src/interactive/status/reasoning.ts +87 -0
- package/src/interactive/status/summary.ts +13 -2
- package/src/interactive/transcript-detail.ts +120 -0
- package/src/interactive/turn-context.ts +238 -88
- package/src/interactive/turn-middleware.ts +6 -6
- package/src/tools/agent-tools.ts +11 -4
- package/src/tools/bash.ts +144 -82
- package/src/tools/builtin-tool-catalog.ts +18 -6
- package/src/tools/context/index.ts +105 -3
- package/src/tools/context/surface.ts +3 -2
- package/src/tools/core-bootstrap.ts +21 -0
- package/src/tools/dispatch-runner.ts +9 -7
- package/src/tools/monitor.ts +28 -20
- package/src/tools/presentation.ts +107 -0
- package/src/tools/registry.ts +65 -7
- package/src/tools/result-disposition.ts +550 -0
- package/src/tools/result-shaping.ts +262 -19
- package/src/tools/safe-exec.ts +2 -0
- package/src/tools/verify/authoring.ts +1119 -0
- package/src/tools/verify/catalog.ts +346 -0
- package/src/tools/verify/index.ts +13 -3
- package/src/tools/verify/scripts.ts +135 -37
- package/src/tools/verify/surface.ts +9 -5
- package/src/tools/worker-evidence.ts +35 -12
- package/dist/chunk-MNA4JGU4.js +0 -255
- package/dist/chunk-SRF2PJNW.js +0 -184
- package/dist/chunk-T6YILFSB.js +0 -80
- package/dist/chunk-VAKQQHWR.js +0 -434
|
@@ -41,6 +41,7 @@ import { type ReceiptIntegrityResult, verifyReceiptIntegrity } from "../domains/
|
|
|
41
41
|
import { explainRouteDecision } from "../domains/dispatch/routing-intent.js";
|
|
42
42
|
import type { RunGateProvenance, RunGateSubjectRef, RunPlanProvenance, RunReceipt } from "../domains/dispatch/types.js";
|
|
43
43
|
import { extractRunProvenance, provenanceCompactSuffix } from "../domains/evidence/provenance.js";
|
|
44
|
+
import { adaptRunReceiptTrustStatus } from "../domains/evidence/trust-status.js";
|
|
44
45
|
import type { AutonomyLevel } from "../domains/safety/autonomy.js";
|
|
45
46
|
import {
|
|
46
47
|
type CandidateWorktree,
|
|
@@ -458,8 +459,8 @@ function formatDispatchOutput(
|
|
|
458
459
|
.map((run) => integrityFailureBanner(run))
|
|
459
460
|
.filter((banner): banner is string => banner !== null);
|
|
460
461
|
const needsSpotCheck = runs.some((run) => {
|
|
461
|
-
const state = run.receipt.
|
|
462
|
-
return state === "
|
|
462
|
+
const state = adaptRunReceiptTrustStatus(run.receipt, { integrity: run.integrity }).validationGrounding.state;
|
|
463
|
+
return state === "absent" || state === "unknown" || state === "ungrounded";
|
|
463
464
|
});
|
|
464
465
|
const lines = [
|
|
465
466
|
`dispatch (${mode}) total=${runs.length} failed=${failed.length}`,
|
|
@@ -486,9 +487,8 @@ function formatDispatchOutput(
|
|
|
486
487
|
// Evidence confidence comes from the sealed receipt. A receipt that fails
|
|
487
488
|
// integrity cannot be read as evidence at all.
|
|
488
489
|
const verification = run.integrity.ok ? receipt.verification : UNVERIFIABLE_RECEIPT_VERIFICATION;
|
|
489
|
-
const
|
|
490
|
-
|
|
491
|
-
}`;
|
|
490
|
+
const trustStatus = adaptRunReceiptTrustStatus({ ...receipt, verification }, { integrity: run.integrity });
|
|
491
|
+
const evidenceSuffix = ` ${receiptEvidenceLabels(receipt, verification, run.integrity).join(" ")}`;
|
|
492
492
|
const routingSuffix =
|
|
493
493
|
run.integrity.ok && receipt.routeDecision !== undefined && receipt.routingIntent !== undefined
|
|
494
494
|
? ` route_decision=${receipt.routeDecision.decisionHash} route_mode=${receipt.routeDecision.mode}`
|
|
@@ -509,9 +509,9 @@ function formatDispatchOutput(
|
|
|
509
509
|
: "(worker text withheld because receipt integrity failed)";
|
|
510
510
|
return [
|
|
511
511
|
`- ${stepLabel}${receipt.runId} agent=${receipt.agentId} exit=${receipt.exitCode} target=${receipt.targetId} model=${receipt.wireModelId} tokens=${receipt.tokenCount} receipt=${receiptPath ?? "n/a"}${evidenceSuffix}${outcomeSuffix}${noteSuffix}${failure}${provenance}${routingSuffix}`,
|
|
512
|
-
` ${workerTextLabel(
|
|
512
|
+
` ${workerTextLabel(trustStatus)}`,
|
|
513
513
|
...output.split("\n").map((line) => ` ${line}`),
|
|
514
|
-
...workerTextNonEvidenceNotices(receipt,
|
|
514
|
+
...workerTextNonEvidenceNotices(receipt, trustStatus, answerText).map((notice) => ` ${notice}`),
|
|
515
515
|
];
|
|
516
516
|
}),
|
|
517
517
|
];
|
|
@@ -555,6 +555,7 @@ function dispatchDetails(
|
|
|
555
555
|
// Additive provenance keys only; folded in when the receipt carries the
|
|
556
556
|
// field so a run entry without them keeps its exact shape.
|
|
557
557
|
const provenance = extractRunProvenance(receipt);
|
|
558
|
+
const trustStatus = adaptRunReceiptTrustStatus(receipt, { integrity });
|
|
558
559
|
return {
|
|
559
560
|
runId: receipt.runId,
|
|
560
561
|
agentId: receipt.agentId,
|
|
@@ -566,6 +567,7 @@ function dispatchDetails(
|
|
|
566
567
|
// receipt is machine-visible here too.
|
|
567
568
|
verification: integrity.ok ? receipt.verification : UNVERIFIABLE_RECEIPT_VERIFICATION,
|
|
568
569
|
receiptIntegrity: integrity,
|
|
570
|
+
trustStatus,
|
|
569
571
|
...(receipt.outcome !== undefined && receipt.outcome !== "succeeded"
|
|
570
572
|
? { outcome: receipt.outcome, outcomeDetail: receipt.outcomeDetail ?? null }
|
|
571
573
|
: {}),
|
package/src/tools/monitor.ts
CHANGED
|
@@ -4,13 +4,18 @@ import { renderAgentLedgerBoard } from "../domains/dispatch/agent-ledger-store.j
|
|
|
4
4
|
import type { DurableAssignmentRecord } from "../domains/dispatch/assignment-store.js";
|
|
5
5
|
import type { DispatchContract } from "../domains/dispatch/contract.js";
|
|
6
6
|
import { UNVERIFIABLE_RECEIPT_VERIFICATION } from "../domains/dispatch/receipt-findings.js";
|
|
7
|
-
import {
|
|
7
|
+
import type { ReceiptIntegrityResult } from "../domains/dispatch/receipt-integrity.js";
|
|
8
8
|
import {
|
|
9
9
|
isTerminalRunEnvelope,
|
|
10
10
|
type RunEnvelope,
|
|
11
11
|
type RunReceipt,
|
|
12
12
|
type RunReceiptVerification,
|
|
13
13
|
} from "../domains/dispatch/types.js";
|
|
14
|
+
import {
|
|
15
|
+
adaptRunReceiptTrustStatus,
|
|
16
|
+
type CanonicalTrustStatus,
|
|
17
|
+
inspectRunReceiptTrustStatus,
|
|
18
|
+
} from "../domains/evidence/trust-status.js";
|
|
14
19
|
import { COST_NOT_MEASURED, costAggregateForAmount, formatCostAggregate } from "../domains/observability/index.js";
|
|
15
20
|
import type { DispatchRunEventRegistry } from "./dispatch.js";
|
|
16
21
|
import { monitorToolSurface } from "./monitor-surface.js";
|
|
@@ -333,14 +338,18 @@ function runReceipt(deps: MonitorToolDeps, runId: string): ToolResult {
|
|
|
333
338
|
const body = truncateUtf8(raw, RECEIPT_MAX_BYTES, `\n[receipt truncated; read ${run.receiptPath} for the rest]`);
|
|
334
339
|
let receipt: RunReceipt | null = null;
|
|
335
340
|
let receiptIntegrity: ReceiptIntegrityResult;
|
|
341
|
+
let trustStatus: CanonicalTrustStatus;
|
|
336
342
|
try {
|
|
337
343
|
receipt = JSON.parse(raw) as RunReceipt;
|
|
338
|
-
|
|
344
|
+
const inspection = inspectRunReceiptTrustStatus(receipt, run);
|
|
345
|
+
receiptIntegrity = inspection.integrity;
|
|
346
|
+
trustStatus = inspection.status;
|
|
339
347
|
} catch (err) {
|
|
340
348
|
receiptIntegrity = {
|
|
341
349
|
ok: false,
|
|
342
350
|
reason: `receipt invalid: ${err instanceof Error ? err.message : String(err)}`,
|
|
343
351
|
};
|
|
352
|
+
trustStatus = adaptRunReceiptTrustStatus(null, { integrity: receiptIntegrity });
|
|
344
353
|
}
|
|
345
354
|
return {
|
|
346
355
|
kind: "ok",
|
|
@@ -350,6 +359,7 @@ function runReceipt(deps: MonitorToolDeps, runId: string): ToolResult {
|
|
|
350
359
|
runId,
|
|
351
360
|
receiptPath: run.receiptPath,
|
|
352
361
|
receiptIntegrity,
|
|
362
|
+
trustStatus,
|
|
353
363
|
...(receipt !== null && receiptIntegrity.ok
|
|
354
364
|
? {
|
|
355
365
|
evidenceVerification: receipt.verification,
|
|
@@ -366,6 +376,7 @@ interface DurableRunEvidence {
|
|
|
366
376
|
output: RunReceipt["output"] | null;
|
|
367
377
|
verification: RunReceiptVerification;
|
|
368
378
|
integrity: ReceiptIntegrityResult;
|
|
379
|
+
trustStatus: CanonicalTrustStatus;
|
|
369
380
|
integrityNote: string | null;
|
|
370
381
|
integrityFailure: boolean;
|
|
371
382
|
}
|
|
@@ -376,6 +387,7 @@ function unavailableRunEvidence(reason: string, note: string, integrityFailure =
|
|
|
376
387
|
output: null,
|
|
377
388
|
verification: UNVERIFIABLE_RECEIPT_VERIFICATION,
|
|
378
389
|
integrity: { ok: false, reason },
|
|
390
|
+
trustStatus: adaptRunReceiptTrustStatus(null, { integrity: { ok: false, reason } }),
|
|
379
391
|
integrityNote: note,
|
|
380
392
|
integrityFailure,
|
|
381
393
|
};
|
|
@@ -410,29 +422,24 @@ function durableRunEvidence(run: RunEnvelope | null): DurableRunEvidence {
|
|
|
410
422
|
`receipt integrity unavailable: cannot read or parse ${run.receiptPath} (${detail}); worker text is unavailable and validation is unknown.`,
|
|
411
423
|
);
|
|
412
424
|
}
|
|
413
|
-
|
|
414
|
-
|
|
415
|
-
integrity = verifyReceiptIntegrity(receipt, run);
|
|
416
|
-
} catch (err) {
|
|
417
|
-
const detail = err instanceof Error ? err.message : String(err);
|
|
418
|
-
return unavailableRunEvidence(
|
|
419
|
-
`receipt invalid: ${detail}`,
|
|
420
|
-
`receipt integrity failed: invalid receipt (${detail}); worker text is withheld as untrusted and validation is unknown.`,
|
|
421
|
-
true,
|
|
422
|
-
);
|
|
423
|
-
}
|
|
425
|
+
const inspection = inspectRunReceiptTrustStatus(receipt, run);
|
|
426
|
+
const integrity = inspection.integrity;
|
|
424
427
|
if (!integrity.ok) {
|
|
425
|
-
return
|
|
426
|
-
|
|
427
|
-
|
|
428
|
-
|
|
429
|
-
|
|
428
|
+
return {
|
|
429
|
+
...unavailableRunEvidence(
|
|
430
|
+
integrity.reason,
|
|
431
|
+
`receipt integrity failed: ${integrity.reason}; worker text is withheld as untrusted and validation is unknown.`,
|
|
432
|
+
true,
|
|
433
|
+
),
|
|
434
|
+
trustStatus: inspection.status,
|
|
435
|
+
};
|
|
430
436
|
}
|
|
431
437
|
return {
|
|
432
438
|
receipt,
|
|
433
439
|
output: receipt.output ?? null,
|
|
434
440
|
verification: receipt.verification,
|
|
435
441
|
integrity,
|
|
442
|
+
trustStatus: inspection.status,
|
|
436
443
|
integrityNote: null,
|
|
437
444
|
integrityFailure: false,
|
|
438
445
|
};
|
|
@@ -534,7 +541,7 @@ function collectRunLine(row: CollectedRunRow): string[] {
|
|
|
534
541
|
);
|
|
535
542
|
}
|
|
536
543
|
if (row.evidence.integrityNote) lines.push(` ${row.evidence.integrityNote}`);
|
|
537
|
-
lines.push(` ${workerTextLabel(row.evidence.
|
|
544
|
+
lines.push(` ${workerTextLabel(row.evidence.trustStatus)}`);
|
|
538
545
|
const output = row.evidence.output;
|
|
539
546
|
if (output) {
|
|
540
547
|
const capped = truncateUtf8(output.text, COLLECT_TEXT_BYTES, "...");
|
|
@@ -543,7 +550,7 @@ function collectRunLine(row: CollectedRunRow): string[] {
|
|
|
543
550
|
lines.push(` agent output${qualifier}${truncatedNote}:`, ...capped.split("\n").map((line) => ` ${line}`));
|
|
544
551
|
if (row.evidence.receipt) {
|
|
545
552
|
lines.push(
|
|
546
|
-
...workerTextNonEvidenceNotices(row.evidence.receipt, row.evidence.
|
|
553
|
+
...workerTextNonEvidenceNotices(row.evidence.receipt, row.evidence.trustStatus, output.text).map(
|
|
547
554
|
(notice) => ` ${notice}`,
|
|
548
555
|
),
|
|
549
556
|
);
|
|
@@ -693,6 +700,7 @@ async function runCollect(
|
|
|
693
700
|
exitCode: row.run?.exitCode ?? null,
|
|
694
701
|
receiptPath: row.run?.receiptPath ?? null,
|
|
695
702
|
receiptIntegrity: row.evidence.integrity,
|
|
703
|
+
trustStatus: row.evidence.trustStatus,
|
|
696
704
|
evidenceVerification: row.evidence.verification,
|
|
697
705
|
briefing: row.evidence.receipt?.briefing ?? null,
|
|
698
706
|
projectContext: row.evidence.receipt?.projectContext ?? null,
|
|
@@ -0,0 +1,107 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Typed tool presentation policy. Answers, per tool, what the transcript's
|
|
3
|
+
* balanced (`/output default`) view needs to know: does the block open folded
|
|
4
|
+
* or expanded, does the folded row keep a mutation diff visible, and does a
|
|
5
|
+
* failed folded row carry an output excerpt.
|
|
6
|
+
*
|
|
7
|
+
* The panel must not decide any of that by tool name. It asks this module,
|
|
8
|
+
* which resolves the answer from two inputs: the registered presentation
|
|
9
|
+
* metadata for the tool (declared once here and attached to `ToolMetadata` by
|
|
10
|
+
* the builtin catalog) and the argument-sensitive resource-read rule. The
|
|
11
|
+
* lookup is a plain object read, so it stays cheap enough to call on every
|
|
12
|
+
* frame, and it needs no registry instance: the live chat panel has none.
|
|
13
|
+
*
|
|
14
|
+
* Pure module: no I/O, no registry construction, no UI imports.
|
|
15
|
+
*/
|
|
16
|
+
|
|
17
|
+
import { ToolNames } from "../core/tool-names.js";
|
|
18
|
+
|
|
19
|
+
export type ToolFoldDefault = "expanded" | "folded";
|
|
20
|
+
|
|
21
|
+
export interface ToolPresentationPolicy {
|
|
22
|
+
/** How a fresh block for this call renders before the operator touches it. */
|
|
23
|
+
foldDefault: ToolFoldDefault;
|
|
24
|
+
/**
|
|
25
|
+
* Keep the mutation diff under the folded row. A folded `edit` that hides
|
|
26
|
+
* what it changed tells the operator nothing they could act on; the diff is
|
|
27
|
+
* the row's whole point and stays visible, bounded, until the body is opened.
|
|
28
|
+
*/
|
|
29
|
+
showDiffWhenFolded: boolean;
|
|
30
|
+
/**
|
|
31
|
+
* Carry the last non-empty output line on a failed folded row. Bash pioneered
|
|
32
|
+
* this so a failed command stays diagnosable without opening its body; every
|
|
33
|
+
* tool that fails with text gets the same courtesy.
|
|
34
|
+
*/
|
|
35
|
+
failureExcerpt: boolean;
|
|
36
|
+
}
|
|
37
|
+
|
|
38
|
+
const FOLDED: ToolPresentationPolicy = { foldDefault: "folded", showDiffWhenFolded: false, failureExcerpt: true };
|
|
39
|
+
const FOLDED_WITH_DIFF: ToolPresentationPolicy = {
|
|
40
|
+
foldDefault: "folded",
|
|
41
|
+
showDiffWhenFolded: true,
|
|
42
|
+
failureExcerpt: true,
|
|
43
|
+
};
|
|
44
|
+
|
|
45
|
+
/**
|
|
46
|
+
* Per-tool presentation declarations. Every builtin folds by default: a
|
|
47
|
+
* routine turn of six reads used to open six bodies, and the one-line row
|
|
48
|
+
* already carries the call, its outcome facts, size, and settlement. Mutations
|
|
49
|
+
* keep their diff under the folded row. Everything unlisted, including dynamic
|
|
50
|
+
* tools, folds the same way.
|
|
51
|
+
*/
|
|
52
|
+
export const TOOL_PRESENTATION: Readonly<Record<string, ToolPresentationPolicy>> = {
|
|
53
|
+
[ToolNames.Read]: FOLDED,
|
|
54
|
+
[ToolNames.Grep]: FOLDED,
|
|
55
|
+
[ToolNames.Find]: FOLDED,
|
|
56
|
+
[ToolNames.Ls]: FOLDED,
|
|
57
|
+
[ToolNames.CodeNav]: FOLDED,
|
|
58
|
+
[ToolNames.Context]: FOLDED,
|
|
59
|
+
[ToolNames.CredentialPresent]: FOLDED,
|
|
60
|
+
[ToolNames.Write]: FOLDED_WITH_DIFF,
|
|
61
|
+
[ToolNames.Edit]: FOLDED_WITH_DIFF,
|
|
62
|
+
[ToolNames.Bash]: FOLDED,
|
|
63
|
+
[ToolNames.Git]: FOLDED,
|
|
64
|
+
[ToolNames.Verify]: FOLDED,
|
|
65
|
+
[ToolNames.Dispatch]: FOLDED,
|
|
66
|
+
[ToolNames.Monitor]: FOLDED,
|
|
67
|
+
[ToolNames.Steer]: FOLDED,
|
|
68
|
+
[ToolNames.Tasks]: FOLDED,
|
|
69
|
+
[ToolNames.Ledger]: FOLDED,
|
|
70
|
+
[ToolNames.WebFetch]: FOLDED,
|
|
71
|
+
[ToolNames.AskUser]: FOLDED,
|
|
72
|
+
[ToolNames.Artifact]: FOLDED,
|
|
73
|
+
};
|
|
74
|
+
|
|
75
|
+
function readStringField(args: unknown, key: string): string | null {
|
|
76
|
+
if (typeof args !== "object" || args === null || Array.isArray(args)) return null;
|
|
77
|
+
const value = (args as Record<string, unknown>)[key];
|
|
78
|
+
return typeof value === "string" && value.length > 0 ? value : null;
|
|
79
|
+
}
|
|
80
|
+
|
|
81
|
+
/**
|
|
82
|
+
* Compact resource-read classification. Reads of skill/handbook/agent
|
|
83
|
+
* instruction files and docs pages collapse to one labeled line and never
|
|
84
|
+
* auto-expand; their bodies are reference material, not task output.
|
|
85
|
+
*/
|
|
86
|
+
export function classifyResourceRead(toolName: string, args: unknown): string | null {
|
|
87
|
+
if (toolName !== ToolNames.Read) return null;
|
|
88
|
+
const path = readStringField(args, "path");
|
|
89
|
+
if (path === null) return null;
|
|
90
|
+
const normalized = path.replace(/\\/g, "/");
|
|
91
|
+
const base = normalized.split("/").pop() ?? "";
|
|
92
|
+
if (base === "SKILL.md") return "skill";
|
|
93
|
+
if (base === "CLIO-CODER.md") return "handbook";
|
|
94
|
+
if (base === "AGENTS.md") return "agents";
|
|
95
|
+
if (/(^|\/)docs\//.test(normalized)) return "docs";
|
|
96
|
+
return null;
|
|
97
|
+
}
|
|
98
|
+
|
|
99
|
+
/**
|
|
100
|
+
* Resolve the presentation policy for one call. Argument-sensitive rules win
|
|
101
|
+
* over the per-tool declaration because a resource read is a property of the
|
|
102
|
+
* path, not of the `read` tool.
|
|
103
|
+
*/
|
|
104
|
+
export function toolPresentationPolicy(toolName: string, args: unknown): ToolPresentationPolicy {
|
|
105
|
+
if (classifyResourceRead(toolName, args) !== null) return FOLDED;
|
|
106
|
+
return TOOL_PRESENTATION[toolName] ?? FOLDED;
|
|
107
|
+
}
|
package/src/tools/registry.ts
CHANGED
|
@@ -26,7 +26,9 @@ import { hashToolCall } from "../domains/safety/loop-detector.js";
|
|
|
26
26
|
import { detectValidationCommand } from "../domains/safety/protected-artifacts.js";
|
|
27
27
|
import { askUserExposure } from "./ask-user.js";
|
|
28
28
|
import { type DispatchPlanView, describeDispatchPlan } from "./dispatch-plan.js";
|
|
29
|
-
import {
|
|
29
|
+
import type { ToolPresentationPolicy } from "./presentation.js";
|
|
30
|
+
import type { ToolResultDisposition } from "./result-disposition.js";
|
|
31
|
+
import { DEFAULT_TOOL_RESULT_MAX_BYTES, shapeToolResult } from "./result-shaping.js";
|
|
30
32
|
|
|
31
33
|
/**
|
|
32
34
|
* Tool registry. Admission point for every tool call. Delegates classification
|
|
@@ -55,6 +57,8 @@ export interface ToolSourceInfo {
|
|
|
55
57
|
export interface ToolResultSizePolicy {
|
|
56
58
|
kind: "exact" | "bounded" | "summary" | "truncate";
|
|
57
59
|
maxBytes?: number;
|
|
60
|
+
/** Scratch retention ceiling for this tool; defaults to the generic 10 MiB cap. */
|
|
61
|
+
offloadMaxBytes?: number;
|
|
58
62
|
followUpHint?: string;
|
|
59
63
|
}
|
|
60
64
|
|
|
@@ -67,6 +71,8 @@ export interface ToolMetadata {
|
|
|
67
71
|
retrySafety: ToolRetrySafety;
|
|
68
72
|
/** Expected result-size behavior at the registry boundary. */
|
|
69
73
|
resultSizePolicy: ToolResultSizePolicy;
|
|
74
|
+
/** Independent operator-presentation and model-context policy for this result. */
|
|
75
|
+
resultDisposition?: ToolResultDisposition;
|
|
70
76
|
/** Coarse cost/latency bucket for dashboard diagnostics. */
|
|
71
77
|
costLatency: ToolCostLatencyClass;
|
|
72
78
|
/**
|
|
@@ -76,6 +82,11 @@ export interface ToolMetadata {
|
|
|
76
82
|
* need none; the schema description covers them.
|
|
77
83
|
*/
|
|
78
84
|
promptHint?: string;
|
|
85
|
+
/**
|
|
86
|
+
* How transcript surfaces present this tool's block under `/output default`.
|
|
87
|
+
* Optional: tools that declare nothing fold like every other tool.
|
|
88
|
+
*/
|
|
89
|
+
presentation?: ToolPresentationPolicy;
|
|
79
90
|
}
|
|
80
91
|
|
|
81
92
|
export interface ToolSpec {
|
|
@@ -108,6 +119,15 @@ export interface ToolSpec {
|
|
|
108
119
|
* the same function at the top of `run`.
|
|
109
120
|
*/
|
|
110
121
|
prepareArguments?(args: Record<string, unknown>): Record<string, unknown>;
|
|
122
|
+
/**
|
|
123
|
+
* Resolve an argument-sensitive canonical disposition after normalization
|
|
124
|
+
* and before execution. The registry applies the result exactly once after
|
|
125
|
+
* middleware has annotated the terminal result.
|
|
126
|
+
*/
|
|
127
|
+
resolveResultDisposition?(
|
|
128
|
+
args: Record<string, unknown>,
|
|
129
|
+
declared: ToolResultDisposition | undefined,
|
|
130
|
+
): ToolResultDisposition | undefined;
|
|
111
131
|
/**
|
|
112
132
|
* Synchronous admission planner. Unlike `prepareArguments`, this runs before
|
|
113
133
|
* safety/autonomy mapping so approval-sensitive tools can attach the exact
|
|
@@ -130,6 +150,8 @@ export type ToolResult =
|
|
|
130
150
|
kind: "ok";
|
|
131
151
|
output: string;
|
|
132
152
|
details?: ToolResultDetails;
|
|
153
|
+
/** Internal registry projection consumed only by the agent-tool adapter. */
|
|
154
|
+
modelContext?: string;
|
|
133
155
|
/**
|
|
134
156
|
* Early-termination hint propagated to pi-agent-core's
|
|
135
157
|
* `AgentToolResult.terminate`. When every finalized tool result in
|
|
@@ -139,7 +161,7 @@ export type ToolResult =
|
|
|
139
161
|
*/
|
|
140
162
|
terminate?: boolean;
|
|
141
163
|
}
|
|
142
|
-
| { kind: "error"; message: string; details?: ToolResultDetails };
|
|
164
|
+
| { kind: "error"; message: string; details?: ToolResultDetails; modelContext?: string };
|
|
143
165
|
|
|
144
166
|
export interface RegistryDeps {
|
|
145
167
|
safety: SafetyContract;
|
|
@@ -378,6 +400,7 @@ export function createRegistry(deps: RegistryDeps): ToolRegistry {
|
|
|
378
400
|
decision: SafetyDecision,
|
|
379
401
|
options?: ToolInvokeOptions,
|
|
380
402
|
): Promise<RegistryVerdict> => {
|
|
403
|
+
let resultDisposition = spec.metadata?.resultDisposition;
|
|
381
404
|
try {
|
|
382
405
|
// The hook layer is the only control stage past safety admission. Guards
|
|
383
406
|
// (loop, protected artifacts, dispatch dedup) are before_tool
|
|
@@ -396,17 +419,18 @@ export function createRegistry(deps: RegistryDeps): ToolRegistry {
|
|
|
396
419
|
}
|
|
397
420
|
try {
|
|
398
421
|
const preparedArgs = prepareToolArgs(spec, call.args ?? {});
|
|
399
|
-
|
|
422
|
+
resultDisposition = resolveToolResultDisposition(spec, preparedArgs);
|
|
423
|
+
const result = await spec.run(preparedArgs, options);
|
|
400
424
|
const afterEffects = runToolHook("after_tool", spec, call, decision, options, result);
|
|
401
|
-
const finalResult = shapeToolResult(spec, applyToolResultEffects(result, afterEffects), options);
|
|
425
|
+
const finalResult = shapeToolResult(spec, applyToolResultEffects(result, afterEffects), options, resultDisposition);
|
|
402
426
|
return { kind: "ok", result: finalResult, decision };
|
|
403
427
|
} catch (err) {
|
|
404
428
|
const message = err instanceof Error ? err.message : String(err);
|
|
405
|
-
const result =
|
|
429
|
+
const result: ToolResult = { kind: "error", message };
|
|
406
430
|
const afterEffects = runToolHook("after_tool", spec, call, decision, options, result);
|
|
407
431
|
return {
|
|
408
432
|
kind: "ok",
|
|
409
|
-
result: shapeToolResult(spec, applyToolResultEffects(result, afterEffects), options),
|
|
433
|
+
result: shapeToolResult(spec, applyToolResultEffects(result, afterEffects), options, resultDisposition),
|
|
410
434
|
decision,
|
|
411
435
|
};
|
|
412
436
|
}
|
|
@@ -851,6 +875,38 @@ function prepareToolArgs(spec: ToolSpec, args: Record<string, unknown>): Record<
|
|
|
851
875
|
}
|
|
852
876
|
}
|
|
853
877
|
|
|
878
|
+
/**
|
|
879
|
+
* Resolve the argument-sensitive disposition, failing closed. A throwing
|
|
880
|
+
* resolver cannot fall back to the declared disposition: the caller may have
|
|
881
|
+
* asked for a narrower context than the tool declares, and silently restoring
|
|
882
|
+
* the declared mode would widen model context while recording the declared mode
|
|
883
|
+
* as the requested one. The failure keeps the declared presentation and applies
|
|
884
|
+
* the narrowest context instead, so the applied and recorded modes stay honest.
|
|
885
|
+
*/
|
|
886
|
+
function resolveToolResultDisposition(
|
|
887
|
+
spec: ToolSpec,
|
|
888
|
+
args: Record<string, unknown>,
|
|
889
|
+
): ToolResultDisposition | undefined {
|
|
890
|
+
const declared = spec.metadata?.resultDisposition;
|
|
891
|
+
if (!spec.resolveResultDisposition) return declared;
|
|
892
|
+
try {
|
|
893
|
+
return spec.resolveResultDisposition(args, declared);
|
|
894
|
+
} catch (error) {
|
|
895
|
+
const maxBytes =
|
|
896
|
+
declared !== undefined && declared.context.maxBytes !== undefined
|
|
897
|
+
? declared.context.maxBytes
|
|
898
|
+
: DEFAULT_TOOL_RESULT_MAX_BYTES;
|
|
899
|
+
return {
|
|
900
|
+
presentation: declared?.presentation ?? { foldDefault: "folded", showDiffWhenFolded: false, failureExcerpt: true },
|
|
901
|
+
context: { mode: "metadata-only", maxBytes },
|
|
902
|
+
// The narrowing is recorded with the result, so a resolver bug shows up
|
|
903
|
+
// on the transcript row and in the model's header instead of reading as
|
|
904
|
+
// a deliberate metadata-only request.
|
|
905
|
+
fallback: { reason: "resolver-error", message: error instanceof Error ? error.message : String(error) },
|
|
906
|
+
};
|
|
907
|
+
}
|
|
908
|
+
}
|
|
909
|
+
|
|
854
910
|
function disposeAdmissionArgs(spec: ToolSpec | undefined, args: Record<string, unknown>): void {
|
|
855
911
|
try {
|
|
856
912
|
spec?.disposeAdmissionArguments?.(args);
|
|
@@ -1155,7 +1211,9 @@ function applyToolResultEffects(result: ToolResult, effects: ReadonlyArray<Middl
|
|
|
1155
1211
|
if (result.terminate === true) annotated.terminate = true;
|
|
1156
1212
|
return annotated;
|
|
1157
1213
|
}
|
|
1158
|
-
|
|
1214
|
+
const annotated: ToolResult = { kind: "error", message: `${result.message}${suffix}` };
|
|
1215
|
+
if (result.details !== undefined) annotated.details = result.details;
|
|
1216
|
+
return annotated;
|
|
1159
1217
|
}
|
|
1160
1218
|
|
|
1161
1219
|
function annotationMessages(effects: ReadonlyArray<MiddlewareEffect>): string[] {
|