@vellumai/assistant 0.11.4 → 0.11.5-staging.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/AGENTS.md +2 -2
- package/ARCHITECTURE.md +29 -4
- package/README.md +1 -1
- package/docs/architecture/memory.md +17 -5
- package/docs/architecture/security.md +96 -152
- package/docs/guardian-request-flow.md +13 -2
- package/node_modules/@vellumai/ces-client/node_modules/@vellumai/service-contracts/package.json +1 -0
- package/node_modules/@vellumai/ces-client/node_modules/@vellumai/service-contracts/src/__tests__/url-normalization.test.ts +135 -0
- package/node_modules/@vellumai/ces-client/node_modules/@vellumai/service-contracts/src/index.ts +1 -0
- package/node_modules/@vellumai/ces-client/node_modules/@vellumai/service-contracts/src/url-normalization.ts +107 -0
- package/node_modules/@vellumai/gateway-client/node_modules/@vellumai/service-contracts/package.json +1 -0
- package/node_modules/@vellumai/gateway-client/node_modules/@vellumai/service-contracts/src/__tests__/url-normalization.test.ts +135 -0
- package/node_modules/@vellumai/gateway-client/node_modules/@vellumai/service-contracts/src/index.ts +1 -0
- package/node_modules/@vellumai/gateway-client/node_modules/@vellumai/service-contracts/src/url-normalization.ts +107 -0
- package/node_modules/@vellumai/gateway-client/src/gateway-ipc-contracts.ts +166 -1
- package/node_modules/@vellumai/gateway-client/src/http-delivery.ts +7 -14
- package/node_modules/@vellumai/gateway-client/src/index.ts +2 -0
- package/node_modules/@vellumai/gateway-client/src/outbound-contract.ts +29 -31
- package/node_modules/@vellumai/service-contracts/package.json +1 -0
- package/node_modules/@vellumai/service-contracts/src/__tests__/url-normalization.test.ts +135 -0
- package/node_modules/@vellumai/service-contracts/src/index.ts +1 -0
- package/node_modules/@vellumai/service-contracts/src/url-normalization.ts +107 -0
- package/openapi.yaml +193 -3
- package/package.json +1 -1
- package/src/__tests__/anthropic-provider.test.ts +10 -3
- package/src/__tests__/approval-interception-trust-gates.test.ts +40 -0
- package/src/__tests__/background-workers-disk-pressure.test.ts +1 -1
- package/src/__tests__/channel-approval-routes.test.ts +4 -0
- package/src/__tests__/channel-inbound-disk-pressure.test.ts +5 -6
- package/src/__tests__/channel-reply-delivery.test.ts +26 -13
- package/src/__tests__/checker.test.ts +20 -314
- package/src/__tests__/cli-memory-v2-reembed-skills.test.ts +6 -2
- package/src/__tests__/client-os-metadata-persistence.test.ts +16 -0
- package/src/__tests__/compactor-retained-image-validation.test.ts +142 -0
- package/src/__tests__/config-loader-backfill.test.ts +19 -10
- package/src/__tests__/container-cpu-sampler.test.ts +153 -0
- package/src/__tests__/context-search-memory-v2-source.test.ts +0 -1
- package/src/__tests__/conversation-attachments.test.ts +0 -1
- package/src/__tests__/conversation-error.test.ts +17 -0
- package/src/__tests__/conversation-lifecycle.test.ts +20 -26
- package/src/__tests__/conversation-pairing.test.ts +186 -0
- package/src/__tests__/conversation-runtime-assembly.test.ts +21 -2
- package/src/__tests__/credential-broker-server-use.test.ts +24 -5
- package/src/__tests__/credential-routes.test.ts +22 -3
- package/src/__tests__/credential-security-invariants.test.ts +2 -1
- package/src/__tests__/daemon-credential-client.test.ts +88 -0
- package/src/__tests__/default-plugin-names-guard.test.ts +12 -1
- package/src/__tests__/dm-backfill.test.ts +63 -0
- package/src/__tests__/dynamic-skill-background-guard.test.ts +0 -2
- package/src/__tests__/edit-propagation.test.ts +107 -4
- package/src/__tests__/events-client-registration.test.ts +28 -0
- package/src/__tests__/file-write-tool.test.ts +4 -2
- package/src/__tests__/guardian-prompt-notice-privacy.test.ts +133 -0
- package/src/__tests__/guardian-verify-setup-skill-regression.test.ts +143 -97
- package/src/__tests__/helpers/gateway-classify-mock.ts +27 -6
- package/src/__tests__/host-proxy-interface.test.ts +11 -1
- package/src/__tests__/host-shell-tool.test.ts +23 -4
- package/src/__tests__/image-conversion.test.ts +143 -1
- package/src/__tests__/inline-skill-load-permissions.test.ts +47 -36
- package/src/__tests__/input-repairs.test.ts +207 -0
- package/src/__tests__/live-workspace-guard.test.ts +75 -0
- package/src/__tests__/managed-profile-guard.test.ts +4 -2
- package/src/__tests__/mcp-abort-signal.test.ts +1 -1
- package/src/__tests__/mcp-client-auth.test.ts +1 -1
- package/src/__tests__/mcp-tool-annotations-risk.test.ts +1 -1
- package/src/__tests__/media-resolve-image-validation.test.ts +310 -0
- package/src/__tests__/mtime-cache.test.ts +2 -0
- package/src/__tests__/notification-vellum-adapter.test.ts +124 -1
- package/src/__tests__/openai-provider.test.ts +22 -0
- package/src/__tests__/personal-memory-auth-bypass.test.ts +22 -6
- package/src/__tests__/platform-bash-auto-approve.test.ts +0 -4
- package/src/__tests__/platform.test.ts +16 -1
- package/src/__tests__/plugin-api-resolve-credential.test.ts +22 -13
- package/src/__tests__/plugin-api-shim.test.ts +5 -0
- package/src/__tests__/plugin-api-store-credential.test.ts +268 -0
- package/src/__tests__/plugin-config-data-migration.test.ts +8 -1
- package/src/__tests__/plugin-disabled-state.test.ts +2 -0
- package/src/__tests__/plugin-effective-enabled-set.test.ts +55 -3
- package/src/__tests__/plugin-execution-context.test.ts +73 -0
- package/src/__tests__/plugin-import-boundary-guard.test.ts +0 -1
- package/src/__tests__/plugin-import-boundary-reverse-guard.test.ts +0 -1
- package/src/__tests__/reaction-persistence.test.ts +200 -7
- package/src/__tests__/require-fresh-approval.test.ts +0 -4
- package/src/__tests__/resource-pressure-guard.test.ts +428 -0
- package/src/__tests__/resource-pressure-routes.test.ts +113 -0
- package/src/__tests__/risk-classification-boundary-guard.test.ts +115 -0
- package/src/__tests__/run-conversation-turn-persistence.test.ts +27 -1
- package/src/__tests__/secret-routes-platform-proxy.test.ts +7 -0
- package/src/__tests__/skill-tool-factory.test.ts +161 -0
- package/src/__tests__/tool-execution-pipeline.benchmark.test.ts +1 -1
- package/src/__tests__/tool-executor-lifecycle-events.test.ts +132 -5
- package/src/__tests__/tool-executor.test.ts +63 -38
- package/src/__tests__/tool-policy.test.ts +97 -1
- package/src/__tests__/trusted-contact-inline-approval-integration.test.ts +7 -4
- package/src/__tests__/user-plugin-loader.test.ts +2 -0
- package/src/__tests__/validate-input.test.ts +243 -7
- package/src/__tests__/verification-control-plane-policy.test.ts +0 -2
- package/src/__tests__/voice-scoped-grant-consumer.test.ts +2 -0
- package/src/__tests__/voice-session-bridge.test.ts +42 -0
- package/src/acp/__tests__/acp-claude-oauth.test.ts +97 -29
- package/src/acp/__tests__/prepare-agent-env.test.ts +376 -153
- package/src/acp/acp-claude-oauth.ts +38 -21
- package/src/acp/prepare-agent-env.ts +146 -55
- package/src/agent/loop-tool-dedup.test.ts +161 -0
- package/src/agent/loop.ts +68 -22
- package/src/api/constants/app-tools.ts +34 -0
- package/src/api/events/resource-pressure-status-changed.ts +53 -0
- package/src/api/index.ts +19 -0
- package/src/api/responses/resource-pressure-status.ts +25 -0
- package/src/approvals/guardian-channel-delivery.ts +70 -6
- package/src/approvals/guardian-request-resolvers.ts +32 -66
- package/src/calls/__tests__/voice-session-bridge.test.ts +114 -0
- package/src/calls/voice-session-bridge.ts +60 -6
- package/src/channels/__tests__/message-audience.test.ts +49 -0
- package/src/channels/__tests__/types.test.ts +17 -3
- package/src/channels/message-audience.ts +40 -0
- package/src/channels/types.ts +16 -12
- package/src/cli/__tests__/catalog-search-help.test.ts +50 -0
- package/src/cli/commands/__tests__/channel-verification-sessions.test.ts +31 -0
- package/src/cli/commands/__tests__/conversations-search.test.ts +289 -0
- package/src/cli/commands/__tests__/conversations-slack.test.ts +1 -0
- package/src/cli/commands/__tests__/inference-providers.test.ts +68 -0
- package/src/cli/commands/__tests__/keys.test.ts +89 -8
- package/src/cli/commands/__tests__/plugins.test.ts +57 -1
- package/src/cli/commands/channel-verification-sessions.help.ts +11 -9
- package/src/cli/commands/channel-verification-sessions.ts +11 -21
- package/src/cli/commands/conversations.help.ts +34 -0
- package/src/cli/commands/conversations.ts +125 -0
- package/src/cli/commands/inference-providers.ts +9 -4
- package/src/cli/commands/inference.help.ts +9 -4
- package/src/cli/commands/keys.help.ts +13 -1
- package/src/cli/commands/keys.ts +26 -15
- package/src/cli/commands/memory/__tests__/memory-v2.test.ts +57 -7
- package/src/cli/commands/memory/index.help.ts +41 -27
- package/src/cli/commands/memory/index.ts +2 -0
- package/src/cli/commands/memory/memory-v2.ts +58 -54
- package/src/cli/commands/memory/memory-validate.ts +18 -0
- package/src/cli/commands/monitoring.ts +1 -0
- package/src/cli/commands/plugins.help.ts +28 -31
- package/src/cli/commands/plugins.ts +10 -0
- package/src/cli/commands/skills.help.ts +13 -16
- package/src/cli/commands/trust.ts +3 -14
- package/src/cli/lib/__tests__/install-from-github.test.ts +38 -0
- package/src/cli/lib/__tests__/merge-plugin-tree.test.ts +34 -0
- package/src/cli/lib/__tests__/plugin-surfaces.test.ts +32 -1
- package/src/cli/lib/__tests__/toggle-plugin.test.ts +2 -1
- package/src/cli/lib/__tests__/upgrade-plugin.test.ts +64 -0
- package/src/cli/lib/bundled-marketplace.json +13 -0
- package/src/cli/lib/daemon-credential-client.ts +25 -3
- package/src/cli/lib/install-from-github.ts +35 -24
- package/src/cli/lib/merge-plugin-tree.ts +22 -4
- package/src/cli/lib/plugin-surfaces.ts +27 -0
- package/src/config/__tests__/balanced-model-experiment.test.ts +7 -7
- package/src/config/__tests__/default-profile-catalog.test.ts +3 -3
- package/src/config/default-profile-catalog.ts +5 -6
- package/src/config/feature-flag-registry.json +32 -32
- package/src/config/webhook-routing.ts +8 -0
- package/src/context/compactor.ts +31 -2
- package/src/daemon/__tests__/provider-rejection-log-fields.test.ts +272 -0
- package/src/daemon/conversation-agent-loop-handlers.ts +4 -1
- package/src/daemon/conversation-error.ts +13 -0
- package/src/daemon/conversation-messaging.ts +22 -2
- package/src/daemon/conversation-process.ts +11 -0
- package/src/daemon/conversation-store.ts +10 -0
- package/src/daemon/conversation-surfaces.ts +21 -8
- package/src/daemon/conversation.ts +12 -10
- package/src/daemon/lifecycle.ts +6 -2
- package/src/daemon/message-types/conversations.ts +5 -7
- package/src/daemon/provider-rejection-log-fields.ts +124 -0
- package/src/daemon/resource-pressure-guard-lifecycle.ts +66 -0
- package/src/daemon/resource-pressure-guard.ts +387 -0
- package/src/daemon/shutdown-handlers.ts +2 -0
- package/src/daemon/startup-error.ts +16 -0
- package/src/daemon/trust-context.ts +14 -10
- package/src/daemon/unsendable-image-notice.ts +76 -0
- package/src/hooks/registry.ts +65 -51
- package/src/ipc/__tests__/socket-path.test.ts +15 -7
- package/src/ipc/gateway-client.test.ts +1 -1
- package/src/ipc/gateway-client.ts +14 -21
- package/src/ipc/socket-cleanup.ts +2 -11
- package/src/mcp/__tests__/mcp-auth-orchestrator.test.ts +1 -1
- package/src/messaging/provider-message-metadata.ts +128 -0
- package/src/messaging/providers/__tests__/transport-dispatch.test.ts +116 -68
- package/src/messaging/providers/channel-transport.ts +90 -13
- package/src/messaging/providers/discord/send.ts +16 -0
- package/src/messaging/providers/discord/transport.ts +11 -1
- package/src/messaging/providers/index.ts +109 -16
- package/src/messaging/providers/slack/message-metadata.ts +45 -0
- package/src/messaging/providers/slack/render-transcript.ts +1 -4
- package/src/messaging/providers/slack/send.test.ts +6 -9
- package/src/messaging/providers/slack/send.ts +40 -28
- package/src/messaging/providers/slack/transport.ts +22 -26
- package/src/messaging/providers/telegram-bot/send.test.ts +2 -8
- package/src/messaging/providers/telegram-bot/transport.ts +3 -6
- package/src/messaging/read-provider-metadata.test.ts +157 -0
- package/src/messaging/read-provider-metadata.ts +51 -0
- package/src/notifications/__tests__/assistant-reply-producer.test.ts +94 -0
- package/src/notifications/__tests__/edit-notification.test.ts +174 -2
- package/src/notifications/__tests__/home-feed-side-effect.test.ts +536 -4
- package/src/notifications/adapters/macos.ts +55 -0
- package/src/notifications/adapters/slack.ts +10 -5
- package/src/notifications/assistant-reply-producer.ts +37 -4
- package/src/notifications/conversation-pairing.ts +111 -21
- package/src/notifications/edit-notification.ts +38 -8
- package/src/notifications/emit-signal.ts +12 -14
- package/src/notifications/home-feed-side-effect.ts +218 -15
- package/src/notifications/types.ts +5 -0
- package/src/permissions/AGENTS.md +16 -0
- package/src/permissions/checker.test.ts +75 -122
- package/src/permissions/checker.ts +44 -560
- package/src/permissions/confirmation-guardian-request.test.ts +28 -3
- package/src/permissions/confirmation-guardian-request.ts +5 -1
- package/src/persistence/conversation-crud.ts +23 -20
- package/src/persistence/conversation-queries.ts +14 -1
- package/src/persistence/conversation-types.ts +22 -0
- package/src/persistence/delivery-crud.ts +122 -15
- package/src/plugin-api/__tests__/import-graph-partial-mock.test.ts +67 -0
- package/src/plugin-api/conversation-turn.ts +27 -0
- package/src/plugin-api/credential-scope.test.ts +15 -0
- package/src/plugin-api/credential-scope.ts +14 -0
- package/src/plugin-api/index.ts +19 -1
- package/src/plugin-api/resolve-credential.ts +15 -10
- package/src/plugin-api/store-credential.ts +148 -0
- package/src/plugin-api/system-card.ts +36 -0
- package/src/plugin-api/vision-support.test.ts +82 -0
- package/src/plugin-api/vision-support.ts +14 -2
- package/src/plugins/defaults/image-fallback/__tests__/image-fallback.test.ts +361 -110
- package/src/plugins/defaults/image-fallback/__tests__/vision-recovery.test.ts +4 -0
- package/src/plugins/defaults/image-fallback/hooks/user-prompt-submit.ts +58 -0
- package/src/plugins/defaults/image-fallback/src/caption-blocks.ts +64 -23
- package/src/plugins/defaults/image-recovery/detect.ts +24 -1
- package/src/plugins/defaults/main.ts +15 -0
- package/src/plugins/defaults/memory/AGENTS.md +11 -3
- package/src/plugins/defaults/memory/__tests__/memory-tier-boundary-guard.test.ts +0 -1
- package/src/plugins/defaults/memory/graph-topology/pending-buffer.ts +9 -12
- package/src/plugins/defaults/memory/memory-retrospective-prompt.ts +1 -1
- package/src/plugins/defaults/memory/src/memory-v2-routes.ts +12 -10
- package/src/plugins/defaults/memory/substrate/__tests__/consolidation-job.test.ts +114 -0
- package/src/plugins/defaults/memory/substrate/__tests__/consolidation-prompt-flag-gating-guard.test.ts +8 -0
- package/src/plugins/defaults/memory/substrate/__tests__/edge-index.test.ts +0 -30
- package/src/plugins/defaults/memory/substrate/__tests__/ingest.test.ts +73 -1
- package/src/plugins/defaults/memory/substrate/__tests__/page-index.test.ts +37 -0
- package/src/plugins/defaults/memory/substrate/__tests__/page-links.test.ts +208 -0
- package/src/plugins/defaults/memory/substrate/__tests__/prompts-consolidation.test.ts +158 -1
- package/src/plugins/defaults/memory/substrate/__tests__/static-context.test.ts +1 -61
- package/src/plugins/defaults/memory/substrate/consolidation-job.ts +68 -6
- package/src/plugins/defaults/memory/substrate/edge-index.ts +0 -31
- package/src/plugins/defaults/memory/substrate/ingest.ts +59 -0
- package/src/plugins/defaults/memory/substrate/page-index.ts +35 -8
- package/src/plugins/defaults/memory/substrate/page-links.ts +133 -0
- package/src/plugins/defaults/memory/substrate/page-store.ts +10 -0
- package/src/plugins/defaults/memory/substrate/prompts/consolidation.ts +124 -34
- package/src/plugins/defaults/memory/substrate/static-context.ts +0 -29
- package/src/plugins/defaults/memory/v3/__tests__/shadow-plugin.test.ts +3 -1
- package/src/plugins/defaults/memory/v3/card.ts +9 -9
- package/src/plugins/defaults/memory/v3/core-set.test.ts +7 -0
- package/src/plugins/defaults/memory/v3/core-set.ts +5 -1
- package/src/plugins/defaults/memory/v3/edge.ts +13 -57
- package/src/plugins/mtime-cache.ts +1 -1
- package/src/plugins/pipeline.ts +14 -7
- package/src/plugins/plugin-execution-context.ts +35 -11
- package/src/plugins/plugin-tree-walk.ts +63 -4
- package/src/providers/anthropic/__tests__/pause-turn-continuation.test.ts +170 -0
- package/src/providers/anthropic/client.ts +332 -241
- package/src/providers/connection-resolution.ts +2 -1
- package/src/providers/content-blocks.ts +22 -0
- package/src/providers/gemini/client.ts +5 -2
- package/src/providers/inference/__tests__/adapter-factory-litellm.test.ts +47 -1
- package/src/providers/inference/__tests__/adapter-factory-ollama.test.ts +83 -0
- package/src/providers/inference/__tests__/adapter-factory-openai-compatible.test.ts +98 -1
- package/src/providers/inference/__tests__/base-url-route-validation.test.ts +38 -0
- package/src/providers/inference/__tests__/base-url-security.test.ts +10 -0
- package/src/providers/inference/__tests__/connections-ollama.test.ts +90 -0
- package/src/providers/inference/__tests__/missing-credential-guard.test.ts +117 -0
- package/src/providers/inference/adapter-factory.ts +33 -2
- package/src/providers/inference/auth.ts +13 -2
- package/src/providers/inference/credential-usage.ts +37 -0
- package/src/providers/inference/missing-credential-guard.ts +110 -0
- package/src/providers/inference/resolve-auth.ts +7 -7
- package/src/providers/media-resolve.ts +176 -15
- package/src/providers/openai/__tests__/chat-completions-provider-reasoning.test.ts +576 -46
- package/src/providers/openai/__tests__/tool-choice-mapping.test.ts +125 -2
- package/src/providers/openai/chat-completions-provider.ts +322 -42
- package/src/routes/route-host-protocol.ts +7 -0
- package/src/routes/worker.ts +31 -10
- package/src/runtime/AGENTS.md +35 -0
- package/src/runtime/__tests__/runtime-http-port-collision.test.ts +131 -0
- package/src/runtime/__tests__/web-presence.test.ts +234 -0
- package/src/runtime/assistant-event-hub.ts +38 -0
- package/src/runtime/channel-reply-delivery.ts +54 -31
- package/src/runtime/channel-retry-sweep.ts +6 -4
- package/src/runtime/effective-capabilities.test.ts +19 -0
- package/src/runtime/effective-capabilities.ts +25 -0
- package/src/runtime/guardian-reply-router.ts +6 -2
- package/src/runtime/http-errors.ts +1 -0
- package/src/runtime/http-server.ts +20 -5
- package/src/runtime/routes/__tests__/client-routes.test.ts +150 -0
- package/src/runtime/routes/__tests__/conversation-list-routes.test.ts +133 -0
- package/src/runtime/routes/__tests__/credential-delete-in-use.test.ts +155 -0
- package/src/runtime/routes/__tests__/inference-provider-connection-routes.test.ts +52 -1
- package/src/runtime/routes/__tests__/monitoring-routes.test.ts +1 -0
- package/src/runtime/routes/__tests__/pending-interactions-route.test.ts +175 -0
- package/src/runtime/routes/__tests__/plugins-routes.test.ts +88 -35
- package/src/runtime/routes/__tests__/surface-action-routes.test.ts +55 -6
- package/src/runtime/routes/__tests__/system-card-turn-termination.test.ts +107 -0
- package/src/runtime/routes/__tests__/user-route-dispatcher-host.test.ts +52 -1
- package/src/runtime/routes/__tests__/user-route-dispatcher.test.ts +75 -4
- package/src/runtime/routes/approval-routes.ts +37 -4
- package/src/runtime/routes/approval-strategies/guardian-text-engine-strategy.ts +16 -22
- package/src/runtime/routes/canned-message-complete.ts +68 -29
- package/src/runtime/routes/client-routes.ts +113 -0
- package/src/runtime/routes/conversation-list-routes.ts +22 -7
- package/src/runtime/routes/conversation-management-routes.ts +15 -9
- package/src/runtime/routes/conversation-routes.ts +36 -22
- package/src/runtime/routes/credential-in-use.ts +85 -0
- package/src/runtime/routes/credential-routes.ts +55 -85
- package/src/runtime/routes/guardian-approval-interception.ts +22 -22
- package/src/runtime/routes/guardian-approval-reply-helpers.ts +14 -12
- package/src/runtime/routes/identity-routes.ts +5 -121
- package/src/runtime/routes/inbound-message-handler.ts +71 -27
- package/src/runtime/routes/inbound-stages/acl-enforcement.ts +16 -17
- package/src/runtime/routes/inbound-stages/background-dispatch.test.ts +77 -50
- package/src/runtime/routes/inbound-stages/background-dispatch.ts +61 -97
- package/src/runtime/routes/inbound-stages/edit-intercept.ts +101 -77
- package/src/runtime/routes/inbound-stages/guardian-reply-intercept.ts +17 -8
- package/src/runtime/routes/inbound-stages/reaction-intercept.test.ts +91 -7
- package/src/runtime/routes/inbound-stages/reaction-intercept.ts +72 -70
- package/src/runtime/routes/index.ts +2 -0
- package/src/runtime/routes/inference-provider-connection-routes.ts +9 -7
- package/src/runtime/routes/monitoring-routes.ts +9 -1
- package/src/runtime/routes/plugins-routes.ts +17 -5
- package/src/runtime/routes/resource-pressure-routes.ts +22 -0
- package/src/runtime/routes/secret-routes.ts +39 -8
- package/src/runtime/routes/settings-routes.ts +6 -6
- package/src/runtime/routes/surface-action-routes.ts +20 -19
- package/src/runtime/routes/user-route-dispatcher.ts +31 -2
- package/src/runtime/routes/user-route-resolution.ts +18 -1
- package/src/runtime/slack-reply-session.test.ts +23 -13
- package/src/runtime/slack-reply-session.ts +38 -55
- package/src/runtime/web-presence.ts +90 -0
- package/src/skills/validate-input.ts +177 -40
- package/src/subagent/__tests__/consult-context-gating.test.ts +6 -10
- package/src/subagent/consult-context.ts +7 -23
- package/src/tools/credentials/broker.ts +24 -79
- package/src/tools/credentials/ref-parse.ts +35 -0
- package/src/tools/credentials/resolve.ts +4 -10
- package/src/tools/credentials/store.ts +168 -0
- package/src/tools/credentials/tool-policy.ts +67 -0
- package/src/tools/executor.ts +53 -15
- package/src/tools/network/__tests__/web-search.test.ts +41 -1
- package/src/tools/network/url-safety.ts +9 -16
- package/src/tools/network/web-search.ts +21 -3
- package/src/tools/permission-checker.ts +43 -52
- package/src/tools/schema-transforms.ts +40 -5
- package/src/tools/shared/input-repairs.ts +160 -0
- package/src/tools/skills/skill-tool-factory.ts +18 -4
- package/src/tools/subagent/spawn.ts +0 -1
- package/src/tools/tool-approval-handler.ts +16 -10
- package/src/tools/tool-types.ts +26 -13
- package/src/tools/types.ts +6 -6
- package/src/util/__tests__/cgroup-memory.test.ts +3 -0
- package/src/util/cgroup-memory.ts +3 -0
- package/src/util/container-cpu-sampler.ts +250 -0
- package/src/util/image-conversion.ts +178 -14
- package/src/util/platform.ts +93 -10
- package/src/permissions/ipc-risk-types.ts +0 -143
- package/src/permissions/risk-types.ts +0 -76
|
@@ -17,16 +17,20 @@ import {
|
|
|
17
17
|
getSecureKeyAsync,
|
|
18
18
|
setSecureKeyAsync,
|
|
19
19
|
} from "../security/secure-keys.js";
|
|
20
|
+
import { getLogger } from "../util/logger.js";
|
|
20
21
|
import {
|
|
21
22
|
ACP_OAUTH_TOKEN_FIELD,
|
|
22
23
|
ACP_SERVICE,
|
|
23
24
|
classifyAnthropicToken,
|
|
24
25
|
} from "./acp-credentials.js";
|
|
25
26
|
import {
|
|
26
|
-
|
|
27
|
-
|
|
27
|
+
ACP_CLAUDE_OAUTH_USAGE_DESCRIPTION,
|
|
28
|
+
acpSpawnCredentialDenialReason,
|
|
29
|
+
repairAcpSpawnPolicy,
|
|
28
30
|
} from "./prepare-agent-env.js";
|
|
29
31
|
|
|
32
|
+
const log = getLogger("acp:claude-oauth");
|
|
33
|
+
|
|
30
34
|
/**
|
|
31
35
|
* Verified Claude Code public OAuth client. PKCE-only (no client secret);
|
|
32
36
|
* the single `user:inference` scope is what the ACP adapter's
|
|
@@ -109,8 +113,9 @@ export function parseManualClaudeCode(input: string): {
|
|
|
109
113
|
|
|
110
114
|
/**
|
|
111
115
|
* Store a captured Claude OAuth token in the `acp/claude_oauth_token` vault
|
|
112
|
-
* field and provision the
|
|
113
|
-
*
|
|
116
|
+
* field and provision the policy the broker applies at spawn time: grant the
|
|
117
|
+
* `acp_spawn` read and lift any domain restriction. Throws when the backing
|
|
118
|
+
* store rejects the write.
|
|
114
119
|
*/
|
|
115
120
|
export async function storeAcpClaudeToken(token: string): Promise<void> {
|
|
116
121
|
const stored = await setSecureKeyAsync(
|
|
@@ -120,13 +125,13 @@ export async function storeAcpClaudeToken(token: string): Promise<void> {
|
|
|
120
125
|
if (!stored) {
|
|
121
126
|
throw new Error("Failed to store Claude OAuth token in secure storage.");
|
|
122
127
|
}
|
|
123
|
-
//
|
|
124
|
-
//
|
|
125
|
-
//
|
|
126
|
-
//
|
|
127
|
-
|
|
128
|
+
// Repair rather than merely ensure the policy: an explicit Connect is a
|
|
129
|
+
// deliberate opt-in to ACP, so this widens a credential the broker would
|
|
130
|
+
// otherwise keep denying the spawn read on, which would dead-loop the Connect
|
|
131
|
+
// card on every auto-continue.
|
|
132
|
+
repairAcpSpawnPolicy(
|
|
128
133
|
ACP_OAUTH_TOKEN_FIELD,
|
|
129
|
-
|
|
134
|
+
ACP_CLAUDE_OAUTH_USAGE_DESCRIPTION,
|
|
130
135
|
);
|
|
131
136
|
}
|
|
132
137
|
|
|
@@ -142,20 +147,32 @@ export async function storeAcpClaudeToken(token: string): Promise<void> {
|
|
|
142
147
|
* spawn, so keeping Connect offered (rather than self-dismissing) lets the user
|
|
143
148
|
* repair the bad entry by connecting a real OAuth token.
|
|
144
149
|
*
|
|
145
|
-
* Likewise, a token the
|
|
146
|
-
* `allowedTools` that omits `acp_spawn
|
|
147
|
-
* value but
|
|
148
|
-
* hide the only repair CTA
|
|
149
|
-
*
|
|
150
|
+
* Likewise, a token the spawn's broker read would be denied (an explicit
|
|
151
|
+
* `allowedTools` that omits `acp_spawn`, or a domain-restricted policy) is NOT
|
|
152
|
+
* connected: the vault holds a value but every spawn fails, so self-dismissing
|
|
153
|
+
* the card would hide the only repair CTA. That half of the answer is delegated
|
|
154
|
+
* to `acpSpawnCredentialDenialReason`, which evaluates the exact policy the
|
|
155
|
+
* spawn-time broker read applies, so "connected" means precisely "the spawn
|
|
156
|
+
* would get this token". The token-shape guard stays here instead: the broker
|
|
157
|
+
* knows nothing about Anthropic token formats.
|
|
150
158
|
*/
|
|
151
159
|
export async function hasAcpClaudeToken(): Promise<boolean> {
|
|
152
160
|
const token = await getSecureKeyAsync(
|
|
153
161
|
credentialKey(ACP_SERVICE, ACP_OAUTH_TOKEN_FIELD),
|
|
154
162
|
);
|
|
155
|
-
|
|
156
|
-
|
|
157
|
-
|
|
158
|
-
|
|
159
|
-
|
|
160
|
-
|
|
163
|
+
if (token == null || token.length === 0) {
|
|
164
|
+
return false;
|
|
165
|
+
}
|
|
166
|
+
if (classifyAnthropicToken(token) === "api_key") {
|
|
167
|
+
return false;
|
|
168
|
+
}
|
|
169
|
+
const denialReason = acpSpawnCredentialDenialReason(ACP_OAUTH_TOKEN_FIELD);
|
|
170
|
+
if (denialReason !== undefined) {
|
|
171
|
+
log.debug(
|
|
172
|
+
{ field: ACP_OAUTH_TOKEN_FIELD, reason: denialReason },
|
|
173
|
+
"Connect Claude status: token present but spawn read would be denied",
|
|
174
|
+
);
|
|
175
|
+
return false;
|
|
176
|
+
}
|
|
177
|
+
return true;
|
|
161
178
|
}
|
|
@@ -25,9 +25,11 @@ import { basename } from "node:path";
|
|
|
25
25
|
import { FailedDependencyError } from "../runtime/routes/errors.js";
|
|
26
26
|
import { credentialBroker } from "../tools/credentials/broker.js";
|
|
27
27
|
import {
|
|
28
|
+
type CredentialMetadata,
|
|
28
29
|
getCredentialMetadata,
|
|
29
30
|
upsertCredentialMetadata,
|
|
30
31
|
} from "../tools/credentials/metadata-store.js";
|
|
32
|
+
import { serverUseDenialReason } from "../tools/credentials/tool-policy.js";
|
|
31
33
|
import { getLogger } from "../util/logger.js";
|
|
32
34
|
import {
|
|
33
35
|
ACP_OAUTH_TOKEN_FIELD,
|
|
@@ -40,91 +42,148 @@ const log = getLogger("acp:prepare-agent-env");
|
|
|
40
42
|
|
|
41
43
|
const ACP_SPAWN_TOOL = "acp_spawn";
|
|
42
44
|
|
|
45
|
+
/**
|
|
46
|
+
* `usageDescription` recorded on `acp/claude_oauth_token` when a record is
|
|
47
|
+
* created, shared by the spawn-time ensure and the Connect repair so the two
|
|
48
|
+
* paths can't describe the same credential differently.
|
|
49
|
+
*/
|
|
50
|
+
export const ACP_CLAUDE_OAUTH_USAGE_DESCRIPTION =
|
|
51
|
+
"Claude OAuth token for ACP agent authentication";
|
|
52
|
+
|
|
43
53
|
/**
|
|
44
54
|
* Stable, machine-readable marker carried on the `FailedDependencyError.details`
|
|
45
55
|
* when a `claude-agent-acp` spawn is missing `CLAUDE_CODE_OAUTH_TOKEN`. Threaded
|
|
46
56
|
* through the tool result / error payload as a structured field so clients can
|
|
47
57
|
* offer the inline "Connect Claude Code" flow instead of re-parsing the human
|
|
48
58
|
* message string. Kept in lockstep with the web literal in
|
|
49
|
-
* `clients/web/src/domains/chat/
|
|
59
|
+
* `clients/web/src/domains/chat/utils/acp-connect.ts`.
|
|
50
60
|
*/
|
|
51
61
|
export const ACP_CLAUDE_OAUTH_MISSING_CODE = "acp_claude_oauth_missing";
|
|
52
62
|
|
|
53
63
|
/**
|
|
54
|
-
*
|
|
55
|
-
*
|
|
64
|
+
* The metadata an `acp/<field>` credential has once the `acp_spawn` read policy
|
|
65
|
+
* is ensured, given what is stored now. This is the single definition of that
|
|
66
|
+
* decision: {@link ensureAcpCredentialPolicy} persists the result and
|
|
67
|
+
* {@link acpSpawnCredentialDenialReason} evaluates it without writing.
|
|
68
|
+
*
|
|
69
|
+
* The policy is only repaired for legacy/unmanaged cases:
|
|
70
|
+
*
|
|
71
|
+
* - No metadata at all: a record with `allowedTools: ["acp_spawn"]` and the
|
|
72
|
+
* caller's `usageDescription`.
|
|
73
|
+
* - Metadata with an empty `allowedTools`: default provisioning path (user ran
|
|
74
|
+
* `credentials set` without `--allowed-tools`), so `acp_spawn` is added.
|
|
75
|
+
* - Metadata with a non-empty `allowedTools`: explicit policy set by the
|
|
76
|
+
* user/admin, returned as the very same object so callers can tell by
|
|
77
|
+
* identity that there is nothing to persist. It stands even when `acp_spawn`
|
|
78
|
+
* is absent; the broker denies the read and the caller decides whether that's
|
|
79
|
+
* fatal.
|
|
56
80
|
*
|
|
57
|
-
*
|
|
58
|
-
*
|
|
59
|
-
* path (user ran `credentials set` without `--allowed-tools`), add it.
|
|
60
|
-
* - Metadata exists with a non-empty `allowedTools`: explicit policy set
|
|
61
|
-
* by the user/admin. Respect it even if `acp_spawn` is absent; the
|
|
62
|
-
* broker will deny the read and the caller decides whether that's fatal.
|
|
81
|
+
* Everything else on an existing record, `allowedDomains` included, is
|
|
82
|
+
* preserved.
|
|
63
83
|
*/
|
|
64
|
-
|
|
84
|
+
function projectEnsuredAcpPolicy(
|
|
85
|
+
meta: CredentialMetadata | undefined,
|
|
65
86
|
field: string,
|
|
66
|
-
usageDescription
|
|
67
|
-
):
|
|
68
|
-
const meta = getCredentialMetadata(ACP_SERVICE, field);
|
|
87
|
+
usageDescription?: string,
|
|
88
|
+
): CredentialMetadata {
|
|
69
89
|
if (!meta) {
|
|
70
|
-
|
|
90
|
+
return {
|
|
91
|
+
credentialId: "",
|
|
92
|
+
service: ACP_SERVICE,
|
|
93
|
+
field,
|
|
71
94
|
allowedTools: [ACP_SPAWN_TOOL],
|
|
95
|
+
allowedDomains: [],
|
|
72
96
|
usageDescription,
|
|
73
|
-
|
|
74
|
-
|
|
97
|
+
createdAt: 0,
|
|
98
|
+
updatedAt: 0,
|
|
99
|
+
};
|
|
75
100
|
}
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
upsertCredentialMetadata(ACP_SERVICE, field, {
|
|
79
|
-
allowedTools: [ACP_SPAWN_TOOL],
|
|
80
|
-
});
|
|
101
|
+
if ((meta.allowedTools ?? []).length === 0) {
|
|
102
|
+
return { ...meta, allowedTools: [ACP_SPAWN_TOOL] };
|
|
81
103
|
}
|
|
104
|
+
return meta;
|
|
82
105
|
}
|
|
83
106
|
|
|
84
107
|
/**
|
|
85
|
-
*
|
|
86
|
-
*
|
|
87
|
-
*
|
|
88
|
-
* it), this is for the EXPLICIT Connect flow: a user connecting Claude is a
|
|
89
|
-
* deliberate opt-in to `acp_spawn`, so granting it makes the CTA actually repair
|
|
90
|
-
* a policy-denied credential instead of dead-looping the missing-token card.
|
|
108
|
+
* Bring the stored metadata for `acp/<field>` up to the policy
|
|
109
|
+
* {@link projectEnsuredAcpPolicy} describes, writing only the fields that
|
|
110
|
+
* projection decides and only when it differs from what is stored.
|
|
91
111
|
*/
|
|
92
|
-
export function
|
|
112
|
+
export function ensureAcpCredentialPolicy(
|
|
93
113
|
field: string,
|
|
94
114
|
usageDescription: string,
|
|
95
115
|
): void {
|
|
96
116
|
const meta = getCredentialMetadata(ACP_SERVICE, field);
|
|
97
|
-
|
|
98
|
-
|
|
99
|
-
allowedTools: [ACP_SPAWN_TOOL],
|
|
100
|
-
usageDescription,
|
|
101
|
-
});
|
|
117
|
+
const ensured = projectEnsuredAcpPolicy(meta, field, usageDescription);
|
|
118
|
+
if (ensured === meta) {
|
|
102
119
|
return;
|
|
103
120
|
}
|
|
104
|
-
|
|
105
|
-
|
|
106
|
-
|
|
107
|
-
|
|
108
|
-
});
|
|
109
|
-
}
|
|
121
|
+
upsertCredentialMetadata(ACP_SERVICE, field, {
|
|
122
|
+
allowedTools: ensured.allowedTools,
|
|
123
|
+
usageDescription: ensured.usageDescription,
|
|
124
|
+
});
|
|
110
125
|
}
|
|
111
126
|
|
|
112
127
|
/**
|
|
113
|
-
*
|
|
114
|
-
*
|
|
115
|
-
*
|
|
116
|
-
*
|
|
117
|
-
*
|
|
118
|
-
*
|
|
119
|
-
*
|
|
128
|
+
* Make `acp/<field>` readable by the spawn: union `acp_spawn` into any existing
|
|
129
|
+
* `allowedTools` and drop any domain restriction, in ONE write and only when the
|
|
130
|
+
* stored record fails either half. This is the whole repair the Connect flow
|
|
131
|
+
* performs, so a new dimension of {@link serverUseDenialReason} is repaired in
|
|
132
|
+
* exactly one place.
|
|
133
|
+
*
|
|
134
|
+
* Unlike {@link ensureAcpCredentialPolicy} (which PRESERVES an explicit non-empty
|
|
135
|
+
* policy so a passive spawn can't silently widen it), this is for the EXPLICIT
|
|
136
|
+
* Connect flow: a user connecting Claude is a deliberate opt-in to `acp_spawn`,
|
|
137
|
+
* so granting it makes the CTA actually repair a policy-denied credential instead
|
|
138
|
+
* of dead-looping the missing-token card. Domains are cleared under the same
|
|
139
|
+
* opt-in: this field is OAuth-only and server-use-only, and the broker refuses a
|
|
140
|
+
* domain-restricted credential server-side, so a lingering restriction would keep
|
|
141
|
+
* every spawn failing even after a successful connect.
|
|
120
142
|
*/
|
|
121
|
-
export function
|
|
143
|
+
export function repairAcpSpawnPolicy(
|
|
144
|
+
field: string,
|
|
145
|
+
usageDescription: string,
|
|
146
|
+
): void {
|
|
122
147
|
const meta = getCredentialMetadata(ACP_SERVICE, field);
|
|
123
|
-
|
|
124
|
-
|
|
148
|
+
const tools = meta?.allowedTools ?? [];
|
|
149
|
+
const spawnAllowed = tools.includes(ACP_SPAWN_TOOL);
|
|
150
|
+
const domainUnrestricted = (meta?.allowedDomains ?? []).length === 0;
|
|
151
|
+
if (meta && spawnAllowed && domainUnrestricted) {
|
|
152
|
+
return;
|
|
125
153
|
}
|
|
126
|
-
|
|
127
|
-
|
|
154
|
+
upsertCredentialMetadata(ACP_SERVICE, field, {
|
|
155
|
+
allowedTools: spawnAllowed ? tools : [...tools, ACP_SPAWN_TOOL],
|
|
156
|
+
allowedDomains: [],
|
|
157
|
+
// Only a fresh record takes the description; an existing one keeps its own.
|
|
158
|
+
...(meta ? {} : { usageDescription }),
|
|
159
|
+
});
|
|
160
|
+
}
|
|
161
|
+
|
|
162
|
+
/**
|
|
163
|
+
* Why the `acp_spawn` broker read for `acp/<field>` would be denied, or
|
|
164
|
+
* `undefined` when it would be permitted. Lets a connected-status check avoid
|
|
165
|
+
* reporting "connected" for a token the spawn is policy-denied from reading
|
|
166
|
+
* (which would otherwise hide the repair CTA and trap the user in a
|
|
167
|
+
* missing-token loop).
|
|
168
|
+
*
|
|
169
|
+
* The verdict comes from `serverUseDenialReason`, the single policy source the
|
|
170
|
+
* broker itself consults, so the status check and the spawn read can never
|
|
171
|
+
* disagree. The stored metadata is first run through
|
|
172
|
+
* {@link projectEnsuredAcpPolicy}, the same repair the spawn persists, computed
|
|
173
|
+
* in memory and never written: this runs on a side-effect-free GET route, so it
|
|
174
|
+
* has to predict what the spawn's ensure-then-read sequence would do rather
|
|
175
|
+
* than perform it.
|
|
176
|
+
*/
|
|
177
|
+
export function acpSpawnCredentialDenialReason(
|
|
178
|
+
field: string,
|
|
179
|
+
): string | undefined {
|
|
180
|
+
const meta = getCredentialMetadata(ACP_SERVICE, field);
|
|
181
|
+
return serverUseDenialReason(
|
|
182
|
+
projectEnsuredAcpPolicy(meta, field),
|
|
183
|
+
ACP_SPAWN_TOOL,
|
|
184
|
+
ACP_SERVICE,
|
|
185
|
+
field,
|
|
186
|
+
);
|
|
128
187
|
}
|
|
129
188
|
|
|
130
189
|
/**
|
|
@@ -243,25 +302,57 @@ export async function prepareAgentEnv(
|
|
|
243
302
|
};
|
|
244
303
|
|
|
245
304
|
dropApiKeyOauthToken();
|
|
305
|
+
let missReason: string | undefined;
|
|
246
306
|
if (!env.CLAUDE_CODE_OAUTH_TOKEN) {
|
|
247
|
-
await injectCredential(
|
|
307
|
+
missReason = await injectCredential(
|
|
248
308
|
env,
|
|
249
309
|
ACP_OAUTH_TOKEN_FIELD,
|
|
250
310
|
"CLAUDE_CODE_OAUTH_TOKEN",
|
|
251
|
-
|
|
311
|
+
ACP_CLAUDE_OAUTH_USAGE_DESCRIPTION,
|
|
252
312
|
);
|
|
253
313
|
}
|
|
314
|
+
// Any api-key-shaped value still standing here came from the vault read:
|
|
315
|
+
// the config override was already dropped above, and the read only runs
|
|
316
|
+
// when the override left the var unset.
|
|
317
|
+
const storedValueIsApiKeyShaped =
|
|
318
|
+
env.CLAUDE_CODE_OAUTH_TOKEN !== undefined &&
|
|
319
|
+
classifyAnthropicToken(env.CLAUDE_CODE_OAUTH_TOKEN) === "api_key";
|
|
254
320
|
dropApiKeyOauthToken();
|
|
255
321
|
if (!env.CLAUDE_CODE_OAUTH_TOKEN) {
|
|
322
|
+
// The operator's record of WHY the spawn has no token. `missReason` is
|
|
323
|
+
// the broker's own reason string and the rest are policy verdicts, so no
|
|
324
|
+
// field can carry the credential value.
|
|
325
|
+
const policyDenialReason = acpSpawnCredentialDenialReason(
|
|
326
|
+
ACP_OAUTH_TOKEN_FIELD,
|
|
327
|
+
);
|
|
328
|
+
log.warn(
|
|
329
|
+
{
|
|
330
|
+
field: ACP_OAUTH_TOKEN_FIELD,
|
|
331
|
+
missReason,
|
|
332
|
+
policyBlocked: policyDenialReason !== undefined,
|
|
333
|
+
apiKeyShaped: storedValueIsApiKeyShaped,
|
|
334
|
+
},
|
|
335
|
+
"Claude OAuth token not injected for acp_spawn",
|
|
336
|
+
);
|
|
256
337
|
// Carry the stable marker as structured `details` so the client renders
|
|
257
338
|
// the inline "Connect Claude Code" card. The message itself is the tool
|
|
258
339
|
// result the model reads at the failure moment, so it directs the model
|
|
259
340
|
// AT that card and away from CLI/token-paste workarounds — otherwise the
|
|
260
341
|
// model relays a `claude setup-token` / paste-a-token flow that the card
|
|
261
342
|
// exists to replace. The CLI command stays only as a headless fallback.
|
|
343
|
+
// A policy-blocked read is a different repair story from an absent value,
|
|
344
|
+
// so the opening states which one happened. The guidance after it is
|
|
345
|
+
// shared: the Connect card fixes both.
|
|
346
|
+
const opening = policyDenialReason
|
|
347
|
+
? "claude-agent-acp cannot read the Claude OAuth token: the credential " +
|
|
348
|
+
"policy on acp/claude_oauth_token blocks the acp_spawn read, so " +
|
|
349
|
+
"CLAUDE_CODE_OAUTH_TOKEN is not set for the spawn. Clicking Connect " +
|
|
350
|
+
"signs in again and repairs that policy. "
|
|
351
|
+
: "claude-agent-acp needs a Claude OAuth token (CLAUDE_CODE_OAUTH_TOKEN), " +
|
|
352
|
+
"which is not set. ";
|
|
262
353
|
throw new FailedDependencyError(
|
|
263
|
-
|
|
264
|
-
'
|
|
354
|
+
opening +
|
|
355
|
+
'The app shows the user an inline "Connect Claude ' +
|
|
265
356
|
'Code" card. Reply with ONE short sentence: ask them to click Connect ' +
|
|
266
357
|
"in that card to sign in, and tell them you'll continue automatically " +
|
|
267
358
|
"once they're connected. Do NOT say where the card is — never say " +
|
|
@@ -0,0 +1,161 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Verifies the agent loop coalesces `tool_use` blocks by call id before it
|
|
3
|
+
* dispatches them: a provider that emits the same call twice under one id gets
|
|
4
|
+
* one execution, one `tool_use` block in history, and one correlated
|
|
5
|
+
* `tool_result`. Distinct ids for the same tool name still run independently.
|
|
6
|
+
* Drives the REAL loop, mocking only the provider boundary.
|
|
7
|
+
*/
|
|
8
|
+
import { describe, expect, test } from "bun:test";
|
|
9
|
+
|
|
10
|
+
import { createMockProvider } from "../__tests__/helpers/mock-provider.js";
|
|
11
|
+
import type { ContentBlock, ProviderResponse } from "../providers/types.js";
|
|
12
|
+
import { AgentLoop } from "./loop.js";
|
|
13
|
+
|
|
14
|
+
const endTurn = (text: string): ProviderResponse => ({
|
|
15
|
+
content: [{ type: "text", text }],
|
|
16
|
+
model: "mock-model",
|
|
17
|
+
usage: { inputTokens: 1, outputTokens: 1 },
|
|
18
|
+
stopReason: "end_turn",
|
|
19
|
+
});
|
|
20
|
+
|
|
21
|
+
const toolUseTurn = (
|
|
22
|
+
blocks: Array<{ id: string; name: string }>,
|
|
23
|
+
): ProviderResponse => ({
|
|
24
|
+
content: [
|
|
25
|
+
{ type: "text", text: "working" },
|
|
26
|
+
...blocks.map((b) => ({
|
|
27
|
+
type: "tool_use" as const,
|
|
28
|
+
id: b.id,
|
|
29
|
+
name: b.name,
|
|
30
|
+
input: {},
|
|
31
|
+
})),
|
|
32
|
+
],
|
|
33
|
+
model: "mock-model",
|
|
34
|
+
usage: { inputTokens: 1, outputTokens: 1 },
|
|
35
|
+
stopReason: "tool_use",
|
|
36
|
+
});
|
|
37
|
+
|
|
38
|
+
function blocksOfType<T extends ContentBlock["type"]>(
|
|
39
|
+
history: Array<{ content: ContentBlock[] }>,
|
|
40
|
+
type: T,
|
|
41
|
+
): Array<Extract<ContentBlock, { type: T }>> {
|
|
42
|
+
return history
|
|
43
|
+
.flatMap((m) => m.content)
|
|
44
|
+
.filter((b): b is Extract<ContentBlock, { type: T }> => b.type === type);
|
|
45
|
+
}
|
|
46
|
+
|
|
47
|
+
function buildLoop(
|
|
48
|
+
provider: ReturnType<typeof createMockProvider>["provider"],
|
|
49
|
+
conversationId: string,
|
|
50
|
+
executed: string[],
|
|
51
|
+
) {
|
|
52
|
+
return new AgentLoop({
|
|
53
|
+
provider,
|
|
54
|
+
systemPrompt: "sys",
|
|
55
|
+
conversationId,
|
|
56
|
+
tools: [
|
|
57
|
+
{ name: "read_file", description: "", input_schema: { type: "object" } },
|
|
58
|
+
],
|
|
59
|
+
toolExecutor: async (name) => {
|
|
60
|
+
executed.push(name);
|
|
61
|
+
return { content: `ran ${name}`, isError: false };
|
|
62
|
+
},
|
|
63
|
+
});
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
const baseRun = {
|
|
67
|
+
requestId: "req-dedup",
|
|
68
|
+
onEvent: () => {},
|
|
69
|
+
callSite: "mainAgent" as const,
|
|
70
|
+
trust: { sourceChannel: "vellum" as const, trustClass: "unknown" as const },
|
|
71
|
+
messages: [
|
|
72
|
+
{ role: "user" as const, content: [{ type: "text" as const, text: "go" }] },
|
|
73
|
+
],
|
|
74
|
+
};
|
|
75
|
+
|
|
76
|
+
describe("AgentLoop: duplicate tool_use ids", () => {
|
|
77
|
+
/** A call the provider emits twice under one id runs a single time. */
|
|
78
|
+
test("executes a call id once when the provider emits it twice", async () => {
|
|
79
|
+
// GIVEN a provider turn carrying the same tool_use id twice
|
|
80
|
+
const { provider } = createMockProvider([
|
|
81
|
+
toolUseTurn([
|
|
82
|
+
{ id: "call-dup", name: "read_file" },
|
|
83
|
+
{ id: "call-dup", name: "read_file" },
|
|
84
|
+
]),
|
|
85
|
+
endTurn("done"),
|
|
86
|
+
]);
|
|
87
|
+
|
|
88
|
+
// WHEN the loop runs the turn
|
|
89
|
+
const executed: string[] = [];
|
|
90
|
+
const toolUseEventIds: string[] = [];
|
|
91
|
+
const { history } = await buildLoop(provider, "dedup-1", executed).run({
|
|
92
|
+
...baseRun,
|
|
93
|
+
onEvent: (event) => {
|
|
94
|
+
if (event.type === "tool_use") {
|
|
95
|
+
toolUseEventIds.push(event.id);
|
|
96
|
+
}
|
|
97
|
+
},
|
|
98
|
+
});
|
|
99
|
+
|
|
100
|
+
// THEN the tool runs once and the client sees one tool_use
|
|
101
|
+
expect(executed).toEqual(["read_file"]);
|
|
102
|
+
expect(toolUseEventIds).toEqual(["call-dup"]);
|
|
103
|
+
|
|
104
|
+
// AND history stays well-formed: providers require one tool_result per
|
|
105
|
+
// tool_use id, so the coalesced copy must not survive into history.
|
|
106
|
+
const toolUses = blocksOfType(history, "tool_use");
|
|
107
|
+
expect(toolUses.map((b) => b.id)).toEqual(["call-dup"]);
|
|
108
|
+
const results = blocksOfType(history, "tool_result");
|
|
109
|
+
expect(results.map((b) => b.tool_use_id)).toEqual(["call-dup"]);
|
|
110
|
+
expect(results[0]!.content).toBe("ran read_file");
|
|
111
|
+
expect(results[0]!.is_error).toBe(false);
|
|
112
|
+
});
|
|
113
|
+
|
|
114
|
+
/** Two independent calls of one tool are not collapsed by name. */
|
|
115
|
+
test("runs repeat calls of one tool when their ids differ", async () => {
|
|
116
|
+
// GIVEN a provider turn calling one tool twice under distinct ids
|
|
117
|
+
const { provider } = createMockProvider([
|
|
118
|
+
toolUseTurn([
|
|
119
|
+
{ id: "call-a", name: "read_file" },
|
|
120
|
+
{ id: "call-b", name: "read_file" },
|
|
121
|
+
]),
|
|
122
|
+
endTurn("done"),
|
|
123
|
+
]);
|
|
124
|
+
|
|
125
|
+
// WHEN the loop runs the turn
|
|
126
|
+
const executed: string[] = [];
|
|
127
|
+
const { history } = await buildLoop(provider, "dedup-2", executed).run(
|
|
128
|
+
baseRun,
|
|
129
|
+
);
|
|
130
|
+
|
|
131
|
+
// THEN both calls execute and each gets its own result
|
|
132
|
+
expect(executed).toEqual(["read_file", "read_file"]);
|
|
133
|
+
expect(
|
|
134
|
+
blocksOfType(history, "tool_result").map((b) => b.tool_use_id),
|
|
135
|
+
).toEqual(["call-a", "call-b"]);
|
|
136
|
+
});
|
|
137
|
+
|
|
138
|
+
/** An id-less call still executes, under a generated id. */
|
|
139
|
+
test("assigns a call id when the provider emits a tool_use without one", async () => {
|
|
140
|
+
// GIVEN a provider turn whose tool_use block carries no id
|
|
141
|
+
const { provider } = createMockProvider([
|
|
142
|
+
toolUseTurn([{ id: "", name: "read_file" }]),
|
|
143
|
+
endTurn("done"),
|
|
144
|
+
]);
|
|
145
|
+
|
|
146
|
+
// WHEN the loop runs the turn
|
|
147
|
+
const executed: string[] = [];
|
|
148
|
+
const { history } = await buildLoop(provider, "dedup-3", executed).run(
|
|
149
|
+
baseRun,
|
|
150
|
+
);
|
|
151
|
+
|
|
152
|
+
// THEN the call executes under a generated id its result correlates to
|
|
153
|
+
expect(executed).toEqual(["read_file"]);
|
|
154
|
+
const toolUses = blocksOfType(history, "tool_use");
|
|
155
|
+
expect(toolUses).toHaveLength(1);
|
|
156
|
+
expect(toolUses[0]!.id.length).toBeGreaterThan(0);
|
|
157
|
+
expect(
|
|
158
|
+
blocksOfType(history, "tool_result").map((b) => b.tool_use_id),
|
|
159
|
+
).toEqual([toolUses[0]!.id]);
|
|
160
|
+
});
|
|
161
|
+
});
|
package/src/agent/loop.ts
CHANGED
|
@@ -31,6 +31,7 @@ import { defaultCompact } from "../plugins/defaults/compaction/compact.js";
|
|
|
31
31
|
import type { ContextWindowResult } from "../plugins/defaults/compaction/window-manager.js";
|
|
32
32
|
import { runHook } from "../plugins/pipeline.js";
|
|
33
33
|
import type { CompactionCircuitEvent } from "../plugins/types.js";
|
|
34
|
+
import { hasVisibleText } from "../providers/content-blocks.js";
|
|
34
35
|
import { isMaxTokensStopReason } from "../providers/stop-reasons.js";
|
|
35
36
|
import { normalizeThinkingConfigForWire } from "../providers/thinking-config.js";
|
|
36
37
|
import type {
|
|
@@ -528,13 +529,6 @@ function assistantTextOf(content: ReadonlyArray<ContentBlock>): string {
|
|
|
528
529
|
return text;
|
|
529
530
|
}
|
|
530
531
|
|
|
531
|
-
/** Whether `content` carries at least one non-empty `text` block. */
|
|
532
|
-
function hasVisibleText(content: ReadonlyArray<ContentBlock>): boolean {
|
|
533
|
-
return content.some(
|
|
534
|
-
(block) => block.type === "text" && block.text.trim().length > 0,
|
|
535
|
-
);
|
|
536
|
-
}
|
|
537
|
-
|
|
538
532
|
type AgentLoopContextWindowResolver = () => {
|
|
539
533
|
maxInputTokens: number;
|
|
540
534
|
overflowRecovery: { enabled: boolean; safetyMarginRatio: number };
|
|
@@ -704,10 +698,56 @@ export type LoopToolExecutor = (
|
|
|
704
698
|
errorCode?: string;
|
|
705
699
|
}>;
|
|
706
700
|
|
|
701
|
+
type ToolUseBlock = Extract<ContentBlock, { type: "tool_use" }>;
|
|
702
|
+
|
|
703
|
+
interface NormalizedToolUse {
|
|
704
|
+
/** Assistant content with at most one `tool_use` block per call id. */
|
|
705
|
+
content: ContentBlock[];
|
|
706
|
+
/** The `tool_use` blocks in `content`, in order. */
|
|
707
|
+
toolUseBlocks: ToolUseBlock[];
|
|
708
|
+
/** Coalesced copies: a call id and name for each block dropped. */
|
|
709
|
+
duplicates: Array<{ id: string; name: string }>;
|
|
710
|
+
}
|
|
711
|
+
|
|
712
|
+
/**
|
|
713
|
+
* Resolves an assistant reply's `tool_use` blocks into an executable set keyed
|
|
714
|
+
* by call id: a block with no id gets one, and a repeat of an id already in the
|
|
715
|
+
* reply is dropped so the call runs once and its single `tool_result` correlates
|
|
716
|
+
* unambiguously. Providers occasionally emit the same call twice under one id,
|
|
717
|
+
* and both Anthropic and OpenAI require one `tool_result` per `tool_use` id, so
|
|
718
|
+
* the duplicate has no well-formed representation downstream.
|
|
719
|
+
*/
|
|
720
|
+
function normalizeToolUseBlocks(
|
|
721
|
+
content: ReadonlyArray<ContentBlock>,
|
|
722
|
+
): NormalizedToolUse {
|
|
723
|
+
const nextContent: ContentBlock[] = [];
|
|
724
|
+
const toolUseBlocks: ToolUseBlock[] = [];
|
|
725
|
+
const duplicates: Array<{ id: string; name: string }> = [];
|
|
726
|
+
const seenIds = new Set<string>();
|
|
727
|
+
|
|
728
|
+
for (const block of content) {
|
|
729
|
+
if (block.type !== "tool_use") {
|
|
730
|
+
nextContent.push(block);
|
|
731
|
+
continue;
|
|
732
|
+
}
|
|
733
|
+
if (seenIds.has(block.id)) {
|
|
734
|
+
duplicates.push({ id: block.id, name: block.name });
|
|
735
|
+
continue;
|
|
736
|
+
}
|
|
737
|
+
const normalized: ToolUseBlock =
|
|
738
|
+
block.id.length === 0 ? { ...block, id: crypto.randomUUID() } : block;
|
|
739
|
+
seenIds.add(normalized.id);
|
|
740
|
+
nextContent.push(normalized);
|
|
741
|
+
toolUseBlocks.push(normalized);
|
|
742
|
+
}
|
|
743
|
+
|
|
744
|
+
return { content: nextContent, toolUseBlocks, duplicates };
|
|
745
|
+
}
|
|
746
|
+
|
|
707
747
|
/**
|
|
708
748
|
* The benign result returned for a sibling tool call that was deferred because
|
|
709
749
|
* an exclusive tool ran in the same turn. Phrased so the model treats it as a
|
|
710
|
-
* "not run yet" signal
|
|
750
|
+
* "not run yet" signal: read the exclusive tool's output, then re-issue this
|
|
711
751
|
* call if it is still the right next step.
|
|
712
752
|
*/
|
|
713
753
|
function deferredForExclusiveMessage(exclusiveToolName: string): string {
|
|
@@ -1176,7 +1216,7 @@ export class AgentLoop {
|
|
|
1176
1216
|
"Agent loop iteration start",
|
|
1177
1217
|
);
|
|
1178
1218
|
|
|
1179
|
-
let toolUseBlocks:
|
|
1219
|
+
let toolUseBlocks: ToolUseBlock[] = [];
|
|
1180
1220
|
// The provider rejection thrown by this iteration's call, if any. Set in
|
|
1181
1221
|
// the inner provider catch and read by the outer catch to confine
|
|
1182
1222
|
// error-stop recovery to genuine provider rejections — a throw from
|
|
@@ -1864,8 +1904,7 @@ export class AgentLoop {
|
|
|
1864
1904
|
// the `post-model-call` hook below, which may add or drop tool calls;
|
|
1865
1905
|
// this raw set drives only the completion log and the max-tokens branch.
|
|
1866
1906
|
const modelToolUseBlocks = response.content.filter(
|
|
1867
|
-
(block): block is
|
|
1868
|
-
block.type === "tool_use",
|
|
1907
|
+
(block): block is ToolUseBlock => block.type === "tool_use",
|
|
1869
1908
|
);
|
|
1870
1909
|
|
|
1871
1910
|
rlog.info(
|
|
@@ -1979,19 +2018,26 @@ export class AgentLoop {
|
|
|
1979
2018
|
// if the model had called it (the supported way for a plugin to surface
|
|
1980
2019
|
// a card or take a follow-up action), or drop one the model emitted, so
|
|
1981
2020
|
// the loop runs whatever the assistant message ends up carrying.
|
|
1982
|
-
//
|
|
1983
|
-
//
|
|
1984
|
-
|
|
1985
|
-
|
|
1986
|
-
block.type === "tool_use",
|
|
2021
|
+
// Normalizing ids keeps executor dispatch and tool_result correlation
|
|
2022
|
+
// 1:1 for the rest of the turn.
|
|
2023
|
+
const normalizedToolUse = normalizeToolUseBlocks(
|
|
2024
|
+
assistantMessage.content,
|
|
1987
2025
|
);
|
|
1988
|
-
const
|
|
1989
|
-
|
|
1990
|
-
|
|
1991
|
-
|
|
1992
|
-
|
|
1993
|
-
|
|
2026
|
+
for (const duplicate of normalizedToolUse.duplicates) {
|
|
2027
|
+
rlog.warn(
|
|
2028
|
+
{
|
|
2029
|
+
turn: toolUseTurns,
|
|
2030
|
+
duplicateId: duplicate.id,
|
|
2031
|
+
duplicateName: duplicate.name,
|
|
2032
|
+
},
|
|
2033
|
+
"Duplicate tool_use id in the assistant reply, coalescing into a single call",
|
|
2034
|
+
);
|
|
1994
2035
|
}
|
|
2036
|
+
assistantMessage = {
|
|
2037
|
+
...assistantMessage,
|
|
2038
|
+
content: normalizedToolUse.content,
|
|
2039
|
+
};
|
|
2040
|
+
toolUseBlocks = normalizedToolUse.toolUseBlocks;
|
|
1995
2041
|
|
|
1996
2042
|
// At the no-tool stop boundary the retry decision is actionable: a
|
|
1997
2043
|
// recovery hook may repair history and ask to re-query (a tool-bearing
|