@vellumai/assistant 0.11.2 → 0.11.3-staging.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/ARCHITECTURE.md +58 -3
- package/docs/trusted-contact-access.md +9 -5
- package/docs/vellum-doctor.md +195 -0
- package/node_modules/@vellumai/ces-client/node_modules/@vellumai/service-contracts/src/channels.ts +39 -0
- package/node_modules/@vellumai/ces-client/node_modules/@vellumai/service-contracts/src/ingress.ts +10 -0
- package/node_modules/@vellumai/ces-client/node_modules/@vellumai/service-contracts/src/remote-web-pairing.ts +6 -0
- package/node_modules/@vellumai/ces-client/src/index.ts +1 -0
- package/node_modules/@vellumai/ces-client/src/rpc-client.ts +19 -2
- package/node_modules/@vellumai/environments/package.json +2 -1
- package/node_modules/@vellumai/environments/src/__tests__/cloud-assistant-hub-url.test.ts +38 -0
- package/node_modules/@vellumai/environments/src/__tests__/install-layout.test.ts +98 -0
- package/node_modules/@vellumai/environments/src/__tests__/package-boundary.test.ts +5 -5
- package/node_modules/@vellumai/environments/src/index.ts +17 -6
- package/node_modules/@vellumai/environments/src/install-layout.ts +49 -0
- package/node_modules/@vellumai/environments/src/seeds.ts +29 -0
- package/node_modules/@vellumai/gateway-client/node_modules/@vellumai/service-contracts/src/channels.ts +39 -0
- package/node_modules/@vellumai/gateway-client/node_modules/@vellumai/service-contracts/src/ingress.ts +10 -0
- package/node_modules/@vellumai/gateway-client/node_modules/@vellumai/service-contracts/src/remote-web-pairing.ts +6 -0
- package/node_modules/@vellumai/gateway-client/src/index.ts +4 -0
- package/node_modules/@vellumai/gateway-client/src/verification-session-contract.ts +91 -6
- package/node_modules/@vellumai/ipc-server-utils/src/endpoint.test.ts +36 -0
- package/node_modules/@vellumai/ipc-server-utils/src/endpoint.ts +142 -0
- package/node_modules/@vellumai/ipc-server-utils/src/index.ts +2 -0
- package/node_modules/@vellumai/ipc-server-utils/src/listen-options.ts +3 -0
- package/node_modules/@vellumai/ipc-server-utils/src/socket-watchdog.test.ts +15 -1
- package/node_modules/@vellumai/ipc-server-utils/src/socket-watchdog.ts +17 -2
- package/node_modules/@vellumai/service-contracts/src/channels.ts +39 -0
- package/node_modules/@vellumai/service-contracts/src/ingress.ts +10 -0
- package/node_modules/@vellumai/service-contracts/src/remote-web-pairing.ts +6 -0
- package/node_modules/@vellumai/slack-text/src/index.ts +16 -2
- package/openapi.yaml +589 -30
- package/package.json +1 -1
- package/src/__tests__/agent-image-optimize.test.ts +4 -4
- package/src/__tests__/agent-loop-mutable-latest-user-message.test.ts +113 -66
- package/src/__tests__/agent-loop-override-profile.test.ts +38 -48
- package/src/__tests__/agent-loop-subagent-billing-headers.test.ts +166 -0
- package/src/__tests__/agent-wake-disk-pressure-callsite.test.ts +1 -0
- package/src/__tests__/agent-wake-override-profile.test.ts +1 -0
- package/src/__tests__/answered-question-persistence.test.ts +240 -0
- package/src/__tests__/anthropic-provider.test.ts +165 -30
- package/src/__tests__/app-control-flow.test.ts +9 -8
- package/src/__tests__/assistant-event-hub.test.ts +48 -0
- package/src/__tests__/attachments-store-heic-normalize.test.ts +16 -12
- package/src/__tests__/attachments-store.test.ts +130 -68
- package/src/__tests__/attachments.test.ts +36 -36
- package/src/__tests__/background-workers-disk-pressure.test.ts +1 -1
- package/src/__tests__/call-start-guardian-guard.test.ts +18 -4
- package/src/__tests__/canned-reply-release.test.ts +86 -0
- package/src/__tests__/channel-guardian.test.ts +59 -29
- package/src/__tests__/channel-readiness-telegram-remote.test.ts +154 -0
- package/src/__tests__/channel-setup-panel-ack.test.ts +6 -5
- package/src/__tests__/client-os-metadata-persistence.test.ts +103 -4
- package/src/__tests__/compactor-image-manifest-trust.test.ts +1 -1
- package/src/__tests__/consult-deadline.test.ts +65 -0
- package/src/__tests__/conversation-agent-loop.test.ts +438 -2
- package/src/__tests__/conversation-disk-view-integration.test.ts +5 -1
- package/src/__tests__/conversation-disk-view.test.ts +4 -4
- package/src/__tests__/conversation-error.test.ts +22 -3
- package/src/__tests__/conversation-fork-crud.test.ts +5 -1
- package/src/__tests__/conversation-fork-referential.test.ts +352 -0
- package/src/__tests__/conversation-fork-retrospective.test.ts +5 -1
- package/src/__tests__/conversation-group-tools.test.ts +32 -2
- package/src/__tests__/conversation-inference-profile-route.test.ts +27 -1
- package/src/__tests__/conversation-initial-prompt.test.ts +51 -22
- package/src/__tests__/conversation-key-store-bootstrap-cleanup.test.ts +119 -0
- package/src/__tests__/conversation-key-store-origin.test.ts +105 -0
- package/src/__tests__/conversation-list-recency-ordering.test.ts +104 -0
- package/src/__tests__/conversation-origin-at-creation.test.ts +153 -0
- package/src/__tests__/conversation-origin-channel-filter.test.ts +95 -0
- package/src/__tests__/conversation-placement-promotion.test.ts +172 -0
- package/src/__tests__/conversation-process-app-control-preactivation.test.ts +123 -0
- package/src/__tests__/conversation-queue.test.ts +433 -1
- package/src/__tests__/conversation-retry-route.test.ts +3 -1
- package/src/__tests__/conversation-routes-hidden-queue.test.ts +64 -1
- package/src/__tests__/conversation-store-ephemeral.test.ts +4 -0
- package/src/__tests__/conversation-store.test.ts +26 -10
- package/src/__tests__/conversation-summarize-route.test.ts +1 -1
- package/src/__tests__/conversation-surfaces-action-delivery.test.ts +15 -6
- package/src/__tests__/conversation-surfaces-activation-emit.test.ts +6 -5
- package/src/__tests__/conversation-surfaces-app-control.test.ts +12 -11
- package/src/__tests__/conversation-surfaces-app-open.test.ts +11 -5
- package/src/__tests__/conversation-surfaces-data-persist.test.ts +5 -4
- package/src/__tests__/conversation-surfaces-history-restored-completion.test.ts +967 -0
- package/src/__tests__/conversation-surfaces-persisted-info.test.ts +312 -0
- package/src/__tests__/conversation-surfaces-queued-emit.test.ts +6 -5
- package/src/__tests__/conversation-surfaces-standalone-payloads.test.ts +12 -8
- package/src/__tests__/conversation-surfaces-standalone.test.ts +21 -11
- package/src/__tests__/conversation-surfaces-state-update.test.ts +7 -11
- package/src/__tests__/conversation-surfaces-table-action.test.ts +7 -6
- package/src/__tests__/conversation-surfaces-task-progress.test.ts +11 -6
- package/src/__tests__/conversation-tool-setup-app-refresh.test.ts +9 -6
- package/src/__tests__/conversation-tool-setup-attribution.test.ts +70 -7
- package/src/__tests__/conversation-wait-for-idle.test.ts +6 -11
- package/src/__tests__/credential-prompt-route.test.ts +57 -7
- package/src/__tests__/credential-security-invariants.test.ts +1 -0
- package/src/__tests__/cu-unified-flow.test.ts +89 -28
- package/src/__tests__/custom-group-listing-visibility.test.ts +1 -1
- package/src/__tests__/db-schedule-syntax-migration.test.ts +4 -1
- package/src/__tests__/discord-access-request-privacy.test.ts +366 -0
- package/src/__tests__/discord-requester-notice-privacy.test.ts +87 -0
- package/src/__tests__/disk-pressure-guard.test.ts +55 -55
- package/src/__tests__/disk-pressure-routes.test.ts +10 -10
- package/src/__tests__/disk-pressure-tools.test.ts +2 -2
- package/src/__tests__/disk-usage.test.ts +35 -20
- package/src/__tests__/document-sync-tags.test.ts +379 -0
- package/src/__tests__/document-update-default-surface.test.ts +28 -0
- package/src/__tests__/dynamic-skill-background-guard.test.ts +2 -2
- package/src/__tests__/emit-signal-routing-intent.test.ts +3 -4
- package/src/__tests__/ensure-conversation-exists.test.ts +8 -5
- package/src/__tests__/events-dev-bypass-actor.test.ts +183 -1
- package/src/__tests__/evict-conversations-for-reload.test.ts +90 -0
- package/src/__tests__/file-list-tool.test.ts +12 -10
- package/src/__tests__/file-ops-service.test.ts +151 -44
- package/src/__tests__/file-read-container-boundary.test.ts +137 -0
- package/src/__tests__/file-read-tool.test.ts +21 -9
- package/src/__tests__/file-url-path-guard.test.ts +83 -0
- package/src/__tests__/filesystem-tools.test.ts +82 -66
- package/src/__tests__/guardian-outbound-http.test.ts +8 -8
- package/src/__tests__/helpers/gateway-verification-sessions-stub.ts +14 -1
- package/src/__tests__/helpers/hub-clients.ts +42 -0
- package/src/__tests__/helpers/mock-conversation.ts +46 -0
- package/src/__tests__/helpers/verification-sessions-ipc-sim.ts +61 -57
- package/src/__tests__/history-repair.test.ts +112 -12
- package/src/__tests__/host-bash-proxy.test.ts +10 -5
- package/src/__tests__/host-bash-routes.test.ts +67 -0
- package/src/__tests__/host-browser-proxy.test.ts +2 -2
- package/src/__tests__/host-cu-proxy.test.ts +3 -7
- package/src/__tests__/host-file-proxy.test.ts +3 -7
- package/src/__tests__/host-file-read-tool.test.ts +5 -1
- package/src/__tests__/host-transfer-proxy-targeted.test.ts +6 -4
- package/src/__tests__/http-conversation-lineage.test.ts +2 -5
- package/src/__tests__/http-user-message-parity.test.ts +44 -0
- package/src/__tests__/identity-routes.test.ts +13 -13
- package/src/__tests__/image-conversion.test.ts +29 -26
- package/src/__tests__/image-source-path-reinject.test.ts +2 -2
- package/src/__tests__/inference-profile-session-handler.test.ts +95 -0
- package/src/__tests__/inference-profile-session-ipc.test.ts +25 -0
- package/src/__tests__/introduction-card-resolver.test.ts +6 -1
- package/src/__tests__/list-messages-attachments.test.ts +56 -32
- package/src/__tests__/list-messages-background-tool-completion.test.ts +4 -4
- package/src/__tests__/list-messages-client-message-id.test.ts +2 -2
- package/src/__tests__/list-messages-hidden-metadata.test.ts +16 -16
- package/src/__tests__/list-messages-page-latest.test.ts +61 -61
- package/src/__tests__/list-messages-provider-error.test.ts +6 -6
- package/src/__tests__/list-messages-queued.test.ts +49 -6
- package/src/__tests__/list-messages-system-card.test.ts +4 -4
- package/src/__tests__/list-messages-tool-merge.test.ts +9 -9
- package/src/__tests__/llm-resolver-override-or-default.test.ts +141 -0
- package/src/__tests__/llm-usage-store.test.ts +55 -0
- package/src/__tests__/manual-token-reconciliation.test.ts +45 -0
- package/src/__tests__/mcp-tool-annotations-risk.test.ts +133 -0
- package/src/__tests__/media-stream-server-integration.test.ts +36 -0
- package/src/__tests__/migration-export-http.test.ts +1 -1
- package/src/__tests__/migration-import-commit-http.test.ts +1 -1
- package/src/__tests__/migration-import-preflight-http.test.ts +1 -1
- package/src/__tests__/migration-validate-http.test.ts +1 -1
- package/src/__tests__/mtime-cache.test.ts +65 -3
- package/src/__tests__/notification-platform-adapter.test.ts +104 -0
- package/src/__tests__/notification-schedule-notify-dedup.test.ts +1 -1
- package/src/__tests__/oauth-provider-profiles.test.ts +4 -2
- package/src/__tests__/oauth-provider-seed-logos.test.ts +65 -0
- package/src/__tests__/oauth-provider-serializer.test.ts +17 -0
- package/src/__tests__/oauth-providers-routes.test.ts +1 -0
- package/src/__tests__/openai-responses-prompt-cache.test.ts +5 -2
- package/src/__tests__/path-policy.test.ts +164 -1
- package/src/__tests__/persist-media-references.test.ts +2 -2
- package/src/__tests__/persist-unsendable-image-downscale.test.ts +5 -5
- package/src/__tests__/persist-unsendable-image.test.ts +23 -23
- package/src/__tests__/persona-resolver.test.ts +154 -0
- package/src/__tests__/plugin-api-webhook-url.test.ts +170 -0
- package/src/__tests__/plugin-import-boundary-guard.test.ts +6 -0
- package/src/__tests__/processing-flag-persist-failure.test.ts +15 -10
- package/src/__tests__/profile-availability-incomplete.test.ts +113 -0
- package/src/__tests__/profiler-routes.test.ts +13 -13
- package/src/__tests__/prompt-cache-cross-turn-stability.test.ts +593 -0
- package/src/__tests__/proxy-approval-callback.test.ts +3 -3
- package/src/__tests__/queued-message-cancel-and-steer.test.ts +189 -0
- package/src/__tests__/queued-message-steer-actor-scoping.test.ts +164 -0
- package/src/__tests__/resolve-trust-class.test.ts +20 -0
- package/src/__tests__/run-conversation-turn-persistence.test.ts +1 -1
- package/src/__tests__/run-due-schedules.test.ts +174 -0
- package/src/__tests__/runtime-attachment-metadata.test.ts +15 -7
- package/src/__tests__/scaffold-managed-skill-tool.test.ts +166 -0
- package/src/__tests__/schedule-routes.test.ts +730 -8
- package/src/__tests__/schedule-store.test.ts +540 -0
- package/src/__tests__/schedule-tools.test.ts +103 -0
- package/src/__tests__/scheduler-wake.test.ts +5 -0
- package/src/__tests__/secret-routes-platform-proxy.test.ts +29 -1
- package/src/__tests__/secure-keys.test.ts +91 -0
- package/src/__tests__/server-history-render.test.ts +97 -7
- package/src/__tests__/size-guard.test.ts +8 -8
- package/src/__tests__/skill-docs-sync-guard.test.ts +3 -2
- package/src/__tests__/skill-tool-factory.test.ts +247 -0
- package/src/__tests__/skills-install-staging.test.ts +15 -6
- package/src/__tests__/slack-edit-ordering-characterization.test.ts +103 -0
- package/src/__tests__/slack-inbound-verification.test.ts +4 -1
- package/src/__tests__/slack-mention-binding-invariance.test.ts +103 -0
- package/src/__tests__/slack-mention-provider-leak-audit.test.ts +135 -0
- package/src/__tests__/starter-task-flow.test.ts +5 -4
- package/src/__tests__/steer-on-enqueue-question.test.ts +3 -2
- package/src/__tests__/stt-language-catalog-parity.test.ts +38 -0
- package/src/__tests__/subagent-allowlist-validation.test.ts +26 -4
- package/src/__tests__/subagent-cron-run-attribution.test.ts +301 -0
- package/src/__tests__/subagent-disposal.test.ts +10 -0
- package/src/__tests__/subagent-fork-notifications.test.ts +10 -0
- package/src/__tests__/subagent-fork-prompt-role.test.ts +85 -11
- package/src/__tests__/subagent-fork-spawn.test.ts +15 -7
- package/src/__tests__/subagent-manager-notify.test.ts +148 -0
- package/src/__tests__/subagent-notify-parent.test.ts +1 -1
- package/src/__tests__/subagent-persistence.test.ts +141 -0
- package/src/__tests__/subagent-role-registry.test.ts +172 -63
- package/src/__tests__/subagent-role-resolution.test.ts +88 -0
- package/src/__tests__/subagent-spawn-and-await.test.ts +48 -0
- package/src/__tests__/subagent-spawn-tool-fork.test.ts +111 -10
- package/src/__tests__/subagent-terminal-message.test.ts +88 -1
- package/src/__tests__/subagent-tool-gate-mode.test.ts +225 -10
- package/src/__tests__/subagent-tools.test.ts +1901 -108
- package/src/__tests__/surface-completion-compaction-boundary.test.ts +124 -0
- package/src/__tests__/surface-completion-in-flight-snapshot.test.ts +298 -0
- package/src/__tests__/system-prompt.test.ts +82 -7
- package/src/__tests__/tool-grant-request-escalation.test.ts +40 -1
- package/src/__tests__/tool-result-spool.test.ts +74 -18
- package/src/__tests__/tool-side-effects-documents.test.ts +174 -0
- package/src/__tests__/tool-side-effects-hook-failures.test.ts +129 -0
- package/src/__tests__/tools-audio-read.test.ts +10 -10
- package/src/__tests__/tools-get-route.test.ts +85 -12
- package/src/__tests__/tools-list-cli.test.ts +171 -0
- package/src/__tests__/top-level-renderer.test.ts +54 -1
- package/src/__tests__/trusted-contact-inline-approval-integration.test.ts +68 -1
- package/src/__tests__/trusted-contact-multichannel.test.ts +4 -1
- package/src/__tests__/turn-tail-chain.test.ts +135 -0
- package/src/__tests__/ui-choice-copy-surfaces.test.ts +6 -4
- package/src/__tests__/ui-visual-surface.test.ts +5 -4
- package/src/__tests__/ui-work-result-surface.test.ts +5 -4
- package/src/__tests__/unified-turn-context-visible-app.test.ts +7 -8
- package/src/__tests__/usage-routes.test.ts +62 -0
- package/src/__tests__/verification-outbound-delivery.test.ts +176 -0
- package/src/__tests__/visible-app-context.test.ts +20 -4
- package/src/__tests__/web-search-history.test.ts +96 -0
- package/src/__tests__/workspace-migration-139-clear-renamed-cost-profile-label.test.ts +6 -6
- package/src/__tests__/workspace-migration-140-repair-seed-pinned-memory-v3-live.test.ts +330 -0
- package/src/acp/failure-error.ts +3 -5
- package/src/agent/attachments.ts +26 -24
- package/src/agent/history-repair/history-repair.ts +27 -26
- package/src/agent/image-optimize.ts +6 -6
- package/src/agent/loop.ts +40 -22
- package/src/agent/message-types.ts +3 -3
- package/src/api/constants/document-tools.ts +55 -0
- package/src/api/events/question-answered.ts +53 -0
- package/src/api/events/tool-result.ts +7 -0
- package/src/api/index.ts +13 -0
- package/src/api/responses/conversation-message.ts +11 -0
- package/src/api/surfaces-oauth-connect.test.ts +96 -0
- package/src/api/surfaces.ts +36 -8
- package/src/approvals/guardian-channel-delivery.ts +83 -0
- package/src/approvals/guardian-expiry-notifier.ts +13 -11
- package/src/approvals/guardian-request-resolvers.ts +222 -135
- package/src/calls/__tests__/voice-session-bridge.test.ts +50 -0
- package/src/calls/media-stream-server.ts +9 -1
- package/src/calls/voice-session-bridge.ts +38 -0
- package/src/channels/__tests__/gateway-verification-sessions.test.ts +3 -3
- package/src/channels/config.ts +7 -2
- package/src/channels/gateway-verification-sessions.ts +7 -2
- package/src/channels/types.ts +7 -1
- package/src/cli/commands/__tests__/cache.test.ts +7 -4
- package/src/cli/commands/__tests__/cli-test-harness.ts +82 -0
- package/src/cli/commands/__tests__/conversations-wake.test.ts +51 -6
- package/src/cli/commands/__tests__/inference-models.test.ts +14 -0
- package/src/cli/commands/__tests__/inference-profiles.test.ts +44 -1
- package/src/cli/commands/__tests__/inference-providers.test.ts +7 -0
- package/src/cli/commands/__tests__/plugins.test.ts +692 -0
- package/src/cli/commands/__tests__/schedules.test.ts +136 -37
- package/src/cli/commands/conversations.ts +7 -3
- package/src/cli/commands/credentials.help.ts +7 -0
- package/src/cli/commands/inference-profiles.ts +9 -3
- package/src/cli/commands/inference.help.ts +16 -2
- package/src/cli/commands/mcp.help.ts +4 -2
- package/src/cli/commands/memory/memory-retrospective.ts +18 -0
- package/src/cli/commands/notifications.ts +3 -6
- package/src/cli/commands/oauth/connect-surface-guidance.test.ts +16 -0
- package/src/cli/commands/oauth/connect-surface-guidance.ts +25 -7
- package/src/cli/commands/oauth/connect.test.ts +26 -1
- package/src/cli/commands/oauth/connect.ts +4 -1
- package/src/cli/commands/oauth/index.help.ts +3 -1
- package/src/cli/commands/oauth/providers.ts +3 -0
- package/src/cli/commands/plugins.help.ts +25 -2
- package/src/cli/commands/plugins.ts +289 -10
- package/src/cli/commands/schedules.help.ts +14 -12
- package/src/cli/commands/schedules.ts +19 -5
- package/src/cli/commands/tools.help.ts +4 -2
- package/src/cli/commands/tools.ts +38 -0
- package/src/cli/lib/__tests__/inspect-plugin.test.ts +1 -0
- package/src/cli/lib/__tests__/install-from-github.test.ts +79 -0
- package/src/cli/lib/__tests__/install-from-platform.test.ts +47 -1
- package/src/cli/lib/__tests__/plugin-surfaces.test.ts +86 -7
- package/src/cli/lib/__tests__/upgrade-plugin.test.ts +122 -0
- package/src/cli/lib/install-from-github.ts +61 -0
- package/src/cli/lib/install-from-platform.ts +13 -2
- package/src/cli/lib/plugin-surfaces.ts +124 -4
- package/src/cli/lib/upgrade-plugin.ts +18 -0
- package/src/cli.ts +1 -1
- package/src/config/__tests__/deployment-context-defaults.test.ts +73 -0
- package/src/config/__tests__/memory-retrospective-schema.test.ts +40 -0
- package/src/config/__tests__/plugin-updates-schema.test.ts +43 -0
- package/src/config/__tests__/profile-tool-support.test.ts +131 -0
- package/src/config/bundled-skills/AGENTS.md +1 -1
- package/src/config/bundled-skills/app-builder/SKILL.md +1 -1
- package/src/config/bundled-skills/app-builder/references/RESPONSIVE.md +3 -3
- package/src/config/bundled-skills/schedule/SKILL.md +5 -1
- package/src/config/bundled-skills/schedule/TOOLS.json +2 -2
- package/src/config/bundled-skills/settings/TOOLS.json +2 -2
- package/src/config/bundled-skills/settings/tools/voice-config-update.ts +1 -0
- package/src/config/bundled-skills/subagent/SKILL.md +53 -22
- package/src/config/bundled-skills/subagent/TOOLS.json +14 -13
- package/src/config/bundled-skills/visualize/SKILL.md +1 -1
- package/src/config/default-profile-catalog.ts +6 -2
- package/src/config/feature-flag-registry.json +25 -17
- package/src/config/llm-resolver.ts +45 -15
- package/src/config/loader.ts +70 -1
- package/src/config/profile-tool-support.ts +110 -0
- package/src/config/sanitize-for-transfer.ts +11 -0
- package/src/config/schema.ts +4 -0
- package/src/config/schemas/__tests__/stt.test.ts +20 -5
- package/src/config/schemas/channels.ts +13 -0
- package/src/config/schemas/memory-retrospective.ts +25 -0
- package/src/config/schemas/plugin-updates.ts +55 -0
- package/src/config/schemas/services.ts +8 -4
- package/src/config/schemas/stt.ts +17 -5
- package/src/config/schemas/timeouts.ts +1 -1
- package/src/contacts/__tests__/guardian-delivery-reader.test.ts +16 -0
- package/src/contacts/guardian-delivery-reader.ts +29 -2
- package/src/context/compactor.ts +4 -4
- package/src/context/post-turn-tool-result-truncation.ts +32 -16
- package/src/context/strip-injections.ts +42 -0
- package/src/context/tool-result-spool.ts +32 -18
- package/src/credential-execution/ces-runtime.ts +3 -0
- package/src/credential-execution/client.ts +22 -1
- package/src/credential-execution/process-manager.test.ts +99 -0
- package/src/credential-execution/process-manager.ts +62 -16
- package/src/daemon/__tests__/abort-null-controller.test.ts +141 -5
- package/src/daemon/__tests__/conversation-surfaces-launch.test.ts +8 -12
- package/src/daemon/__tests__/install-assistant-command.test.ts +234 -0
- package/src/daemon/__tests__/turn-tail-assistant-reply-notify.test.ts +2 -4
- package/src/daemon/__tests__/turn-tail-deleted-conversation.test.ts +20 -29
- package/src/daemon/channel-ui-capability.ts +4 -4
- package/src/daemon/conversation-agent-loop-handlers.ts +117 -61
- package/src/daemon/conversation-agent-loop.ts +318 -148
- package/src/daemon/conversation-attachments.ts +1 -1
- package/src/daemon/conversation-error.ts +8 -1
- package/src/daemon/conversation-initial-prompt.ts +16 -38
- package/src/daemon/conversation-lifecycle.ts +154 -32
- package/src/daemon/conversation-messaging.ts +106 -43
- package/src/daemon/conversation-process.ts +131 -33
- package/src/daemon/conversation-queue-manager.ts +10 -0
- package/src/daemon/conversation-runtime-assembly.ts +9 -4
- package/src/daemon/conversation-store.ts +52 -3
- package/src/daemon/conversation-surface-state.ts +9 -0
- package/src/daemon/conversation-surfaces.ts +403 -210
- package/src/daemon/conversation-tool-setup.ts +56 -7
- package/src/daemon/conversation-turn-finalize.ts +134 -89
- package/src/daemon/conversation.ts +111 -27
- package/src/daemon/disk-pressure-guard-lifecycle.ts +2 -2
- package/src/daemon/disk-pressure-guard.ts +3 -3
- package/src/daemon/doordash-steps.ts +5 -5
- package/src/daemon/external-plugins-bootstrap.ts +7 -1
- package/src/daemon/handlers/config-channels.ts +48 -5
- package/src/daemon/handlers/conversation-history.ts +1 -1
- package/src/daemon/handlers/conversations.ts +62 -17
- package/src/daemon/handlers/shared.ts +12 -0
- package/src/daemon/handlers/skills.ts +11 -1
- package/src/daemon/identity-helpers.ts +38 -10
- package/src/daemon/install-assistant-command.ts +322 -0
- package/src/daemon/lifecycle.ts +14 -3
- package/src/daemon/message-types/conversations.ts +6 -7
- package/src/daemon/message-types/sync.ts +1 -0
- package/src/daemon/persist-media-references.ts +23 -14
- package/src/daemon/process-message.ts +6 -6
- package/src/daemon/tool-setup-types.ts +19 -122
- package/src/daemon/tool-side-effects.ts +17 -94
- package/src/daemon/trust-context-types.ts +39 -0
- package/src/daemon/trust-context.ts +14 -6
- package/src/daemon/turn-tail-chain.ts +73 -0
- package/src/daemon/web-search-history.ts +47 -33
- package/src/heartbeat/heartbeat-service.ts +6 -6
- package/src/ipc/__tests__/socket-path.test.ts +21 -0
- package/src/ipc/assistant-server.ts +17 -10
- package/src/ipc/cli-client.ts +10 -1
- package/src/ipc/routes/__tests__/documents-sync-ipc-routes.test.ts +56 -0
- package/src/ipc/routes/documents-sync-ipc-routes.ts +42 -0
- package/src/ipc/socket-cleanup.ts +9 -7
- package/src/ipc/socket-path.ts +4 -96
- package/src/live-voice/__tests__/live-voice-archive.test.ts +16 -16
- package/src/live-voice/__tests__/live-voice-attach-image.test.ts +209 -0
- package/src/live-voice/__tests__/live-voice-connection.test.ts +20 -0
- package/src/live-voice/__tests__/live-voice-integration.test.ts +7 -5
- package/src/live-voice/__tests__/live-voice-session-telemetry.test.ts +265 -0
- package/src/live-voice/__tests__/live-voice-stt.test.ts +6 -1
- package/src/live-voice/__tests__/live-voice-vad.test.ts +35 -15
- package/src/live-voice/__tests__/protocol.test.ts +52 -0
- package/src/live-voice/__tests__/runtime-websocket-shell.test.ts +3 -0
- package/src/live-voice/live-voice-archive.ts +13 -13
- package/src/live-voice/live-voice-connection.ts +7 -1
- package/src/live-voice/live-voice-photo.ts +169 -0
- package/src/live-voice/live-voice-session.ts +108 -4
- package/src/live-voice/protocol.ts +89 -1
- package/src/mcp/client.ts +10 -0
- package/src/messaging/providers/__tests__/callback-routing.test.ts +5 -1
- package/src/messaging/providers/__tests__/transport-dispatch.test.ts +66 -4
- package/src/messaging/providers/callback-routing.ts +1 -0
- package/src/messaging/providers/discord/api.ts +312 -0
- package/src/messaging/providers/discord/dm-delivery.test.ts +204 -0
- package/src/messaging/providers/discord/render.test.ts +204 -0
- package/src/messaging/providers/discord/render.ts +195 -0
- package/src/messaging/providers/discord/send.test.ts +179 -0
- package/src/messaging/providers/discord/send.ts +212 -0
- package/src/messaging/providers/discord/transport.ts +81 -0
- package/src/messaging/providers/index.ts +3 -1
- package/src/messaging/providers/retry-policy.test.ts +247 -0
- package/src/messaging/providers/retry-policy.ts +206 -0
- package/src/messaging/providers/slack/mention-source.test.ts +285 -0
- package/src/messaging/providers/slack/mention-source.ts +287 -0
- package/src/messaging/providers/telegram-bot/api.ts +47 -130
- package/src/messaging/providers/whatsapp/api.ts +47 -136
- package/src/monitoring/__tests__/db-integrity-sample.test.ts +4 -1
- package/src/monitoring/__tests__/plugin-auto-update.test.ts +332 -0
- package/src/monitoring/db-integrity-sample.ts +6 -1
- package/src/monitoring/plugin-auto-update.ts +359 -0
- package/src/monitoring/resource-sampler.ts +21 -5
- package/src/monitoring/worker.ts +13 -0
- package/src/notifications/__tests__/assistant-reply-producer.test.ts +375 -2
- package/src/notifications/__tests__/broadcaster.test.ts +37 -1
- package/src/notifications/__tests__/copy-composer.test.ts +165 -0
- package/src/notifications/__tests__/emit-signal-pipeline-failure.test.ts +205 -0
- package/src/notifications/__tests__/notification-utils.test.ts +249 -0
- package/src/notifications/adapters/platform.ts +31 -2
- package/src/notifications/assistant-reply-producer.ts +79 -32
- package/src/notifications/broadcaster.ts +30 -10
- package/src/notifications/copy-composer.ts +69 -0
- package/src/notifications/emit-signal.ts +82 -15
- package/src/notifications/events-store.ts +10 -2
- package/src/notifications/home-feed-side-effect.ts +6 -3
- package/src/notifications/notification-utils.ts +173 -5
- package/src/notifications/signal.ts +12 -0
- package/src/oauth/AGENTS.md +4 -2
- package/src/oauth/byo-connection.ts +6 -1
- package/src/oauth/credential-token-resolver.ts +3 -10
- package/src/oauth/manual-token-connection.ts +29 -64
- package/src/oauth/manual-token-providers.ts +62 -0
- package/src/oauth/provider-serializer.ts +13 -0
- package/src/oauth/seed-providers.ts +62 -11
- package/src/onboarding/onboarding-events-store.ts +91 -1
- package/src/permissions/question-prompter.test.ts +4 -0
- package/src/permissions/question-prompter.ts +21 -4
- package/src/persistence/__tests__/conversation-lineage.test.ts +248 -0
- package/src/persistence/attachments-store.ts +15 -15
- package/src/persistence/conversation-attention-store.ts +30 -10
- package/src/persistence/conversation-bootstrap.ts +17 -0
- package/src/persistence/conversation-crud.ts +363 -97
- package/src/persistence/conversation-group-migration.ts +1 -1
- package/src/persistence/conversation-key-store.ts +38 -10
- package/src/persistence/conversation-lineage.ts +226 -0
- package/src/persistence/conversation-plugin-facade.ts +28 -3
- package/src/persistence/conversation-queries.ts +207 -33
- package/src/persistence/conversation-types.ts +107 -1
- package/src/persistence/jobs-store.ts +32 -8
- package/src/persistence/llm-usage-store.ts +22 -0
- package/src/persistence/migrations/361-normalize-managed-connection-rows.test.ts +168 -0
- package/src/persistence/migrations/361-normalize-managed-connection-rows.ts +76 -0
- package/src/persistence/migrations/362-add-conversation-subagent-kind.test.ts +187 -0
- package/src/persistence/migrations/362-add-conversation-subagent-kind.ts +93 -0
- package/src/persistence/migrations/363-backfill-schedule-inference-profile.test.ts +229 -0
- package/src/persistence/migrations/363-backfill-schedule-inference-profile.ts +127 -0
- package/src/persistence/migrations/364-add-schedule-source-key.test.ts +120 -0
- package/src/persistence/migrations/364-add-schedule-source-key.ts +43 -0
- package/src/persistence/migrations/365-add-conversation-fork-strategy.ts +27 -0
- package/src/persistence/schema/conversations.ts +32 -0
- package/src/persistence/schema/infrastructure.ts +3 -0
- package/src/persistence/steps.ts +30 -0
- package/src/persistence/subagent-store.ts +148 -6
- package/src/platform/managed-speech.test.ts +20 -0
- package/src/platform/managed-speech.ts +9 -0
- package/src/plugin-api/index.ts +8 -0
- package/src/plugin-api/webhook-url.ts +151 -0
- package/src/plugins/__tests__/installed-plugin-dirs.test.ts +82 -0
- package/src/plugins/__tests__/source-fingerprint.test.ts +5 -1
- package/src/plugins/collect-source-versions.ts +4 -27
- package/src/plugins/defaults/exploration-drift/hooks/post-tool-use.ts +5 -5
- package/src/plugins/defaults/exploration-drift/package.json +1 -1
- package/src/plugins/defaults/image-recovery/hooks/post-model-call.ts +4 -2
- package/src/plugins/defaults/image-recovery/recover.ts +67 -57
- package/src/plugins/defaults/index.ts +1 -1
- package/src/plugins/defaults/memory/__tests__/conversation-queries.test.ts +64 -18
- package/src/plugins/defaults/memory/__tests__/jobs-worker-outcome-contract.test.ts +242 -0
- package/src/plugins/defaults/memory/__tests__/jobs-worker-retrospective-sweep.test.ts +24 -1
- package/src/plugins/defaults/memory/__tests__/memory-retrospective-enqueue.test.ts +48 -3
- package/src/plugins/defaults/memory/__tests__/memory-retrospective-job.test.ts +511 -19
- package/src/plugins/defaults/memory/__tests__/memory-retrospective-prompt.test.ts +39 -3
- package/src/plugins/defaults/memory/__tests__/memory-retrospective-provider-path.test.ts +493 -0
- package/src/plugins/defaults/memory/__tests__/memory-retrospective-wake-chain.test.ts +5 -4
- package/src/plugins/defaults/memory/__tests__/memory-tier-boundary-guard.test.ts +5 -1
- package/src/plugins/defaults/memory/job-handlers.ts +87 -5
- package/src/plugins/defaults/memory/jobs-worker.ts +60 -6
- package/src/plugins/defaults/memory/memory-retrospective-constants.ts +11 -0
- package/src/plugins/defaults/memory/memory-retrospective-enqueue.ts +29 -0
- package/src/plugins/defaults/memory/memory-retrospective-job.ts +342 -60
- package/src/plugins/defaults/memory/memory-retrospective-prompt.ts +23 -3
- package/src/plugins/defaults/memory/substrate/consolidation-lock.ts +9 -4
- package/src/plugins/defaults/memory/substrate/sweep-job.ts +7 -0
- package/src/plugins/defaults/memory/v1/graph/injection.ts +1 -1
- package/src/plugins/defaults/memory/v2/router.ts +8 -0
- package/src/plugins/defaults/turn-context/unified-turn-context.ts +1 -2
- package/src/plugins/external-plugin-loader.ts +17 -8
- package/src/plugins/installed-plugin-dirs.ts +101 -0
- package/src/plugins/mtime-cache.ts +34 -22
- package/src/plugins/types.ts +69 -8
- package/src/prompts/persona-resolver.ts +6 -2
- package/src/prompts/system-prompt.ts +72 -17
- package/src/providers/__tests__/inference.test.ts +40 -0
- package/src/providers/__tests__/preflight-resolved-config.test.ts +58 -1
- package/src/providers/__tests__/retry-callsite.test.ts +67 -0
- package/src/providers/__tests__/vellum-connection-routing.test.ts +42 -11
- package/src/providers/anthropic/client.ts +112 -95
- package/src/providers/connection-resolution.ts +35 -1
- package/src/providers/gemini/client.ts +5 -5
- package/src/providers/inference/adapter-factory.ts +16 -10
- package/src/providers/inference/auth.ts +16 -0
- package/src/providers/inference/connection-availability.ts +164 -12
- package/src/providers/inference/connections.ts +25 -1
- package/src/providers/media-resolve.ts +21 -11
- package/src/providers/model-catalog.ts +15 -0
- package/src/providers/openai/chat-completions-provider.ts +4 -4
- package/src/providers/openai/responses-provider.ts +12 -25
- package/src/providers/registry.ts +9 -4
- package/src/providers/retry.ts +25 -0
- package/src/providers/server-tool-pairing.ts +134 -0
- package/src/providers/speech-to-text/__tests__/resolve.test.ts +187 -6
- package/src/providers/speech-to-text/deepgram.ts +1 -0
- package/src/providers/speech-to-text/resolve.ts +80 -10
- package/src/providers/speech-to-text/vellum-managed.ts +7 -1
- package/src/providers/speech-to-text/vellum-speech-relay-connection.ts +3 -1
- package/src/providers/types.ts +26 -6
- package/src/providers/vellum-model-routing.ts +6 -7
- package/src/runtime/AGENTS.md +2 -0
- package/src/runtime/__tests__/agent-wake.test.ts +16 -16
- package/src/runtime/__tests__/background-job-runner.test.ts +0 -144
- package/src/runtime/__tests__/desktop-presence.test.ts +162 -0
- package/src/runtime/agent-wake.ts +63 -24
- package/src/runtime/assistant-event-hub.ts +87 -12
- package/src/runtime/auth/__tests__/guard-tests.test.ts +2 -0
- package/src/runtime/auth/__tests__/scopes.test.ts +20 -0
- package/src/runtime/auth/same-actor.ts +56 -8
- package/src/runtime/auth/scopes.ts +12 -0
- package/src/runtime/auth/types.ts +2 -0
- package/src/runtime/background-job-runner.ts +5 -27
- package/src/runtime/channel-readiness-service.ts +59 -3
- package/src/runtime/channel-readiness-types.ts +18 -0
- package/src/runtime/desktop-presence.ts +69 -0
- package/src/runtime/http-server.ts +12 -0
- package/src/runtime/local-actor-identity.ts +11 -9
- package/src/runtime/migrations/__tests__/vbundle-symlink-streaming-importer.test.ts +0 -0
- package/src/runtime/routes/__tests__/channel-verification-routes.test.ts +1 -1
- package/src/runtime/routes/__tests__/client-routes.test.ts +186 -14
- package/src/runtime/routes/__tests__/conversation-list-routes.test.ts +547 -1
- package/src/runtime/routes/__tests__/conversation-management-routes.test.ts +162 -7
- package/src/runtime/routes/__tests__/conversation-query-routes.test.ts +14 -10
- package/src/runtime/routes/__tests__/inference-profiles-routes.test.ts +340 -3
- package/src/runtime/routes/__tests__/llm-call-sites-routes.test.ts +28 -1
- package/src/runtime/routes/__tests__/plugins-routes.test.ts +173 -1
- package/src/runtime/routes/__tests__/question-routes.test.ts +64 -0
- package/src/runtime/routes/__tests__/retrospective-routes.test.ts +20 -0
- package/src/runtime/routes/__tests__/skill-history-route.test.ts +116 -0
- package/src/runtime/routes/__tests__/wake-conversation-routes.test.ts +60 -2
- package/src/runtime/routes/archive-utils.ts +16 -12
- package/src/runtime/routes/attachment-routes.ts +13 -6
- package/src/runtime/routes/canned-reply-release.ts +54 -0
- package/src/runtime/routes/channel-verification-routes.ts +2 -27
- package/src/runtime/routes/chatgpt-subscription-auth-routes.ts +5 -0
- package/src/runtime/routes/client-routes.ts +114 -2
- package/src/runtime/routes/conversation-cli-routes.ts +10 -12
- package/src/runtime/routes/conversation-list-routes.ts +59 -10
- package/src/runtime/routes/conversation-management-routes.ts +47 -18
- package/src/runtime/routes/conversation-query-routes.ts +32 -53
- package/src/runtime/routes/conversation-routes.ts +360 -293
- package/src/runtime/routes/credential-prompt-routes.ts +23 -2
- package/src/runtime/routes/debug-routes.ts +4 -1
- package/src/runtime/routes/default-provider-routes.ts +7 -11
- package/src/runtime/routes/documents-routes.ts +44 -1
- package/src/runtime/routes/events-routes.ts +28 -1
- package/src/runtime/routes/host-app-control-routes.ts +2 -0
- package/src/runtime/routes/host-bash-routes.ts +2 -0
- package/src/runtime/routes/host-browser-routes.ts +2 -0
- package/src/runtime/routes/host-cu-routes.ts +2 -0
- package/src/runtime/routes/host-file-routes.ts +2 -0
- package/src/runtime/routes/host-transfer-routes.ts +4 -0
- package/src/runtime/routes/identity-routes.ts +6 -6
- package/src/runtime/routes/inbound-message-handler.ts +5 -5
- package/src/runtime/routes/inbound-stages/acl-enforcement.ts +7 -3
- package/src/runtime/routes/inbound-stages/background-dispatch.ts +14 -0
- package/src/runtime/routes/inbound-stages/bootstrap-intercept.ts +2 -2
- package/src/runtime/routes/inbound-stages/guardian-activation-intercept.test.ts +14 -11
- package/src/runtime/routes/inbound-stages/guardian-activation-intercept.ts +14 -17
- package/src/runtime/routes/inference-profile-availability-guard.ts +171 -0
- package/src/runtime/routes/inference-profile-session-handler.ts +26 -2
- package/src/runtime/routes/inference-profile-session-routes.ts +2 -1
- package/src/runtime/routes/inference-profiles-routes.ts +152 -73
- package/src/runtime/routes/inference-provider-connection-routes.ts +0 -16
- package/src/runtime/routes/llm-call-sites-routes.ts +13 -0
- package/src/runtime/routes/log-export-routes.ts +1 -1
- package/src/runtime/routes/notification-routes.ts +0 -14
- package/src/runtime/routes/oauth-providers.ts +1 -0
- package/src/runtime/routes/plugins-routes.ts +106 -1
- package/src/runtime/routes/profiler-routes.ts +4 -2
- package/src/runtime/routes/question-routes.ts +28 -10
- package/src/runtime/routes/retrospective-routes.ts +9 -3
- package/src/runtime/routes/schedule-routes.ts +210 -12
- package/src/runtime/routes/secret-routes.ts +40 -23
- package/src/runtime/routes/settings-routes.ts +41 -13
- package/src/runtime/routes/skills-routes.ts +79 -0
- package/src/runtime/routes/ui-snapshot-routes.ts +1 -0
- package/src/runtime/routes/wake-conversation-routes.ts +23 -0
- package/src/runtime/services/__tests__/conversation-serializer.test.ts +1 -0
- package/src/runtime/services/conversation-serializer.ts +24 -9
- package/src/runtime/sync/documents-sidecar-publish.test.ts +101 -0
- package/src/runtime/sync/resource-sync-events.ts +79 -1
- package/src/runtime/sync/worker-daemon-notify.test.ts +34 -0
- package/src/runtime/sync/worker-daemon-notify.ts +44 -6
- package/src/runtime/verification-outbound-actions.ts +250 -222
- package/src/runtime/verification-templates.ts +41 -30
- package/src/schedule/__tests__/plugin-schedule-declarations.test.ts +614 -0
- package/src/schedule/__tests__/plugin-schedule-reconciler.test.ts +1066 -0
- package/src/schedule/inference-profile.ts +64 -3
- package/src/schedule/plugin-schedule-declarations.ts +518 -0
- package/src/schedule/plugin-schedule-reconciler.ts +605 -0
- package/src/schedule/plugin-schedules-gate.ts +23 -0
- package/src/schedule/schedule-store.ts +520 -26
- package/src/schedule/scheduler.ts +38 -0
- package/src/security/credential-backend.ts +51 -4
- package/src/security/encrypted-store.ts +64 -42
- package/src/security/secure-keys.ts +55 -7
- package/src/skills/catalog-install.ts +20 -7
- package/src/skills/inline-command-runner.ts +2 -7
- package/src/skills/skill-history.test.ts +269 -0
- package/src/skills/skill-history.ts +250 -0
- package/src/skills/skillssh-registry.ts +1 -1
- package/src/stt/daemon-batch-transcriber.ts +8 -6
- package/src/subagent/__tests__/consult-prompt.test.ts +25 -0
- package/src/subagent/consult-prompt.ts +3 -1
- package/src/subagent/consult-transcript.ts +1 -1
- package/src/subagent/manager.ts +304 -44
- package/src/subagent/progress-events.ts +35 -0
- package/src/subagent/role-resolution.ts +88 -0
- package/src/subagent/types.ts +302 -67
- package/src/subagent/validate-allowlists.ts +3 -4
- package/src/telegram/__tests__/webhook-health.test.ts +73 -1
- package/src/telegram/webhook-health.ts +101 -3
- package/src/telemetry/__tests__/telemetry-event-fixtures.ts +2 -0
- package/src/telemetry/activation-funnel.ts +6 -1
- package/src/telemetry/live-voice-funnel.ts +75 -0
- package/src/telemetry/telemetry-event-sources.ts +6 -0
- package/src/telemetry/telemetry-wire-source.json +1 -1
- package/src/telemetry/telemetry-wire.generated.ts +2 -0
- package/src/telemetry/types.ts +23 -4
- package/src/telemetry/usage-telemetry-reporter.test.ts +4 -0
- package/src/tools/__tests__/subagent-workflow-tool-input-schemas.test.ts +1 -1
- package/src/tools/ask-question/ask-question-tool.test.ts +118 -3
- package/src/tools/ask-question/ask-question-tool.ts +56 -2
- package/src/tools/browser/browser-execution.ts +1 -1
- package/src/tools/calls/call-start.ts +8 -5
- package/src/tools/conversation-groups/move_to_group.ts +10 -12
- package/src/tools/document/document-tool.ts +11 -4
- package/src/tools/execution-timeout.ts +9 -2
- package/src/tools/executor.ts +7 -6
- package/src/tools/filesystem/edit.ts +1 -1
- package/src/tools/filesystem/list.ts +1 -1
- package/src/tools/filesystem/read.ts +7 -13
- package/src/tools/filesystem/search.ts +23 -21
- package/src/tools/filesystem/write.ts +4 -1
- package/src/tools/host-filesystem/edit.ts +7 -2
- package/src/tools/host-filesystem/read.ts +21 -8
- package/src/tools/host-filesystem/write.ts +10 -2
- package/src/tools/host-terminal/host-shell.ts +2 -2
- package/src/tools/mcp/mcp-tool-factory.ts +25 -1
- package/src/tools/permission-checker.ts +8 -3
- package/src/tools/schedule/delete.ts +13 -1
- package/src/tools/schedule/list.ts +14 -3
- package/src/tools/schedule/update.ts +56 -33
- package/src/tools/shared/filesystem/audio-read.ts +6 -4
- package/src/tools/shared/filesystem/file-ops-service.ts +176 -116
- package/src/tools/shared/filesystem/image-read.ts +12 -12
- package/src/tools/shared/filesystem/path-policy.ts +70 -5
- package/src/tools/shared/filesystem/size-guard.ts +6 -6
- package/src/tools/shared/input-misuse.ts +72 -0
- package/src/tools/skills/scaffold-managed.ts +82 -0
- package/src/tools/skills/skill-tool-factory.ts +14 -1
- package/src/tools/subagent/consult-deadline.ts +12 -7
- package/src/tools/subagent/message.ts +3 -1
- package/src/tools/subagent/read.ts +66 -4
- package/src/tools/subagent/spawn.ts +500 -78
- package/src/tools/tool-approval-handler.ts +45 -4
- package/src/tools/types.ts +22 -3
- package/src/tools/ui-surface/surface-shape-docs.ts +1 -1
- package/src/tools/ui-surface/visual-validation.ts +2 -0
- package/src/usage/__tests__/subagent-attribution.test.ts +109 -0
- package/src/usage/subagent-attribution.ts +108 -0
- package/src/util/__tests__/sqlite-retry.test.ts +112 -0
- package/src/util/abort-reasons.ts +16 -0
- package/src/util/ansi.ts +44 -0
- package/src/util/disk-usage.ts +17 -13
- package/src/util/image-conversion.ts +20 -13
- package/src/util/schedule-source-key.ts +23 -0
- package/src/util/worker-process.ts +7 -1
- package/src/workspace/git-service.ts +33 -9
- package/src/workspace/migrations/140-repair-seed-pinned-memory-v3-live.ts +260 -0
- package/src/workspace/migrations/141-stt-english-default-to-multilingual.ts +111 -0
- package/src/workspace/migrations/__tests__/141-stt-english-default-to-multilingual.test.ts +126 -0
- package/src/workspace/migrations/registry.ts +4 -0
- package/src/workspace/top-level-renderer.ts +15 -2
- package/src/__tests__/tool-side-effects-slack-dm.test.ts +0 -357
- package/src/daemon/install-symlink.ts +0 -207
- package/src/notifications/deferred-emit.ts +0 -147
|
@@ -4,24 +4,42 @@ import { describe, expect, mock, test } from "bun:test";
|
|
|
4
4
|
|
|
5
5
|
import { setConfig } from "./helpers/set-config.js";
|
|
6
6
|
|
|
7
|
-
// Seed the
|
|
8
|
-
// drives the "profile is disabled" spawn error,
|
|
9
|
-
// consult's default `advisorProfile
|
|
7
|
+
// Seed the non-catalog inference profiles these tests exercise: `disabled`
|
|
8
|
+
// drives the "profile is disabled" spawn error, `frontier` is the advisor
|
|
9
|
+
// consult's default `advisorProfile`, and the two model-pinned profiles stand
|
|
10
|
+
// on either side of the catalog's tool-use verdict (`no-tool-model` is a
|
|
11
|
+
// catalog model declared `supportsToolUse: false`, `byok-unknown-model` is a
|
|
12
|
+
// model the catalog has never heard of). The catalog profiles (balanced,
|
|
10
13
|
// cost-optimized, quality-optimized) always resolve through the code catalog,
|
|
11
14
|
// so they need no seeding.
|
|
12
|
-
|
|
15
|
+
// Exported as a constant so a test that needs an extra `llm` key (a call-site
|
|
16
|
+
// pin, say) can re-seed the whole block and restore this baseline afterwards.
|
|
17
|
+
const BASE_LLM_CONFIG = {
|
|
13
18
|
profiles: {
|
|
14
19
|
disabled: { status: "disabled" },
|
|
15
20
|
frontier: {},
|
|
21
|
+
"no-tool-model": {
|
|
22
|
+
provider: "openrouter",
|
|
23
|
+
model: "minimax/minimax-01",
|
|
24
|
+
},
|
|
25
|
+
"byok-unknown-model": {
|
|
26
|
+
provider: "openrouter",
|
|
27
|
+
model: "acme/private-llm-9",
|
|
28
|
+
},
|
|
16
29
|
},
|
|
17
30
|
advisorProfile: "frontier",
|
|
18
|
-
}
|
|
31
|
+
};
|
|
32
|
+
setConfig("llm", BASE_LLM_CONFIG);
|
|
19
33
|
|
|
20
34
|
// Mock conversation-crud before importing tool executors that depend on it.
|
|
21
35
|
let mockGetMessages: (
|
|
22
36
|
conversationId: string,
|
|
23
37
|
) => Array<{ role: string; content: unknown }> | null = () => null;
|
|
24
38
|
|
|
39
|
+
// The profile pinned on the parent conversation, as the spawn tool's
|
|
40
|
+
// inheritance rung reads it.
|
|
41
|
+
let mockConversationOverrideProfile: string | undefined = undefined;
|
|
42
|
+
|
|
25
43
|
// Mock the conversation registry so the advisor consult can resolve a fake
|
|
26
44
|
// parent conversation (snapshot messages + system prompt) without a live
|
|
27
45
|
// Conversation. Other executors in this suite never call `findConversation`.
|
|
@@ -60,6 +78,7 @@ mock.module("../persistence/conversation-crud.js", () => ({
|
|
|
60
78
|
getConversationOriginInterface: () => null,
|
|
61
79
|
getConversationOriginChannel: () => null,
|
|
62
80
|
getMessages: (conversationId: string) => mockGetMessages(conversationId),
|
|
81
|
+
getConversationOverrideProfile: () => mockConversationOverrideProfile,
|
|
63
82
|
createConversation: () => ({ id: "mock-conv" }),
|
|
64
83
|
reserveMessage: mock(async () => ({ id: "msg-reserve" })),
|
|
65
84
|
}));
|
|
@@ -73,8 +92,16 @@ import {
|
|
|
73
92
|
upsertSubagentRecord,
|
|
74
93
|
} from "../persistence/subagent-store.js";
|
|
75
94
|
import { getSubagentManager } from "../subagent/index.js";
|
|
76
|
-
import {
|
|
77
|
-
|
|
95
|
+
import {
|
|
96
|
+
buildSubagentSystemPrompt,
|
|
97
|
+
SubagentAbortedError,
|
|
98
|
+
SubagentManager,
|
|
99
|
+
} from "../subagent/manager.js";
|
|
100
|
+
import {
|
|
101
|
+
SUBAGENT_READ_STILL_PROCESSING,
|
|
102
|
+
SUBAGENT_ROLE_REGISTRY,
|
|
103
|
+
type SubagentState,
|
|
104
|
+
} from "../subagent/types.js";
|
|
78
105
|
import { executeSubagentAbort } from "../tools/subagent/abort.js";
|
|
79
106
|
import { executeSubagentMessage } from "../tools/subagent/message.js";
|
|
80
107
|
import { executeSubagentRead } from "../tools/subagent/read.js";
|
|
@@ -102,13 +129,16 @@ const findTool = (name: string) =>
|
|
|
102
129
|
/**
|
|
103
130
|
* Inject a fake subagent into the singleton manager so tool executors
|
|
104
131
|
* can find it. Uses the same private-internals trick as the notify tests.
|
|
132
|
+
*
|
|
133
|
+
* `rehydrated` is a manager-entry property rather than a state field, so it is
|
|
134
|
+
* passed alongside the state overrides and applied to the entry.
|
|
105
135
|
*/
|
|
106
136
|
function injectSubagent(
|
|
107
137
|
manager: SubagentManager,
|
|
108
138
|
subagentId: string,
|
|
109
139
|
parentConversationId: string,
|
|
110
140
|
status: SubagentState["status"] = "running",
|
|
111
|
-
overrides: Partial<SubagentState> = {},
|
|
141
|
+
overrides: Partial<SubagentState> & { rehydrated?: boolean } = {},
|
|
112
142
|
): SubagentState {
|
|
113
143
|
const internals = manager as unknown as {
|
|
114
144
|
subagents: Map<
|
|
@@ -117,11 +147,13 @@ function injectSubagent(
|
|
|
117
147
|
conversation: unknown;
|
|
118
148
|
state: SubagentState;
|
|
119
149
|
parentSendToClient: () => void;
|
|
150
|
+
rehydrated?: boolean;
|
|
120
151
|
}
|
|
121
152
|
>;
|
|
122
153
|
parentToChildren: Map<string, Set<string>>;
|
|
123
154
|
labelIndex: Map<string, string>;
|
|
124
155
|
};
|
|
156
|
+
const { rehydrated, ...stateOverrides } = overrides;
|
|
125
157
|
const state: SubagentState = {
|
|
126
158
|
config: {
|
|
127
159
|
id: subagentId,
|
|
@@ -134,7 +166,7 @@ function injectSubagent(
|
|
|
134
166
|
isFork: false,
|
|
135
167
|
createdAt: Date.now(),
|
|
136
168
|
usage: { inputTokens: 0, outputTokens: 0, estimatedCost: 0 },
|
|
137
|
-
...
|
|
169
|
+
...stateOverrides,
|
|
138
170
|
};
|
|
139
171
|
const fakeConversation = {
|
|
140
172
|
abort: () => {},
|
|
@@ -145,11 +177,48 @@ function injectSubagent(
|
|
|
145
177
|
enqueueMessage: () => ({ queued: false }),
|
|
146
178
|
persistUserMessage: async () => ({ id: "msg-1", deduplicated: false }),
|
|
147
179
|
runAgentLoop: async () => {},
|
|
180
|
+
// The live counters a retained conversation keeps, seeded to agree with
|
|
181
|
+
// any injected `stats`: in production that field is only ever a reading of
|
|
182
|
+
// THIS conversation's counters, and readers re-read them while the
|
|
183
|
+
// conversation is around.
|
|
184
|
+
subagentToolStats: {
|
|
185
|
+
calls: state.stats?.calls ?? 0,
|
|
186
|
+
succeeded: state.stats?.succeeded ?? 0,
|
|
187
|
+
filesWritten: new Set(
|
|
188
|
+
Array.from(
|
|
189
|
+
{ length: state.stats?.filesWritten ?? 0 },
|
|
190
|
+
(_unused, i) => `/written-${i}.ts`,
|
|
191
|
+
),
|
|
192
|
+
),
|
|
193
|
+
},
|
|
194
|
+
// Drain state, as the queued-turn settle wait observes it. Idle here, so
|
|
195
|
+
// an injected subagent reads as having nothing left to run; a test that
|
|
196
|
+
// wants a follow-up turn in flight drives it through `queuedFollowUpTurn`.
|
|
197
|
+
processing: false,
|
|
198
|
+
queueDepth: 0,
|
|
199
|
+
isProcessing(): boolean {
|
|
200
|
+
return this.processing;
|
|
201
|
+
},
|
|
202
|
+
hasQueuedMessages(): boolean {
|
|
203
|
+
return this.queueDepth > 0;
|
|
204
|
+
},
|
|
205
|
+
async waitForIdle({ timeoutMs }: { timeoutMs: number }): Promise<boolean> {
|
|
206
|
+
// Resolve in slices rather than sitting on the caller's whole budget, so
|
|
207
|
+
// the settle loop re-observes and a turn a test ends on a timer is
|
|
208
|
+
// picked up promptly.
|
|
209
|
+
await new Promise((resolve) =>
|
|
210
|
+
setTimeout(resolve, Math.min(timeoutMs, 5)),
|
|
211
|
+
);
|
|
212
|
+
return !this.processing;
|
|
213
|
+
},
|
|
148
214
|
};
|
|
215
|
+
// A rehydrated entry is metadata rebuilt from the durable row, so it never
|
|
216
|
+
// has a live conversation behind it.
|
|
149
217
|
internals.subagents.set(subagentId, {
|
|
150
|
-
conversation: fakeConversation,
|
|
218
|
+
conversation: rehydrated ? null : fakeConversation,
|
|
151
219
|
state,
|
|
152
220
|
parentSendToClient: () => {},
|
|
221
|
+
...(rehydrated ? { rehydrated: true } : {}),
|
|
153
222
|
});
|
|
154
223
|
if (!internals.parentToChildren.has(parentConversationId)) {
|
|
155
224
|
internals.parentToChildren.set(parentConversationId, new Set());
|
|
@@ -166,6 +235,69 @@ function injectSubagent(
|
|
|
166
235
|
return state;
|
|
167
236
|
}
|
|
168
237
|
|
|
238
|
+
/**
|
|
239
|
+
* The live tool-call counters behind an injected subagent's fake conversation,
|
|
240
|
+
* so a test can move them the way a queued follow-up turn does after the run's
|
|
241
|
+
* own harvest.
|
|
242
|
+
*/
|
|
243
|
+
function liveToolStats(
|
|
244
|
+
manager: SubagentManager,
|
|
245
|
+
subagentId: string,
|
|
246
|
+
): { calls: number; succeeded: number; filesWritten: Set<string> } {
|
|
247
|
+
const internals = manager as unknown as {
|
|
248
|
+
subagents: Map<
|
|
249
|
+
string,
|
|
250
|
+
{
|
|
251
|
+
conversation: {
|
|
252
|
+
subagentToolStats: {
|
|
253
|
+
calls: number;
|
|
254
|
+
succeeded: number;
|
|
255
|
+
filesWritten: Set<string>;
|
|
256
|
+
};
|
|
257
|
+
} | null;
|
|
258
|
+
}
|
|
259
|
+
>;
|
|
260
|
+
};
|
|
261
|
+
return internals.subagents.get(subagentId)!.conversation!.subagentToolStats;
|
|
262
|
+
}
|
|
263
|
+
|
|
264
|
+
/** The drain state a test drives to stand in for a queued follow-up turn. */
|
|
265
|
+
interface QueuedTurnDrainState {
|
|
266
|
+
/** Messages waiting in the child's queue. */
|
|
267
|
+
queueDepth: number;
|
|
268
|
+
/** Whether the child is mid-turn. */
|
|
269
|
+
processing: boolean;
|
|
270
|
+
}
|
|
271
|
+
|
|
272
|
+
/**
|
|
273
|
+
* Put an injected subagent into the window that opens when guidance is queued
|
|
274
|
+
* during its run: the subagent is terminal because its own run returned, but
|
|
275
|
+
* the queued turn is still ahead of it on a conversation the manager retains.
|
|
276
|
+
*
|
|
277
|
+
* Returns the drain state so the test can move the turn through it. Queued and
|
|
278
|
+
* not yet dispatched to start with, which is where the drain sits at the
|
|
279
|
+
* moment the parent is told to read.
|
|
280
|
+
*/
|
|
281
|
+
function queuedFollowUpTurn(
|
|
282
|
+
manager: SubagentManager,
|
|
283
|
+
subagentId: string,
|
|
284
|
+
): QueuedTurnDrainState {
|
|
285
|
+
const internals = manager as unknown as {
|
|
286
|
+
subagents: Map<
|
|
287
|
+
string,
|
|
288
|
+
{
|
|
289
|
+
conversation: QueuedTurnDrainState | null;
|
|
290
|
+
hadEnqueuedMessages?: boolean;
|
|
291
|
+
}
|
|
292
|
+
>;
|
|
293
|
+
};
|
|
294
|
+
const managed = internals.subagents.get(subagentId)!;
|
|
295
|
+
managed.hadEnqueuedMessages = true;
|
|
296
|
+
const drain = managed.conversation!;
|
|
297
|
+
drain.queueDepth = 1;
|
|
298
|
+
return drain;
|
|
299
|
+
}
|
|
300
|
+
|
|
169
301
|
function makeContext(
|
|
170
302
|
conversationId: string,
|
|
171
303
|
extras: Record<string, unknown> = {},
|
|
@@ -186,6 +318,7 @@ describe("Subagent tool definitions", () => {
|
|
|
186
318
|
expect(def).toBeDefined();
|
|
187
319
|
expect(def.input_schema.required).toEqual(["label", "objective"]);
|
|
188
320
|
expect(def.input_schema.properties.inference_profile).toBeDefined();
|
|
321
|
+
expect(def.input_schema.properties.confirm_repeat.type).toBe("boolean");
|
|
189
322
|
});
|
|
190
323
|
|
|
191
324
|
test("abort tool has correct definition", () => {
|
|
@@ -207,6 +340,9 @@ describe("Subagent tool definitions", () => {
|
|
|
207
340
|
expect(def).toBeDefined();
|
|
208
341
|
expect(def.input_schema.required).toEqual([]);
|
|
209
342
|
expect(def.input_schema.properties.label).toBeDefined();
|
|
343
|
+
expect(def.input_schema.properties.last_n.type).toBe("integer");
|
|
344
|
+
expect(def.description).toContain("NOT a file reader");
|
|
345
|
+
expect(def.description).toContain("file_read");
|
|
210
346
|
});
|
|
211
347
|
|
|
212
348
|
test("status tool has correct definition", () => {
|
|
@@ -500,7 +636,7 @@ describe("Subagent spawn success and failure", () => {
|
|
|
500
636
|
}
|
|
501
637
|
});
|
|
502
638
|
|
|
503
|
-
test("spawn
|
|
639
|
+
test("spawn passes no override, landing the child on the subagentSpawn default", async () => {
|
|
504
640
|
const manager = getSubagentManager();
|
|
505
641
|
const originalSpawn = manager.spawn.bind(manager);
|
|
506
642
|
let capturedConfig: Record<string, unknown> | undefined;
|
|
@@ -520,17 +656,17 @@ describe("Subagent spawn success and failure", () => {
|
|
|
520
656
|
);
|
|
521
657
|
|
|
522
658
|
expect(result.isError).toBe(false);
|
|
523
|
-
// No
|
|
524
|
-
//
|
|
525
|
-
//
|
|
526
|
-
expect(capturedConfig!.overrideProfile).
|
|
659
|
+
// No override travels with the spawn. The child runs its loop under
|
|
660
|
+
// `callSite: "subagentSpawn"` and resolves that call site's own profile,
|
|
661
|
+
// which also keeps its usage attribution off `conversation`.
|
|
662
|
+
expect(capturedConfig!.overrideProfile).toBeUndefined();
|
|
527
663
|
expect(capturedConfig!.forceOverrideProfile).toBeUndefined();
|
|
528
664
|
} finally {
|
|
529
665
|
manager.spawn = originalSpawn;
|
|
530
666
|
}
|
|
531
667
|
});
|
|
532
668
|
|
|
533
|
-
test("
|
|
669
|
+
test("a non-main invoker's call-site default does not reach the child", async () => {
|
|
534
670
|
const manager = getSubagentManager();
|
|
535
671
|
const originalSpawn = manager.spawn.bind(manager);
|
|
536
672
|
let capturedConfig: Record<string, unknown> | undefined;
|
|
@@ -550,15 +686,16 @@ describe("Subagent spawn success and failure", () => {
|
|
|
550
686
|
);
|
|
551
687
|
|
|
552
688
|
expect(result.isError).toBe(false);
|
|
553
|
-
// A subagent spawned from a heartbeat turn
|
|
554
|
-
// cost-optimized default
|
|
555
|
-
|
|
689
|
+
// A subagent spawned from a heartbeat turn does not pick up
|
|
690
|
+
// heartbeatAgent's cost-optimized default. Delegated work is priced by
|
|
691
|
+
// where it runs, not by which call site happened to delegate it.
|
|
692
|
+
expect(capturedConfig!.overrideProfile).toBeUndefined();
|
|
556
693
|
} finally {
|
|
557
694
|
manager.spawn = originalSpawn;
|
|
558
695
|
}
|
|
559
696
|
});
|
|
560
697
|
|
|
561
|
-
test("
|
|
698
|
+
test("a per-turn override profile does not reach the child", async () => {
|
|
562
699
|
const manager = getSubagentManager();
|
|
563
700
|
const originalSpawn = manager.spawn.bind(manager);
|
|
564
701
|
let capturedConfig: Record<string, unknown> | undefined;
|
|
@@ -579,9 +716,10 @@ describe("Subagent spawn success and failure", () => {
|
|
|
579
716
|
);
|
|
580
717
|
|
|
581
718
|
expect(result.isError).toBe(false);
|
|
582
|
-
//
|
|
583
|
-
// the
|
|
584
|
-
|
|
719
|
+
// A profile switched mid-conversation is a choice about that
|
|
720
|
+
// conversation, not about the work it delegates, so it stops at the
|
|
721
|
+
// spawn boundary.
|
|
722
|
+
expect(capturedConfig!.overrideProfile).toBeUndefined();
|
|
585
723
|
expect(capturedConfig!.forceOverrideProfile).toBeUndefined();
|
|
586
724
|
} finally {
|
|
587
725
|
manager.spawn = originalSpawn;
|
|
@@ -702,6 +840,454 @@ describe("Subagent spawn success and failure", () => {
|
|
|
702
840
|
});
|
|
703
841
|
});
|
|
704
842
|
|
|
843
|
+
// ── Profile isolation ───────────────────────────────────────────────
|
|
844
|
+
|
|
845
|
+
describe("Subagent spawn profile isolation", () => {
|
|
846
|
+
/** Capture the config `executeSubagentSpawn` hands the manager. */
|
|
847
|
+
async function spawnCapturingConfig(
|
|
848
|
+
input: Record<string, unknown>,
|
|
849
|
+
contextExtras: Record<string, unknown> = {},
|
|
850
|
+
): Promise<{
|
|
851
|
+
result: { content: string; isError: boolean };
|
|
852
|
+
config: Record<string, unknown>;
|
|
853
|
+
}> {
|
|
854
|
+
const manager = getSubagentManager();
|
|
855
|
+
const originalSpawn = manager.spawn.bind(manager);
|
|
856
|
+
let capturedConfig: Record<string, unknown> = {};
|
|
857
|
+
manager.spawn = async (config: Record<string, unknown>) => {
|
|
858
|
+
capturedConfig = config;
|
|
859
|
+
return "isolation-subagent-id";
|
|
860
|
+
};
|
|
861
|
+
try {
|
|
862
|
+
const result = await executeSubagentSpawn(
|
|
863
|
+
input,
|
|
864
|
+
makeContext("sess-isolation", {
|
|
865
|
+
sendToClient: () => {},
|
|
866
|
+
...contextExtras,
|
|
867
|
+
}),
|
|
868
|
+
);
|
|
869
|
+
return { result, config: capturedConfig };
|
|
870
|
+
} finally {
|
|
871
|
+
mockConversationOverrideProfile = undefined;
|
|
872
|
+
manager.spawn = originalSpawn;
|
|
873
|
+
}
|
|
874
|
+
}
|
|
875
|
+
|
|
876
|
+
test("keeps the conversation-pinned profile away from the child", async () => {
|
|
877
|
+
mockConversationOverrideProfile = "quality-optimized";
|
|
878
|
+
const { config } = await spawnCapturingConfig({
|
|
879
|
+
label: "Pinned parent",
|
|
880
|
+
objective: "Do it",
|
|
881
|
+
});
|
|
882
|
+
// No override travels with the spawn, which is what lands the child on
|
|
883
|
+
// the subagentSpawn call site's own profile while leaving usage
|
|
884
|
+
// attribution on `call_site` instead of reporting a pin nobody set.
|
|
885
|
+
expect(config.overrideProfile).toBeUndefined();
|
|
886
|
+
expect(config.forceOverrideProfile).toBeUndefined();
|
|
887
|
+
});
|
|
888
|
+
|
|
889
|
+
test("keeps the per-turn override profile away from the child", async () => {
|
|
890
|
+
const { config } = await spawnCapturingConfig(
|
|
891
|
+
{ label: "Override parent", objective: "Do it" },
|
|
892
|
+
{ invokingCallSite: "mainAgent", overrideProfile: "quality-optimized" },
|
|
893
|
+
);
|
|
894
|
+
expect(config.overrideProfile).toBeUndefined();
|
|
895
|
+
});
|
|
896
|
+
|
|
897
|
+
test("still honors an explicit inference_profile", async () => {
|
|
898
|
+
mockConversationOverrideProfile = "cost-optimized";
|
|
899
|
+
const { result, config } = await spawnCapturingConfig({
|
|
900
|
+
label: "Explicit",
|
|
901
|
+
objective: "Do it",
|
|
902
|
+
inference_profile: "quality-optimized",
|
|
903
|
+
});
|
|
904
|
+
expect(config.overrideProfile).toBe("quality-optimized");
|
|
905
|
+
expect(config.forceOverrideProfile).toBe(true);
|
|
906
|
+
expect(JSON.parse(result.content).note).toBeUndefined();
|
|
907
|
+
});
|
|
908
|
+
|
|
909
|
+
test("falls back with a note when the catalog denies tool use", async () => {
|
|
910
|
+
const { result, config } = await spawnCapturingConfig({
|
|
911
|
+
label: "No tools",
|
|
912
|
+
objective: "Do it",
|
|
913
|
+
inference_profile: "no-tool-model",
|
|
914
|
+
});
|
|
915
|
+
// Dropping the override is what redirects the child to the call site's
|
|
916
|
+
// profile; naming it explicitly would re-file the spend as a pin.
|
|
917
|
+
expect(config.overrideProfile).toBeUndefined();
|
|
918
|
+
expect(config.forceOverrideProfile).toBeUndefined();
|
|
919
|
+
const parsed = JSON.parse(result.content);
|
|
920
|
+
expect(parsed.subagentId).toBe("isolation-subagent-id");
|
|
921
|
+
expect(parsed.status).toBe("pending");
|
|
922
|
+
expect(parsed.note).toBe(
|
|
923
|
+
'requested profile "no-tool-model" is not verified for tool calling; ran on the default profile instead.',
|
|
924
|
+
);
|
|
925
|
+
});
|
|
926
|
+
|
|
927
|
+
test("fails open for a model the catalog does not list", async () => {
|
|
928
|
+
const { result, config } = await spawnCapturingConfig({
|
|
929
|
+
label: "BYOK",
|
|
930
|
+
objective: "Do it",
|
|
931
|
+
inference_profile: "byok-unknown-model",
|
|
932
|
+
});
|
|
933
|
+
expect(config.overrideProfile).toBe("byok-unknown-model");
|
|
934
|
+
expect(config.forceOverrideProfile).toBe(true);
|
|
935
|
+
expect(JSON.parse(result.content).note).toBeUndefined();
|
|
936
|
+
});
|
|
937
|
+
});
|
|
938
|
+
|
|
939
|
+
// ── Repeat-spawn guard ──────────────────────────────────────────────
|
|
940
|
+
|
|
941
|
+
describe("Subagent spawn repeat-loop guard", () => {
|
|
942
|
+
const guardParent = "guard-parent";
|
|
943
|
+
|
|
944
|
+
/** Seed a finished spawn of `objective`, the way the manager records one. */
|
|
945
|
+
function seedSpawn(
|
|
946
|
+
id: string,
|
|
947
|
+
objective: string,
|
|
948
|
+
over: Partial<SubagentRecord> = {},
|
|
949
|
+
): void {
|
|
950
|
+
upsertSubagentRecord({
|
|
951
|
+
id,
|
|
952
|
+
parentConversationId: guardParent,
|
|
953
|
+
conversationId: `conv-${id}`,
|
|
954
|
+
label: id,
|
|
955
|
+
objective,
|
|
956
|
+
role: "builder",
|
|
957
|
+
isFork: false,
|
|
958
|
+
sendResultToUser: true,
|
|
959
|
+
parentToolUseId: null,
|
|
960
|
+
status: "completed",
|
|
961
|
+
error: null,
|
|
962
|
+
createdAt: Date.now(),
|
|
963
|
+
startedAt: null,
|
|
964
|
+
completedAt: Date.now(),
|
|
965
|
+
inputTokens: 10,
|
|
966
|
+
outputTokens: 20,
|
|
967
|
+
estimatedCost: 1.25,
|
|
968
|
+
...over,
|
|
969
|
+
});
|
|
970
|
+
}
|
|
971
|
+
|
|
972
|
+
/** Seed `count` finished spawns of one objective. */
|
|
973
|
+
function seedSpawns(
|
|
974
|
+
idPrefix: string,
|
|
975
|
+
objective: string,
|
|
976
|
+
count: number,
|
|
977
|
+
): void {
|
|
978
|
+
for (let i = 0; i < count; i++) {
|
|
979
|
+
seedSpawn(`${idPrefix}-${i}`, objective);
|
|
980
|
+
}
|
|
981
|
+
}
|
|
982
|
+
|
|
983
|
+
/**
|
|
984
|
+
* Run the spawn tool, reporting whether the manager was actually asked to
|
|
985
|
+
* spawn.
|
|
986
|
+
*/
|
|
987
|
+
async function spawnWithGuard(
|
|
988
|
+
input: Record<string, unknown>,
|
|
989
|
+
conversationId = guardParent,
|
|
990
|
+
) {
|
|
991
|
+
const manager = getSubagentManager();
|
|
992
|
+
const originalSpawn = manager.spawn.bind(manager);
|
|
993
|
+
let spawned = false;
|
|
994
|
+
manager.spawn = async () => {
|
|
995
|
+
spawned = true;
|
|
996
|
+
return "guard-subagent-id";
|
|
997
|
+
};
|
|
998
|
+
try {
|
|
999
|
+
const result = await executeSubagentSpawn(
|
|
1000
|
+
input,
|
|
1001
|
+
makeContext(conversationId, { sendToClient: () => {} }),
|
|
1002
|
+
);
|
|
1003
|
+
return { result, spawned };
|
|
1004
|
+
} finally {
|
|
1005
|
+
manager.spawn = originalSpawn;
|
|
1006
|
+
}
|
|
1007
|
+
}
|
|
1008
|
+
|
|
1009
|
+
test("a conversation under the threshold spawns normally", async () => {
|
|
1010
|
+
const objective = "Audit the under-threshold pipeline for drift";
|
|
1011
|
+
seedSpawns("guard-under", objective, 2);
|
|
1012
|
+
|
|
1013
|
+
const { spawned } = await spawnWithGuard({
|
|
1014
|
+
label: "Repeat",
|
|
1015
|
+
objective,
|
|
1016
|
+
});
|
|
1017
|
+
|
|
1018
|
+
expect(spawned).toBe(true);
|
|
1019
|
+
});
|
|
1020
|
+
|
|
1021
|
+
test("a fourth repeat in one conversation is held for confirmation", async () => {
|
|
1022
|
+
const objective = "Audit the retention pipeline for drift";
|
|
1023
|
+
seedSpawns("guard-conv", objective, 3);
|
|
1024
|
+
|
|
1025
|
+
const { result, spawned } = await spawnWithGuard({
|
|
1026
|
+
label: "Repeat",
|
|
1027
|
+
objective,
|
|
1028
|
+
});
|
|
1029
|
+
|
|
1030
|
+
expect(spawned).toBe(false);
|
|
1031
|
+
expect(result.isError).toBe(false);
|
|
1032
|
+
expect(result.content).toContain(
|
|
1033
|
+
"3 near-identical subagents already completed in this conversation in the last 24 hours",
|
|
1034
|
+
);
|
|
1035
|
+
expect(result.content).toContain("about $3.75");
|
|
1036
|
+
expect(result.content).toContain("confirm_repeat: true");
|
|
1037
|
+
});
|
|
1038
|
+
|
|
1039
|
+
test("the assistant-wide threshold catches a repeat spread across conversations", async () => {
|
|
1040
|
+
const objective = "Audit the fleet-wide pipeline for drift";
|
|
1041
|
+
for (let i = 0; i < 10; i++) {
|
|
1042
|
+
seedSpawn(`guard-wide-${i}`, objective, {
|
|
1043
|
+
parentConversationId: `guard-wide-parent-${i}`,
|
|
1044
|
+
});
|
|
1045
|
+
}
|
|
1046
|
+
|
|
1047
|
+
const { result, spawned } = await spawnWithGuard(
|
|
1048
|
+
{ label: "Repeat", objective },
|
|
1049
|
+
"guard-fresh-conversation",
|
|
1050
|
+
);
|
|
1051
|
+
|
|
1052
|
+
expect(spawned).toBe(false);
|
|
1053
|
+
expect(result.content).toContain(
|
|
1054
|
+
"10 near-identical subagents already completed across this assistant in the last 24 hours",
|
|
1055
|
+
);
|
|
1056
|
+
});
|
|
1057
|
+
|
|
1058
|
+
test("confirm_repeat spawns past the guard", async () => {
|
|
1059
|
+
const objective = "Audit the confirmed pipeline for drift";
|
|
1060
|
+
seedSpawns("guard-confirm", objective, 5);
|
|
1061
|
+
|
|
1062
|
+
const { result, spawned } = await spawnWithGuard({
|
|
1063
|
+
label: "Repeat",
|
|
1064
|
+
objective,
|
|
1065
|
+
confirm_repeat: true,
|
|
1066
|
+
});
|
|
1067
|
+
|
|
1068
|
+
expect(spawned).toBe(true);
|
|
1069
|
+
expect(JSON.parse(result.content).status).toBe("pending");
|
|
1070
|
+
});
|
|
1071
|
+
|
|
1072
|
+
test("advisor consults are never guarded", async () => {
|
|
1073
|
+
const objective = "Audit the advisor pipeline for drift";
|
|
1074
|
+
seedSpawns("guard-advisor", objective, 6);
|
|
1075
|
+
|
|
1076
|
+
const { result } = await spawnWithGuard({
|
|
1077
|
+
label: "Consult",
|
|
1078
|
+
objective,
|
|
1079
|
+
role: "advisor",
|
|
1080
|
+
});
|
|
1081
|
+
|
|
1082
|
+
// The advisor branch runs and reports its own missing-parent notice, so the
|
|
1083
|
+
// guard never saw the repetition.
|
|
1084
|
+
expect(result.content).toContain("advisor unavailable");
|
|
1085
|
+
expect(result.content).not.toContain("confirm_repeat");
|
|
1086
|
+
});
|
|
1087
|
+
|
|
1088
|
+
test("copies still running trip the guard before any of them completes", async () => {
|
|
1089
|
+
// The runaway shape the guard exists for: a burst issued faster than
|
|
1090
|
+
// anything can finish, which a completed-only count cannot see.
|
|
1091
|
+
const objective = "Audit the in-flight pipeline for drift";
|
|
1092
|
+
seedSpawn("guard-flight-0", objective, { status: "running" });
|
|
1093
|
+
seedSpawn("guard-flight-1", objective, { status: "pending" });
|
|
1094
|
+
|
|
1095
|
+
const { result, spawned } = await spawnWithGuard({
|
|
1096
|
+
label: "Repeat",
|
|
1097
|
+
objective,
|
|
1098
|
+
});
|
|
1099
|
+
|
|
1100
|
+
expect(spawned).toBe(false);
|
|
1101
|
+
expect(result.isError).toBe(false);
|
|
1102
|
+
expect(result.content).toContain(
|
|
1103
|
+
"2 near-identical subagents are already running in this conversation",
|
|
1104
|
+
);
|
|
1105
|
+
expect(result.content).toContain("none of them has returned yet");
|
|
1106
|
+
// Nothing has been produced, so the caller must not be sent reading.
|
|
1107
|
+
expect(result.content).not.toContain("subagent_read");
|
|
1108
|
+
expect(result.content).toContain("confirm_repeat: true");
|
|
1109
|
+
});
|
|
1110
|
+
|
|
1111
|
+
test("a single in-flight copy is not a loop", async () => {
|
|
1112
|
+
const objective = "Audit the single-flight pipeline for drift";
|
|
1113
|
+
seedSpawn("guard-flight-solo", objective, { status: "running" });
|
|
1114
|
+
|
|
1115
|
+
const { spawned } = await spawnWithGuard({
|
|
1116
|
+
label: "Repeat",
|
|
1117
|
+
objective,
|
|
1118
|
+
});
|
|
1119
|
+
|
|
1120
|
+
expect(spawned).toBe(true);
|
|
1121
|
+
});
|
|
1122
|
+
|
|
1123
|
+
test("the assistant-wide in-flight ceiling catches a burst across conversations", async () => {
|
|
1124
|
+
const objective = "Audit the fleet-wide in-flight pipeline for drift";
|
|
1125
|
+
for (let i = 0; i < 4; i++) {
|
|
1126
|
+
seedSpawn(`guard-flight-wide-${i}`, objective, {
|
|
1127
|
+
status: "running",
|
|
1128
|
+
parentConversationId: `guard-flight-parent-${i}`,
|
|
1129
|
+
});
|
|
1130
|
+
}
|
|
1131
|
+
|
|
1132
|
+
const { result, spawned } = await spawnWithGuard(
|
|
1133
|
+
{ label: "Repeat", objective },
|
|
1134
|
+
"guard-flight-fresh-conversation",
|
|
1135
|
+
);
|
|
1136
|
+
|
|
1137
|
+
expect(spawned).toBe(false);
|
|
1138
|
+
expect(result.content).toContain(
|
|
1139
|
+
"4 near-identical subagents are already running in this assistant",
|
|
1140
|
+
);
|
|
1141
|
+
});
|
|
1142
|
+
|
|
1143
|
+
test("completed runs are reported ahead of in-flight ones", async () => {
|
|
1144
|
+
// Both ceilings are met; the answer that already exists is the more
|
|
1145
|
+
// actionable thing to hand back.
|
|
1146
|
+
const objective = "Audit the mixed pipeline for drift";
|
|
1147
|
+
seedSpawns("guard-mixed", objective, 3);
|
|
1148
|
+
seedSpawn("guard-mixed-running-0", objective, { status: "running" });
|
|
1149
|
+
seedSpawn("guard-mixed-running-1", objective, { status: "running" });
|
|
1150
|
+
|
|
1151
|
+
const { result, spawned } = await spawnWithGuard({
|
|
1152
|
+
label: "Repeat",
|
|
1153
|
+
objective,
|
|
1154
|
+
});
|
|
1155
|
+
|
|
1156
|
+
expect(spawned).toBe(false);
|
|
1157
|
+
expect(result.content).toContain(
|
|
1158
|
+
"3 near-identical subagents already completed in this conversation",
|
|
1159
|
+
);
|
|
1160
|
+
expect(result.content).toContain("subagent_read");
|
|
1161
|
+
});
|
|
1162
|
+
|
|
1163
|
+
test("confirm_repeat spawns past the in-flight guard too", async () => {
|
|
1164
|
+
const objective = "Audit the confirmed in-flight pipeline for drift";
|
|
1165
|
+
seedSpawns("guard-flight-confirm", objective, 0);
|
|
1166
|
+
for (let i = 0; i < 3; i++) {
|
|
1167
|
+
seedSpawn(`guard-flight-confirm-${i}`, objective, { status: "running" });
|
|
1168
|
+
}
|
|
1169
|
+
|
|
1170
|
+
const { spawned } = await spawnWithGuard({
|
|
1171
|
+
label: "Repeat",
|
|
1172
|
+
objective,
|
|
1173
|
+
confirm_repeat: true,
|
|
1174
|
+
});
|
|
1175
|
+
|
|
1176
|
+
expect(spawned).toBe(true);
|
|
1177
|
+
});
|
|
1178
|
+
|
|
1179
|
+
test("runs that ended without an answer never trip either ceiling", async () => {
|
|
1180
|
+
const objective = "Audit the failed pipeline for drift";
|
|
1181
|
+
for (const status of ["failed", "aborted", "interrupted"]) {
|
|
1182
|
+
for (let i = 0; i < 4; i++) {
|
|
1183
|
+
seedSpawn(`guard-dead-${status}-${i}`, objective, { status });
|
|
1184
|
+
}
|
|
1185
|
+
}
|
|
1186
|
+
|
|
1187
|
+
const { spawned } = await spawnWithGuard({
|
|
1188
|
+
label: "Retry",
|
|
1189
|
+
objective,
|
|
1190
|
+
});
|
|
1191
|
+
|
|
1192
|
+
expect(spawned).toBe(true);
|
|
1193
|
+
});
|
|
1194
|
+
|
|
1195
|
+
test("case and spacing differences count as the same objective", async () => {
|
|
1196
|
+
const objective = "Audit the normalized pipeline for drift";
|
|
1197
|
+
seedSpawn("guard-norm-0", objective.toUpperCase());
|
|
1198
|
+
seedSpawn("guard-norm-1", ` ${objective} `);
|
|
1199
|
+
seedSpawn("guard-norm-2", objective.replace(/ /gu, "\n"));
|
|
1200
|
+
seedSpawn("guard-norm-3", objective.replace(/ /gu, " "));
|
|
1201
|
+
|
|
1202
|
+
const { result, spawned } = await spawnWithGuard({
|
|
1203
|
+
label: "Repeat",
|
|
1204
|
+
objective,
|
|
1205
|
+
});
|
|
1206
|
+
|
|
1207
|
+
expect(spawned).toBe(false);
|
|
1208
|
+
expect(result.content).toContain("4 near-identical subagents");
|
|
1209
|
+
});
|
|
1210
|
+
|
|
1211
|
+
test("a genuinely different objective is not a repeat", async () => {
|
|
1212
|
+
const objective = "Audit the distinct pipeline for drift";
|
|
1213
|
+
seedSpawns("guard-distinct", objective, 5);
|
|
1214
|
+
|
|
1215
|
+
const { spawned } = await spawnWithGuard({
|
|
1216
|
+
label: "Different",
|
|
1217
|
+
objective: "Audit the payouts ledger for drift",
|
|
1218
|
+
});
|
|
1219
|
+
|
|
1220
|
+
expect(spawned).toBe(true);
|
|
1221
|
+
});
|
|
1222
|
+
|
|
1223
|
+
test("a retry after runs that produced no answer spawns normally", async () => {
|
|
1224
|
+
const objective = "Audit the flaky pipeline for drift";
|
|
1225
|
+
const unfinished = ["failed", "aborted", "interrupted"];
|
|
1226
|
+
unfinished.forEach((status, i) => {
|
|
1227
|
+
seedSpawn(`guard-retry-${i}`, objective, { status });
|
|
1228
|
+
});
|
|
1229
|
+
// Well past the assistant-wide limit too, so neither scope may count them.
|
|
1230
|
+
for (let i = 0; i < 12; i++) {
|
|
1231
|
+
seedSpawn(`guard-retry-wide-${i}`, objective, {
|
|
1232
|
+
parentConversationId: `guard-retry-parent-${i}`,
|
|
1233
|
+
status: unfinished[i % unfinished.length],
|
|
1234
|
+
});
|
|
1235
|
+
}
|
|
1236
|
+
|
|
1237
|
+
const { spawned } = await spawnWithGuard({
|
|
1238
|
+
label: "Retry",
|
|
1239
|
+
objective,
|
|
1240
|
+
});
|
|
1241
|
+
|
|
1242
|
+
expect(spawned).toBe(true);
|
|
1243
|
+
});
|
|
1244
|
+
|
|
1245
|
+
test("runs still in flight do not count as answers already produced", async () => {
|
|
1246
|
+
// Two answers plus one copy still executing. Each ceiling is judged on its
|
|
1247
|
+
// own tally, so neither is reached; folding them into one count would hold
|
|
1248
|
+
// this spawn on work that has produced nothing.
|
|
1249
|
+
const objective = "Audit the mixed-tally pipeline for drift";
|
|
1250
|
+
seedSpawns("guard-inflight-done", objective, 2);
|
|
1251
|
+
seedSpawn("guard-inflight-0", objective, { status: "awaiting_input" });
|
|
1252
|
+
|
|
1253
|
+
const { spawned } = await spawnWithGuard({
|
|
1254
|
+
label: "Repeat",
|
|
1255
|
+
objective,
|
|
1256
|
+
});
|
|
1257
|
+
|
|
1258
|
+
expect(spawned).toBe(true);
|
|
1259
|
+
});
|
|
1260
|
+
|
|
1261
|
+
test("objectives sharing a boilerplate prefix are distinct tasks", async () => {
|
|
1262
|
+
const preamble =
|
|
1263
|
+
"Review the module against every item in the team checklist, then write up " +
|
|
1264
|
+
"what you found and what should change, focusing on ";
|
|
1265
|
+
seedSpawns("guard-batch", `${preamble}the billing service`, 5);
|
|
1266
|
+
|
|
1267
|
+
const { spawned } = await spawnWithGuard({
|
|
1268
|
+
label: "Next in batch",
|
|
1269
|
+
objective: `${preamble}the payouts service`,
|
|
1270
|
+
});
|
|
1271
|
+
|
|
1272
|
+
expect(spawned).toBe(true);
|
|
1273
|
+
});
|
|
1274
|
+
|
|
1275
|
+
test("the same long objective is still caught however long its preamble", async () => {
|
|
1276
|
+
const objective =
|
|
1277
|
+
"Review the module against every item in the team checklist, then write up " +
|
|
1278
|
+
"what you found and what should change, focusing on the ledger service";
|
|
1279
|
+
seedSpawns("guard-long", objective, 3);
|
|
1280
|
+
|
|
1281
|
+
const { result, spawned } = await spawnWithGuard({
|
|
1282
|
+
label: "Repeat",
|
|
1283
|
+
objective,
|
|
1284
|
+
});
|
|
1285
|
+
|
|
1286
|
+
expect(spawned).toBe(false);
|
|
1287
|
+
expect(result.content).toContain("3 near-identical subagents");
|
|
1288
|
+
});
|
|
1289
|
+
});
|
|
1290
|
+
|
|
705
1291
|
// ── Message success path ────────────────────────────────────────────
|
|
706
1292
|
|
|
707
1293
|
describe("Subagent message success path", () => {
|
|
@@ -825,7 +1411,13 @@ describe("Subagent read tool", () => {
|
|
|
825
1411
|
);
|
|
826
1412
|
expect(result.isError).toBe(false);
|
|
827
1413
|
expect(result.content).toContain("still running");
|
|
828
|
-
expect(result.content).toContain("
|
|
1414
|
+
expect(result.content).toContain("Do not poll");
|
|
1415
|
+
expect(result.content).toContain(
|
|
1416
|
+
"you will be notified automatically when it completes",
|
|
1417
|
+
);
|
|
1418
|
+
// A deferred run is announced with a read pointer rather than an inlined
|
|
1419
|
+
// result, so the wait message must not promise the result itself.
|
|
1420
|
+
expect(result.content).not.toContain("including its result");
|
|
829
1421
|
});
|
|
830
1422
|
|
|
831
1423
|
test("read returns wait message for pending subagent", async () => {
|
|
@@ -1347,44 +1939,346 @@ describe("Subagent read tool", () => {
|
|
|
1347
1939
|
});
|
|
1348
1940
|
});
|
|
1349
1941
|
|
|
1350
|
-
// ──
|
|
1942
|
+
// ── Read stats footer (machine truth envelope) ──────────────────────
|
|
1351
1943
|
|
|
1352
|
-
describe("Subagent
|
|
1353
|
-
|
|
1944
|
+
describe("Subagent read stats footer", () => {
|
|
1945
|
+
const ownerConversation = "read-stats-owner";
|
|
1946
|
+
|
|
1947
|
+
function stubOutput(subagentId: string, text: string) {
|
|
1948
|
+
mockGetMessages = (convId: string) =>
|
|
1949
|
+
convId === `conv-${subagentId}`
|
|
1950
|
+
? [{ role: "assistant", content: [{ type: "text", text }] }]
|
|
1951
|
+
: null;
|
|
1952
|
+
}
|
|
1953
|
+
|
|
1954
|
+
test("reports what the subagent actually ran alongside its output", async () => {
|
|
1354
1955
|
const manager = getSubagentManager();
|
|
1355
|
-
const subagentId = "
|
|
1356
|
-
injectSubagent(manager, subagentId, "
|
|
1956
|
+
const subagentId = "read-stats-1";
|
|
1957
|
+
injectSubagent(manager, subagentId, ownerConversation, "completed", {
|
|
1958
|
+
stats: { calls: 5, succeeded: 4, filesWritten: 2 },
|
|
1959
|
+
});
|
|
1960
|
+
stubOutput(subagentId, "Refactored the parser.");
|
|
1357
1961
|
|
|
1358
|
-
|
|
1359
|
-
|
|
1360
|
-
|
|
1361
|
-
|
|
1362
|
-
|
|
1363
|
-
|
|
1364
|
-
|
|
1365
|
-
|
|
1366
|
-
|
|
1962
|
+
try {
|
|
1963
|
+
const result = await executeSubagentRead(
|
|
1964
|
+
{ subagent_id: subagentId },
|
|
1965
|
+
makeContext(ownerConversation),
|
|
1966
|
+
);
|
|
1967
|
+
expect(result.isError).toBe(false);
|
|
1968
|
+
expect(result.content).toBe(
|
|
1969
|
+
"Refactored the parser.\n\n[stats: 5 tool calls, 4 succeeded, files written via file_write/file_edit: 2]",
|
|
1970
|
+
);
|
|
1971
|
+
} finally {
|
|
1972
|
+
mockGetMessages = () => null;
|
|
1973
|
+
}
|
|
1367
1974
|
});
|
|
1368
1975
|
|
|
1369
|
-
test("
|
|
1976
|
+
test("a zero-call subagent's output carries the unverified warning", async () => {
|
|
1370
1977
|
const manager = getSubagentManager();
|
|
1371
|
-
const subagentId = "
|
|
1372
|
-
injectSubagent(manager, subagentId,
|
|
1978
|
+
const subagentId = "read-stats-zero";
|
|
1979
|
+
injectSubagent(manager, subagentId, ownerConversation, "completed", {
|
|
1980
|
+
stats: { calls: 0, succeeded: 0, filesWritten: 0 },
|
|
1981
|
+
});
|
|
1982
|
+
stubOutput(subagentId, "I ran the tests and they all passed.");
|
|
1373
1983
|
|
|
1374
|
-
|
|
1375
|
-
|
|
1376
|
-
|
|
1377
|
-
|
|
1378
|
-
|
|
1379
|
-
|
|
1984
|
+
try {
|
|
1985
|
+
const result = await executeSubagentRead(
|
|
1986
|
+
{ subagent_id: subagentId },
|
|
1987
|
+
makeContext(ownerConversation),
|
|
1988
|
+
);
|
|
1989
|
+
expect(result.content).toContain("I ran the tests and they all passed.");
|
|
1990
|
+
expect(result.content).toContain(
|
|
1991
|
+
"[stats: no tools were used by this subagent; treat any claims of executed work as unverified]",
|
|
1992
|
+
);
|
|
1993
|
+
} finally {
|
|
1994
|
+
mockGetMessages = () => null;
|
|
1995
|
+
}
|
|
1380
1996
|
});
|
|
1381
1997
|
|
|
1382
|
-
test("
|
|
1998
|
+
test("the footer also lands on a subagent that produced no text", async () => {
|
|
1383
1999
|
const manager = getSubagentManager();
|
|
1384
|
-
const subagentId = "
|
|
1385
|
-
injectSubagent(manager, subagentId, "
|
|
2000
|
+
const subagentId = "read-stats-silent";
|
|
2001
|
+
injectSubagent(manager, subagentId, ownerConversation, "completed", {
|
|
2002
|
+
stats: { calls: 3, succeeded: 3, filesWritten: 1 },
|
|
2003
|
+
});
|
|
2004
|
+
mockGetMessages = () => [{ role: "user", content: "go" }];
|
|
1386
2005
|
|
|
1387
|
-
|
|
2006
|
+
try {
|
|
2007
|
+
const result = await executeSubagentRead(
|
|
2008
|
+
{ subagent_id: subagentId },
|
|
2009
|
+
makeContext(ownerConversation),
|
|
2010
|
+
);
|
|
2011
|
+
expect(result.content).toContain("no text output");
|
|
2012
|
+
expect(result.content).toContain(
|
|
2013
|
+
"[stats: 3 tool calls, 3 succeeded, files written via file_write/file_edit: 1]",
|
|
2014
|
+
);
|
|
2015
|
+
} finally {
|
|
2016
|
+
mockGetMessages = () => null;
|
|
2017
|
+
}
|
|
2018
|
+
});
|
|
2019
|
+
|
|
2020
|
+
test("a live subagent that never recorded a run claims nothing", async () => {
|
|
2021
|
+
const manager = getSubagentManager();
|
|
2022
|
+
const subagentId = "read-stats-none";
|
|
2023
|
+
injectSubagent(manager, subagentId, ownerConversation, "aborted");
|
|
2024
|
+
stubOutput(subagentId, "Partial output.");
|
|
2025
|
+
|
|
2026
|
+
try {
|
|
2027
|
+
const result = await executeSubagentRead(
|
|
2028
|
+
{ subagent_id: subagentId },
|
|
2029
|
+
makeContext(ownerConversation),
|
|
2030
|
+
);
|
|
2031
|
+
// No harvest ever happened, so there is nothing measured to report, and
|
|
2032
|
+
// nothing was lost either, so it must not claim the counters are gone.
|
|
2033
|
+
expect(result.content).toBe("Partial output.");
|
|
2034
|
+
expect(result.content).not.toContain("[stats:");
|
|
2035
|
+
} finally {
|
|
2036
|
+
mockGetMessages = () => null;
|
|
2037
|
+
}
|
|
2038
|
+
});
|
|
2039
|
+
|
|
2040
|
+
test("a follow-up turn's tool calls land in the footer, not just the run's", async () => {
|
|
2041
|
+
const manager = getSubagentManager();
|
|
2042
|
+
const subagentId = "read-stats-queued";
|
|
2043
|
+
injectSubagent(manager, subagentId, ownerConversation, "completed", {
|
|
2044
|
+
stats: { calls: 2, succeeded: 2, filesWritten: 0 },
|
|
2045
|
+
});
|
|
2046
|
+
// Guidance queued during the run drains after the run harvested, into the
|
|
2047
|
+
// same retained conversation, so the counters keep moving past that
|
|
2048
|
+
// reading. The read has to look again rather than quote the harvest.
|
|
2049
|
+
const live = liveToolStats(manager, subagentId);
|
|
2050
|
+
live.calls += 3;
|
|
2051
|
+
live.succeeded += 2;
|
|
2052
|
+
live.filesWritten.add("/queued-turn.md");
|
|
2053
|
+
stubOutput(subagentId, "Applied the follow-up guidance.");
|
|
2054
|
+
|
|
2055
|
+
try {
|
|
2056
|
+
const result = await executeSubagentRead(
|
|
2057
|
+
{ subagent_id: subagentId },
|
|
2058
|
+
makeContext(ownerConversation),
|
|
2059
|
+
);
|
|
2060
|
+
expect(result.content).toBe(
|
|
2061
|
+
"Applied the follow-up guidance.\n\n[stats: 5 tool calls, 4 succeeded, files written via file_write/file_edit: 1]",
|
|
2062
|
+
);
|
|
2063
|
+
} finally {
|
|
2064
|
+
mockGetMessages = () => null;
|
|
2065
|
+
}
|
|
2066
|
+
});
|
|
2067
|
+
|
|
2068
|
+
test("a subagent rebuilt from its row reports its counters unavailable", async () => {
|
|
2069
|
+
const manager = getSubagentManager();
|
|
2070
|
+
const subagentId = "read-stats-rehydrated";
|
|
2071
|
+
injectSubagent(manager, subagentId, ownerConversation, "completed", {
|
|
2072
|
+
rehydrated: true,
|
|
2073
|
+
});
|
|
2074
|
+
stubOutput(subagentId, "Output from before the restart.");
|
|
2075
|
+
|
|
2076
|
+
try {
|
|
2077
|
+
const result = await executeSubagentRead(
|
|
2078
|
+
{ subagent_id: subagentId },
|
|
2079
|
+
makeContext(ownerConversation),
|
|
2080
|
+
);
|
|
2081
|
+
// Being in the manager is not evidence of a live run: this entry was
|
|
2082
|
+
// rebuilt from the durable row, which carries no counters at all.
|
|
2083
|
+
expect(manager.getState(subagentId)).toBeDefined();
|
|
2084
|
+
expect(result.content).toBe(
|
|
2085
|
+
"Output from before the restart.\n\n[stats: unavailable (tool counters are not retained for this subagent)]",
|
|
2086
|
+
);
|
|
2087
|
+
} finally {
|
|
2088
|
+
mockGetMessages = () => null;
|
|
2089
|
+
}
|
|
2090
|
+
});
|
|
2091
|
+
});
|
|
2092
|
+
|
|
2093
|
+
// ── Reads against a queued follow-up turn ───────────────────────────
|
|
2094
|
+
|
|
2095
|
+
describe("Subagent read while a queued follow-up turn is still in flight", () => {
|
|
2096
|
+
const ownerConversation = "read-queued-owner";
|
|
2097
|
+
|
|
2098
|
+
function stubOutput(subagentId: string, texts: () => string[]) {
|
|
2099
|
+
mockGetMessages = (convId: string) =>
|
|
2100
|
+
convId === `conv-${subagentId}`
|
|
2101
|
+
? texts().map((text) => ({
|
|
2102
|
+
role: "assistant",
|
|
2103
|
+
content: [{ type: "text", text }],
|
|
2104
|
+
}))
|
|
2105
|
+
: null;
|
|
2106
|
+
}
|
|
2107
|
+
|
|
2108
|
+
test("waits for the queued turn rather than answering from the run before it", async () => {
|
|
2109
|
+
const manager = getSubagentManager();
|
|
2110
|
+
const subagentId = "read-queued-settles";
|
|
2111
|
+
injectSubagent(manager, subagentId, ownerConversation, "completed", {
|
|
2112
|
+
stats: { calls: 2, succeeded: 2, filesWritten: 0 },
|
|
2113
|
+
});
|
|
2114
|
+
const drain = queuedFollowUpTurn(manager, subagentId);
|
|
2115
|
+
const live = liveToolStats(manager, subagentId);
|
|
2116
|
+
let followUpLanded = false;
|
|
2117
|
+
stubOutput(subagentId, () =>
|
|
2118
|
+
followUpLanded
|
|
2119
|
+
? ["Initial run output.", "Applied the follow-up guidance."]
|
|
2120
|
+
: ["Initial run output."],
|
|
2121
|
+
);
|
|
2122
|
+
|
|
2123
|
+
// The drain picks the message up, runs the turn, and only then does the
|
|
2124
|
+
// transcript and the counters cover it.
|
|
2125
|
+
setTimeout(() => {
|
|
2126
|
+
drain.queueDepth = 0;
|
|
2127
|
+
drain.processing = true;
|
|
2128
|
+
}, 10);
|
|
2129
|
+
setTimeout(() => {
|
|
2130
|
+
live.calls += 3;
|
|
2131
|
+
live.succeeded += 3;
|
|
2132
|
+
live.filesWritten.add("/follow-up.md");
|
|
2133
|
+
followUpLanded = true;
|
|
2134
|
+
drain.processing = false;
|
|
2135
|
+
}, 30);
|
|
2136
|
+
|
|
2137
|
+
try {
|
|
2138
|
+
const result = await executeSubagentRead(
|
|
2139
|
+
{ subagent_id: subagentId },
|
|
2140
|
+
makeContext(ownerConversation),
|
|
2141
|
+
);
|
|
2142
|
+
expect(result.content).toBe(
|
|
2143
|
+
"Initial run output.\n\nApplied the follow-up guidance.\n\n" +
|
|
2144
|
+
"[stats: 5 tool calls, 5 succeeded, files written via file_write/file_edit: 1]",
|
|
2145
|
+
);
|
|
2146
|
+
expect(result.content).not.toContain("still processing");
|
|
2147
|
+
} finally {
|
|
2148
|
+
mockGetMessages = () => null;
|
|
2149
|
+
}
|
|
2150
|
+
});
|
|
2151
|
+
|
|
2152
|
+
test("a queue the drain has taken but not yet started is not read as finished", async () => {
|
|
2153
|
+
const manager = getSubagentManager();
|
|
2154
|
+
const subagentId = "read-queued-dispatch-gap";
|
|
2155
|
+
injectSubagent(manager, subagentId, ownerConversation, "completed", {
|
|
2156
|
+
stats: { calls: 1, succeeded: 1, filesWritten: 0 },
|
|
2157
|
+
});
|
|
2158
|
+
const drain = queuedFollowUpTurn(manager, subagentId);
|
|
2159
|
+
let followUpLanded = false;
|
|
2160
|
+
stubOutput(subagentId, () =>
|
|
2161
|
+
followUpLanded ? ["Guidance applied."] : ["Initial run output."],
|
|
2162
|
+
);
|
|
2163
|
+
|
|
2164
|
+
// The gap between the drain shifting the message off the queue and the
|
|
2165
|
+
// turn taking the processing lock: nothing is queued and nothing is
|
|
2166
|
+
// running, yet the turn is on its way.
|
|
2167
|
+
drain.queueDepth = 0;
|
|
2168
|
+
drain.processing = false;
|
|
2169
|
+
setTimeout(() => {
|
|
2170
|
+
drain.processing = true;
|
|
2171
|
+
}, 15);
|
|
2172
|
+
setTimeout(() => {
|
|
2173
|
+
followUpLanded = true;
|
|
2174
|
+
drain.processing = false;
|
|
2175
|
+
}, 40);
|
|
2176
|
+
|
|
2177
|
+
try {
|
|
2178
|
+
const result = await executeSubagentRead(
|
|
2179
|
+
{ subagent_id: subagentId },
|
|
2180
|
+
makeContext(ownerConversation),
|
|
2181
|
+
);
|
|
2182
|
+
expect(result.content).toContain("Guidance applied.");
|
|
2183
|
+
expect(result.content).not.toContain("Initial run output.");
|
|
2184
|
+
} finally {
|
|
2185
|
+
mockGetMessages = () => null;
|
|
2186
|
+
}
|
|
2187
|
+
});
|
|
2188
|
+
|
|
2189
|
+
test("a turn still running at the deadline is reported as unfinished, not as the result", async () => {
|
|
2190
|
+
const manager = getSubagentManager();
|
|
2191
|
+
const subagentId = "read-queued-still-running";
|
|
2192
|
+
injectSubagent(manager, subagentId, ownerConversation, "completed", {
|
|
2193
|
+
stats: { calls: 2, succeeded: 2, filesWritten: 1 },
|
|
2194
|
+
});
|
|
2195
|
+
const drain = queuedFollowUpTurn(manager, subagentId);
|
|
2196
|
+
// The guidance turn outlives the read's patience, as a real one does.
|
|
2197
|
+
drain.queueDepth = 0;
|
|
2198
|
+
drain.processing = true;
|
|
2199
|
+
stubOutput(subagentId, () => ["Initial run output."]);
|
|
2200
|
+
|
|
2201
|
+
try {
|
|
2202
|
+
const result = await executeSubagentRead(
|
|
2203
|
+
{ subagent_id: subagentId },
|
|
2204
|
+
makeContext(ownerConversation),
|
|
2205
|
+
);
|
|
2206
|
+
expect(result.isError).toBe(false);
|
|
2207
|
+
// What there is so far, said to be what there is so far: the counts
|
|
2208
|
+
// are an interim reading and the parent is told to come back.
|
|
2209
|
+
expect(result.content).toBe(
|
|
2210
|
+
"Initial run output.\n\n" +
|
|
2211
|
+
`${SUBAGENT_READ_STILL_PROCESSING}\n\n` +
|
|
2212
|
+
"[stats: 2 tool calls, 2 succeeded, files written via file_write/file_edit: 1]",
|
|
2213
|
+
);
|
|
2214
|
+
} finally {
|
|
2215
|
+
drain.processing = false;
|
|
2216
|
+
mockGetMessages = () => null;
|
|
2217
|
+
}
|
|
2218
|
+
}, 15_000);
|
|
2219
|
+
|
|
2220
|
+
test("a subagent that had nothing queued is read without waiting", async () => {
|
|
2221
|
+
const manager = getSubagentManager();
|
|
2222
|
+
const subagentId = "read-queued-none";
|
|
2223
|
+
injectSubagent(manager, subagentId, ownerConversation, "completed", {
|
|
2224
|
+
stats: { calls: 1, succeeded: 1, filesWritten: 0 },
|
|
2225
|
+
});
|
|
2226
|
+
stubOutput(subagentId, () => ["Ran to completion."]);
|
|
2227
|
+
|
|
2228
|
+
try {
|
|
2229
|
+
const startedAt = Date.now();
|
|
2230
|
+
const result = await executeSubagentRead(
|
|
2231
|
+
{ subagent_id: subagentId },
|
|
2232
|
+
makeContext(ownerConversation),
|
|
2233
|
+
);
|
|
2234
|
+
expect(Date.now() - startedAt).toBeLessThan(500);
|
|
2235
|
+
expect(result.content).toBe(
|
|
2236
|
+
"Ran to completion.\n\n[stats: 1 tool call, 1 succeeded, files written via file_write/file_edit: 0]",
|
|
2237
|
+
);
|
|
2238
|
+
} finally {
|
|
2239
|
+
mockGetMessages = () => null;
|
|
2240
|
+
}
|
|
2241
|
+
});
|
|
2242
|
+
});
|
|
2243
|
+
|
|
2244
|
+
// ── Abort success path details ──────────────────────────────────────
|
|
2245
|
+
|
|
2246
|
+
describe("Subagent abort success responses", () => {
|
|
2247
|
+
test("abort returns subagentId and aborted status on success", async () => {
|
|
2248
|
+
const manager = getSubagentManager();
|
|
2249
|
+
const subagentId = "abort-detail-1";
|
|
2250
|
+
injectSubagent(manager, subagentId, "abort-owner-sess", "running");
|
|
2251
|
+
|
|
2252
|
+
const result = await executeSubagentAbort(
|
|
2253
|
+
{ subagent_id: subagentId },
|
|
2254
|
+
makeContext("abort-owner-sess"),
|
|
2255
|
+
);
|
|
2256
|
+
expect(result.isError).toBe(false);
|
|
2257
|
+
const parsed = JSON.parse(result.content);
|
|
2258
|
+
expect(parsed.subagentId).toBe(subagentId);
|
|
2259
|
+
expect(parsed.status).toBe("aborted");
|
|
2260
|
+
expect(parsed.message).toContain("aborted successfully");
|
|
2261
|
+
});
|
|
2262
|
+
|
|
2263
|
+
test("abort fails for already-completed subagent", async () => {
|
|
2264
|
+
const manager = getSubagentManager();
|
|
2265
|
+
const subagentId = "abort-completed-1";
|
|
2266
|
+
injectSubagent(manager, subagentId, "abort-owner-sess", "completed");
|
|
2267
|
+
|
|
2268
|
+
const result = await executeSubagentAbort(
|
|
2269
|
+
{ subagent_id: subagentId },
|
|
2270
|
+
makeContext("abort-owner-sess"),
|
|
2271
|
+
);
|
|
2272
|
+
expect(result.isError).toBe(true);
|
|
2273
|
+
expect(result.content).toContain("Could not abort");
|
|
2274
|
+
});
|
|
2275
|
+
|
|
2276
|
+
test("abort fails for already-failed subagent", async () => {
|
|
2277
|
+
const manager = getSubagentManager();
|
|
2278
|
+
const subagentId = "abort-failed-1";
|
|
2279
|
+
injectSubagent(manager, subagentId, "abort-owner-sess", "failed");
|
|
2280
|
+
|
|
2281
|
+
const result = await executeSubagentAbort(
|
|
1388
2282
|
{ subagent_id: subagentId },
|
|
1389
2283
|
makeContext("abort-owner-sess"),
|
|
1390
2284
|
);
|
|
@@ -1569,7 +2463,7 @@ describe("Subagent role-based spawn", () => {
|
|
|
1569
2463
|
}
|
|
1570
2464
|
});
|
|
1571
2465
|
|
|
1572
|
-
test("spawn without role
|
|
2466
|
+
test("spawn without role runs as builder", async () => {
|
|
1573
2467
|
const manager = getSubagentManager();
|
|
1574
2468
|
const originalSpawn = manager.spawn.bind(manager);
|
|
1575
2469
|
let capturedConfig: Record<string, unknown> | undefined;
|
|
@@ -1581,52 +2475,523 @@ describe("Subagent role-based spawn", () => {
|
|
|
1581
2475
|
|
|
1582
2476
|
try {
|
|
1583
2477
|
const result = await executeSubagentSpawn(
|
|
1584
|
-
{ label: "
|
|
2478
|
+
{ label: "Unlabelled task", objective: "Do something" },
|
|
1585
2479
|
makeContext("sess-role-2", { sendToClient: () => {} }),
|
|
1586
2480
|
);
|
|
1587
2481
|
expect(result.isError).toBe(false);
|
|
1588
2482
|
const parsed = JSON.parse(result.content);
|
|
1589
2483
|
expect(parsed.subagentId).toBe("role-default-id");
|
|
2484
|
+
expect(parsed.role).toBe("builder");
|
|
2485
|
+
// Naming no role stays write-capable, and says nothing extra about it.
|
|
2486
|
+
expect(parsed.roleNote).toBeUndefined();
|
|
1590
2487
|
expect(capturedConfig).toBeDefined();
|
|
1591
|
-
|
|
1592
|
-
|
|
2488
|
+
expect(capturedConfig!.role).toBe("builder");
|
|
2489
|
+
// And keeps the whole surface: builder imposes no allowlist, so the
|
|
2490
|
+
// manager filters nothing for a spawn that named no role.
|
|
2491
|
+
expect(SUBAGENT_ROLE_REGISTRY.builder.allowedTools).toBeUndefined();
|
|
2492
|
+
} finally {
|
|
2493
|
+
manager.spawn = originalSpawn;
|
|
2494
|
+
}
|
|
2495
|
+
});
|
|
2496
|
+
|
|
2497
|
+
test.each([
|
|
2498
|
+
["planner", "researcher"],
|
|
2499
|
+
["investigator", "researcher"],
|
|
2500
|
+
["coder", "builder"],
|
|
2501
|
+
["general", "builder"],
|
|
2502
|
+
])("legacy role %s spawns a %s", async (legacy, expected) => {
|
|
2503
|
+
const manager = getSubagentManager();
|
|
2504
|
+
const originalSpawn = manager.spawn.bind(manager);
|
|
2505
|
+
let capturedConfig: Record<string, unknown> | undefined;
|
|
2506
|
+
|
|
2507
|
+
manager.spawn = async (config: Record<string, unknown>) => {
|
|
2508
|
+
capturedConfig = config;
|
|
2509
|
+
return `alias-${legacy}-id`;
|
|
2510
|
+
};
|
|
2511
|
+
|
|
2512
|
+
try {
|
|
2513
|
+
const result = await executeSubagentSpawn(
|
|
2514
|
+
{ label: `${legacy} task`, objective: "Do something", role: legacy },
|
|
2515
|
+
makeContext(`sess-alias-${legacy}`, { sendToClient: () => {} }),
|
|
2516
|
+
);
|
|
2517
|
+
expect(result.isError).toBe(false);
|
|
2518
|
+
expect(capturedConfig!.role).toBe(expected);
|
|
2519
|
+
expect(capturedConfig!.persona).toBeUndefined();
|
|
2520
|
+
|
|
2521
|
+
const parsed = JSON.parse(result.content);
|
|
2522
|
+
expect(parsed.role).toBe(expected);
|
|
2523
|
+
expect(parsed.roleNote).toContain(legacy);
|
|
2524
|
+
expect(parsed.roleNote).toContain(expected);
|
|
1593
2525
|
} finally {
|
|
1594
2526
|
manager.spawn = originalSpawn;
|
|
1595
2527
|
}
|
|
1596
2528
|
});
|
|
1597
2529
|
|
|
1598
|
-
test("
|
|
1599
|
-
|
|
2530
|
+
test("the child of a legacy role gets the new type's tool surface", () => {
|
|
2531
|
+
// The alias resolves to a type, and the type owns the tools.
|
|
2532
|
+
expect(SUBAGENT_ROLE_REGISTRY.researcher.allowedTools).toContain(
|
|
2533
|
+
"code_search",
|
|
2534
|
+
);
|
|
2535
|
+
expect(SUBAGENT_ROLE_REGISTRY.researcher.allowedTools).not.toContain(
|
|
2536
|
+
"bash",
|
|
2537
|
+
);
|
|
2538
|
+
// `general` resolves to builder, and builder is unrestricted exactly as
|
|
2539
|
+
// `general` was: a fixed list would drop connectors, MCP tools, and
|
|
2540
|
+
// browser and computer use from every spawn that names neither.
|
|
2541
|
+
expect(SUBAGENT_ROLE_REGISTRY.builder.allowedTools).toBeUndefined();
|
|
2542
|
+
});
|
|
2543
|
+
|
|
2544
|
+
test("an unrecognized role spawns a researcher with the text as persona", async () => {
|
|
2545
|
+
const manager = getSubagentManager();
|
|
2546
|
+
const originalSpawn = manager.spawn.bind(manager);
|
|
2547
|
+
let capturedConfig: Record<string, unknown> | undefined;
|
|
2548
|
+
|
|
2549
|
+
manager.spawn = async (config: Record<string, unknown>) => {
|
|
2550
|
+
capturedConfig = config;
|
|
2551
|
+
return "role-persona-id";
|
|
2552
|
+
};
|
|
2553
|
+
|
|
2554
|
+
try {
|
|
2555
|
+
const result = await executeSubagentSpawn(
|
|
2556
|
+
{
|
|
2557
|
+
label: "Persona task",
|
|
2558
|
+
objective: "Assess the filing",
|
|
2559
|
+
role: "financial journalist",
|
|
2560
|
+
},
|
|
2561
|
+
makeContext("sess-role-persona", { sendToClient: () => {} }),
|
|
2562
|
+
);
|
|
2563
|
+
expect(result.isError).toBe(false);
|
|
2564
|
+
expect(capturedConfig!.role).toBe("researcher");
|
|
2565
|
+
expect(capturedConfig!.persona).toBe("financial journalist");
|
|
2566
|
+
|
|
2567
|
+
const parsed = JSON.parse(result.content);
|
|
2568
|
+
expect(parsed.role).toBe("researcher");
|
|
2569
|
+
expect(parsed.roleNote).toContain("financial journalist");
|
|
2570
|
+
expect(parsed.roleNote).toContain("persona");
|
|
2571
|
+
expect(parsed.roleNote).toContain("builder");
|
|
2572
|
+
} finally {
|
|
2573
|
+
manager.spawn = originalSpawn;
|
|
2574
|
+
}
|
|
2575
|
+
});
|
|
2576
|
+
|
|
2577
|
+
test("the persona reaches the child's system prompt", () => {
|
|
2578
|
+
const prompt = buildSubagentSystemPrompt(
|
|
2579
|
+
{
|
|
2580
|
+
id: "sub-persona",
|
|
2581
|
+
parentConversationId: "conv-1",
|
|
2582
|
+
label: "Persona task",
|
|
2583
|
+
objective: "Assess the filing",
|
|
2584
|
+
role: "researcher",
|
|
2585
|
+
persona: "financial journalist",
|
|
2586
|
+
},
|
|
2587
|
+
"researcher",
|
|
2588
|
+
);
|
|
2589
|
+
expect(prompt).toContain(
|
|
2590
|
+
"- Persona: act as financial journalist for this task.",
|
|
2591
|
+
);
|
|
2592
|
+
});
|
|
2593
|
+
|
|
2594
|
+
test("a whitespace-only role is treated as no role at all", async () => {
|
|
2595
|
+
const manager = getSubagentManager();
|
|
2596
|
+
const originalSpawn = manager.spawn.bind(manager);
|
|
2597
|
+
let capturedConfig: Record<string, unknown> | undefined;
|
|
2598
|
+
|
|
2599
|
+
manager.spawn = async (config: Record<string, unknown>) => {
|
|
2600
|
+
capturedConfig = config;
|
|
2601
|
+
return "role-blank-id";
|
|
2602
|
+
};
|
|
2603
|
+
|
|
2604
|
+
try {
|
|
2605
|
+
const result = await executeSubagentSpawn(
|
|
2606
|
+
{ label: "Blank role", objective: "Do something", role: " " },
|
|
2607
|
+
makeContext("sess-role-blank", { sendToClient: () => {} }),
|
|
2608
|
+
);
|
|
2609
|
+
expect(result.isError).toBe(false);
|
|
2610
|
+
expect(capturedConfig!.role).toBe("builder");
|
|
2611
|
+
expect(capturedConfig!.persona).toBeUndefined();
|
|
2612
|
+
expect(JSON.parse(result.content).roleNote).toBeUndefined();
|
|
2613
|
+
} finally {
|
|
2614
|
+
manager.spawn = originalSpawn;
|
|
2615
|
+
}
|
|
2616
|
+
});
|
|
2617
|
+
|
|
2618
|
+
test("a sentence-length role still spawns, bounded, as a researcher persona", async () => {
|
|
2619
|
+
const manager = getSubagentManager();
|
|
2620
|
+
const originalSpawn = manager.spawn.bind(manager);
|
|
2621
|
+
let capturedConfig: Record<string, unknown> | undefined;
|
|
2622
|
+
|
|
2623
|
+
manager.spawn = async (config: Record<string, unknown>) => {
|
|
2624
|
+
capturedConfig = config;
|
|
2625
|
+
return "role-sentence-id";
|
|
2626
|
+
};
|
|
2627
|
+
|
|
2628
|
+
const sentence =
|
|
2629
|
+
"You are a meticulous senior staff engineer who reviews every change against the design document and reports every discrepancy, however small.";
|
|
2630
|
+
try {
|
|
2631
|
+
const result = await executeSubagentSpawn(
|
|
2632
|
+
{
|
|
2633
|
+
label: "Sentence role",
|
|
2634
|
+
objective: "Review the change",
|
|
2635
|
+
role: sentence,
|
|
2636
|
+
},
|
|
2637
|
+
makeContext("sess-role-sentence", { sendToClient: () => {} }),
|
|
2638
|
+
);
|
|
2639
|
+
expect(result.isError).toBe(false);
|
|
2640
|
+
expect(capturedConfig!.role).toBe("researcher");
|
|
2641
|
+
expect((capturedConfig!.persona as string).length).toBeLessThan(
|
|
2642
|
+
sentence.length,
|
|
2643
|
+
);
|
|
2644
|
+
} finally {
|
|
2645
|
+
manager.spawn = originalSpawn;
|
|
2646
|
+
}
|
|
2647
|
+
});
|
|
2648
|
+
|
|
2649
|
+
test("spawn tool definition takes role as an unconstrained string", () => {
|
|
2650
|
+
const def = findTool("subagent_spawn");
|
|
2651
|
+
expect(def).toBeDefined();
|
|
2652
|
+
expect(def.input_schema.properties.role).toBeDefined();
|
|
2653
|
+
expect(def.input_schema.properties.role.type).toBe("string");
|
|
2654
|
+
// Manifest validation runs ahead of the executor, so an enum here would
|
|
2655
|
+
// reject the legacy names and personas `resolveSubagentRole` handles. The
|
|
2656
|
+
// three types are named in the description instead.
|
|
2657
|
+
expect(def.input_schema.properties.role.enum).toBeUndefined();
|
|
2658
|
+
for (const type of ["researcher", "builder", "advisor"]) {
|
|
2659
|
+
expect(def.input_schema.properties.role.description).toContain(type);
|
|
2660
|
+
}
|
|
2661
|
+
// role is not required
|
|
2662
|
+
expect(def.input_schema.required).not.toContain("role");
|
|
2663
|
+
});
|
|
2664
|
+
});
|
|
2665
|
+
|
|
2666
|
+
// ── Output contract ─────────────────────────────────────────────────
|
|
2667
|
+
|
|
2668
|
+
describe("Subagent output contract", () => {
|
|
2669
|
+
/**
|
|
2670
|
+
* Run a spawn with the manager stubbed, reporting both the config it was
|
|
2671
|
+
* handed and whether it was called at all (a rejected contract must not
|
|
2672
|
+
* spawn).
|
|
2673
|
+
*/
|
|
2674
|
+
async function spawnCapturing(
|
|
2675
|
+
input: Record<string, unknown>,
|
|
2676
|
+
contextExtras: Record<string, unknown> = {},
|
|
2677
|
+
): Promise<{
|
|
2678
|
+
result: { content: string; isError: boolean };
|
|
2679
|
+
config: Record<string, unknown> | undefined;
|
|
2680
|
+
}> {
|
|
2681
|
+
const manager = getSubagentManager();
|
|
2682
|
+
const originalSpawn = manager.spawn.bind(manager);
|
|
2683
|
+
let capturedConfig: Record<string, unknown> | undefined;
|
|
2684
|
+
manager.spawn = async (config: Record<string, unknown>) => {
|
|
2685
|
+
capturedConfig = config;
|
|
2686
|
+
return "contract-subagent-id";
|
|
2687
|
+
};
|
|
2688
|
+
try {
|
|
2689
|
+
const result = await executeSubagentSpawn(
|
|
2690
|
+
input,
|
|
2691
|
+
makeContext("sess-contract", {
|
|
2692
|
+
sendToClient: () => {},
|
|
2693
|
+
...contextExtras,
|
|
2694
|
+
}),
|
|
2695
|
+
);
|
|
2696
|
+
return { result, config: capturedConfig };
|
|
2697
|
+
} finally {
|
|
2698
|
+
manager.spawn = originalSpawn;
|
|
2699
|
+
}
|
|
2700
|
+
}
|
|
2701
|
+
|
|
2702
|
+
test.each([
|
|
2703
|
+
["report", "builder"],
|
|
2704
|
+
["verdict", "researcher"],
|
|
2705
|
+
["artifact", "builder"],
|
|
2706
|
+
])("output_contract %s is accepted for a %s", async (contract, role) => {
|
|
2707
|
+
const { result, config } = await spawnCapturing({
|
|
2708
|
+
label: "Contract task",
|
|
2709
|
+
objective: "Do it",
|
|
2710
|
+
role,
|
|
2711
|
+
output_contract: contract,
|
|
2712
|
+
});
|
|
2713
|
+
expect(result.isError).toBe(false);
|
|
2714
|
+
expect(config!.outputContract).toBe(contract);
|
|
2715
|
+
});
|
|
2716
|
+
|
|
2717
|
+
test("an unknown output_contract fails validation without spawning", async () => {
|
|
2718
|
+
const { result, config } = await spawnCapturing({
|
|
2719
|
+
label: "Bad contract",
|
|
2720
|
+
objective: "Do it",
|
|
2721
|
+
role: "researcher",
|
|
2722
|
+
output_contract: "checklist",
|
|
2723
|
+
});
|
|
2724
|
+
expect(result.isError).toBe(true);
|
|
2725
|
+
expect(result.content).toContain('Invalid input for tool "subagent_spawn"');
|
|
2726
|
+
expect(result.content).toContain("output_contract");
|
|
2727
|
+
expect(config).toBeUndefined();
|
|
2728
|
+
});
|
|
2729
|
+
|
|
2730
|
+
test("verdict on a builder returns a mismatch error instead of spawning", async () => {
|
|
2731
|
+
const { result, config } = await spawnCapturing({
|
|
2732
|
+
label: "Check it",
|
|
2733
|
+
objective: "Verify the migration ran",
|
|
2734
|
+
role: "builder",
|
|
2735
|
+
output_contract: "verdict",
|
|
2736
|
+
});
|
|
2737
|
+
expect(result.isError).toBe(true);
|
|
2738
|
+
expect(result.content).toContain("only available to researcher-typed");
|
|
2739
|
+
expect(result.content).toContain('resolved to "builder"');
|
|
2740
|
+
expect(config).toBeUndefined();
|
|
2741
|
+
});
|
|
2742
|
+
|
|
2743
|
+
test("verdict on a spawn that names no role is a mismatch (the default is builder)", async () => {
|
|
2744
|
+
const { result, config } = await spawnCapturing({
|
|
2745
|
+
label: "Check it",
|
|
2746
|
+
objective: "Verify the migration ran",
|
|
2747
|
+
output_contract: "verdict",
|
|
2748
|
+
});
|
|
2749
|
+
expect(result.isError).toBe(true);
|
|
2750
|
+
expect(result.content).toContain('role "researcher"');
|
|
2751
|
+
expect(config).toBeUndefined();
|
|
2752
|
+
});
|
|
2753
|
+
|
|
2754
|
+
test("verdict rides the researcher fallback an unknown role resolves to", async () => {
|
|
2755
|
+
const { result, config } = await spawnCapturing({
|
|
2756
|
+
label: "Check it",
|
|
2757
|
+
objective: "Verify the migration ran",
|
|
2758
|
+
role: "release auditor",
|
|
2759
|
+
output_contract: "verdict",
|
|
2760
|
+
});
|
|
2761
|
+
expect(result.isError).toBe(false);
|
|
2762
|
+
expect(config!.role).toBe("researcher");
|
|
2763
|
+
expect(config!.outputContract).toBe("verdict");
|
|
2764
|
+
});
|
|
2765
|
+
|
|
2766
|
+
test("artifact on a researcher returns a mismatch error instead of spawning", async () => {
|
|
2767
|
+
const { result, config } = await spawnCapturing({
|
|
2768
|
+
label: "Write it",
|
|
2769
|
+
objective: "Produce the migration file",
|
|
2770
|
+
role: "researcher",
|
|
2771
|
+
output_contract: "artifact",
|
|
2772
|
+
});
|
|
2773
|
+
expect(result.isError).toBe(true);
|
|
2774
|
+
expect(result.content).toContain("only available to builder-typed");
|
|
2775
|
+
expect(config).toBeUndefined();
|
|
2776
|
+
});
|
|
2777
|
+
|
|
2778
|
+
test("the advisor takes no output contract and never consults", async () => {
|
|
2779
|
+
const manager = getSubagentManager();
|
|
2780
|
+
const originalAwait = manager.spawnAndAwait.bind(manager);
|
|
2781
|
+
let consulted = false;
|
|
2782
|
+
manager.spawnAndAwait = async () => {
|
|
2783
|
+
consulted = true;
|
|
2784
|
+
return "advice";
|
|
2785
|
+
};
|
|
2786
|
+
try {
|
|
2787
|
+
const result = await executeSubagentSpawn(
|
|
2788
|
+
{
|
|
2789
|
+
label: "Consult",
|
|
2790
|
+
objective: "Check the plan",
|
|
2791
|
+
role: "advisor",
|
|
2792
|
+
output_contract: "verdict",
|
|
2793
|
+
},
|
|
2794
|
+
makeContext("sess-contract-advisor", { sendToClient: () => {} }),
|
|
2795
|
+
);
|
|
2796
|
+
expect(result.isError).toBe(true);
|
|
2797
|
+
expect(result.content).toContain("does not apply to the advisor");
|
|
2798
|
+
expect(consulted).toBe(false);
|
|
2799
|
+
} finally {
|
|
2800
|
+
manager.spawnAndAwait = originalAwait;
|
|
2801
|
+
}
|
|
2802
|
+
});
|
|
2803
|
+
|
|
2804
|
+
test("an explicit report is rejected for the advisor too", async () => {
|
|
2805
|
+
const manager = getSubagentManager();
|
|
2806
|
+
const originalAwait = manager.spawnAndAwait.bind(manager);
|
|
2807
|
+
let consulted = false;
|
|
2808
|
+
manager.spawnAndAwait = async () => {
|
|
2809
|
+
consulted = true;
|
|
2810
|
+
return "advice";
|
|
2811
|
+
};
|
|
2812
|
+
try {
|
|
2813
|
+
const result = await executeSubagentSpawn(
|
|
2814
|
+
{
|
|
2815
|
+
label: "Consult",
|
|
2816
|
+
objective: "Check the plan",
|
|
2817
|
+
role: "advisor",
|
|
2818
|
+
output_contract: "report",
|
|
2819
|
+
},
|
|
2820
|
+
makeContext("sess-contract-advisor-report", {
|
|
2821
|
+
sendToClient: () => {},
|
|
2822
|
+
}),
|
|
2823
|
+
);
|
|
2824
|
+
expect(result.isError).toBe(true);
|
|
2825
|
+
expect(result.content).toContain("does not apply to the advisor");
|
|
2826
|
+
expect(result.content).toContain('"report"');
|
|
2827
|
+
expect(consulted).toBe(false);
|
|
2828
|
+
} finally {
|
|
2829
|
+
manager.spawnAndAwait = originalAwait;
|
|
2830
|
+
}
|
|
2831
|
+
});
|
|
2832
|
+
|
|
2833
|
+
test("a null contract reads as omitted and reaches the advisor branch", async () => {
|
|
2834
|
+
const result = await executeSubagentSpawn(
|
|
2835
|
+
{
|
|
2836
|
+
label: "Consult",
|
|
2837
|
+
objective: "Check the plan",
|
|
2838
|
+
role: "advisor",
|
|
2839
|
+
output_contract: null,
|
|
2840
|
+
},
|
|
2841
|
+
makeContext("sess-contract-advisor-none", { sendToClient: () => {} }),
|
|
2842
|
+
);
|
|
2843
|
+
// The advisor branch runs and reports its own missing-parent notice, so
|
|
2844
|
+
// the contract check waved this spawn through.
|
|
2845
|
+
expect(result.content).toContain("advisor unavailable");
|
|
2846
|
+
expect(result.content).not.toContain("does not apply to the advisor");
|
|
2847
|
+
});
|
|
2848
|
+
|
|
2849
|
+
test("verdict defaults the child to the cost-optimized tier", async () => {
|
|
2850
|
+
const { result, config } = await spawnCapturing(
|
|
2851
|
+
{
|
|
2852
|
+
label: "Check it",
|
|
2853
|
+
objective: "Verify each acceptance criterion",
|
|
2854
|
+
role: "researcher",
|
|
2855
|
+
output_contract: "verdict",
|
|
2856
|
+
},
|
|
2857
|
+
{ invokingCallSite: "mainAgent" },
|
|
2858
|
+
);
|
|
2859
|
+
expect(result.isError).toBe(false);
|
|
2860
|
+
// Ahead of the mainAgent default (balanced) this spawn would otherwise
|
|
2861
|
+
// inherit, and unforced like every other inheritance rung.
|
|
2862
|
+
expect(config!.overrideProfile).toBe("cost-optimized");
|
|
2863
|
+
expect(config!.forceOverrideProfile).toBeUndefined();
|
|
2864
|
+
});
|
|
2865
|
+
|
|
2866
|
+
test("a subagentSpawn call-site pin outranks the verdict preset", async () => {
|
|
2867
|
+
// A pinned call site is a user's choice about delegated work; the verdict
|
|
2868
|
+
// tier is this tool's preset. An override wins over a call-site profile
|
|
2869
|
+
// outright under single-winner resolution, so the preset must not be
|
|
2870
|
+
// forwarded as one here or the pin is silently downgraded.
|
|
2871
|
+
setConfig("llm", {
|
|
2872
|
+
...BASE_LLM_CONFIG,
|
|
2873
|
+
callSites: { subagentSpawn: { profile: "quality-optimized" } },
|
|
2874
|
+
});
|
|
2875
|
+
try {
|
|
2876
|
+
const { result, config } = await spawnCapturing(
|
|
2877
|
+
{
|
|
2878
|
+
label: "Check it",
|
|
2879
|
+
objective: "Verify each acceptance criterion",
|
|
2880
|
+
role: "researcher",
|
|
2881
|
+
output_contract: "verdict",
|
|
2882
|
+
},
|
|
2883
|
+
{ invokingCallSite: "mainAgent" },
|
|
2884
|
+
);
|
|
2885
|
+
expect(result.isError).toBe(false);
|
|
2886
|
+
// No override at all, so the child resolves on the pinned call site.
|
|
2887
|
+
expect(config!.overrideProfile).toBeUndefined();
|
|
2888
|
+
expect(config!.forceOverrideProfile).toBeUndefined();
|
|
2889
|
+
} finally {
|
|
2890
|
+
setConfig("llm", BASE_LLM_CONFIG);
|
|
2891
|
+
}
|
|
2892
|
+
});
|
|
2893
|
+
|
|
2894
|
+
test("the verdict preset still applies when the call site is unpinned", async () => {
|
|
2895
|
+
setConfig("llm", {
|
|
2896
|
+
...BASE_LLM_CONFIG,
|
|
2897
|
+
callSites: { mainAgent: { profile: "quality-optimized" } },
|
|
2898
|
+
});
|
|
2899
|
+
try {
|
|
2900
|
+
const { config } = await spawnCapturing(
|
|
2901
|
+
{
|
|
2902
|
+
label: "Check it",
|
|
2903
|
+
objective: "Verify each acceptance criterion",
|
|
2904
|
+
role: "researcher",
|
|
2905
|
+
output_contract: "verdict",
|
|
2906
|
+
},
|
|
2907
|
+
{ invokingCallSite: "mainAgent" },
|
|
2908
|
+
);
|
|
2909
|
+
// A pin on some other call site says nothing about delegated checks.
|
|
2910
|
+
expect(config!.overrideProfile).toBe("cost-optimized");
|
|
2911
|
+
} finally {
|
|
2912
|
+
setConfig("llm", BASE_LLM_CONFIG);
|
|
2913
|
+
}
|
|
2914
|
+
});
|
|
2915
|
+
|
|
2916
|
+
test("an explicit inference_profile beats the verdict default", async () => {
|
|
2917
|
+
const { config } = await spawnCapturing({
|
|
2918
|
+
label: "Check it",
|
|
2919
|
+
objective: "Verify each acceptance criterion",
|
|
2920
|
+
role: "researcher",
|
|
2921
|
+
output_contract: "verdict",
|
|
2922
|
+
inference_profile: "quality-optimized",
|
|
2923
|
+
});
|
|
2924
|
+
expect(config!.overrideProfile).toBe("quality-optimized");
|
|
2925
|
+
expect(config!.forceOverrideProfile).toBe(true);
|
|
2926
|
+
});
|
|
2927
|
+
|
|
2928
|
+
test("a report contract changes neither the profile nor the framing", async () => {
|
|
2929
|
+
const { config } = await spawnCapturing(
|
|
2930
|
+
{
|
|
2931
|
+
label: "Research it",
|
|
2932
|
+
objective: "Find the pricing data",
|
|
2933
|
+
role: "researcher",
|
|
2934
|
+
output_contract: "report",
|
|
2935
|
+
},
|
|
2936
|
+
{ invokingCallSite: "mainAgent" },
|
|
2937
|
+
);
|
|
2938
|
+
// A report is the default contract, so it applies no preset of its own and
|
|
2939
|
+
// leaves the child on the subagentSpawn default like any other spawn.
|
|
2940
|
+
expect(config!.overrideProfile).toBeUndefined();
|
|
2941
|
+
expect(
|
|
2942
|
+
buildSubagentSystemPrompt(
|
|
2943
|
+
{
|
|
2944
|
+
id: "sub-report",
|
|
2945
|
+
parentConversationId: "conv-1",
|
|
2946
|
+
label: "Research it",
|
|
2947
|
+
objective: "Find the pricing data",
|
|
2948
|
+
outputContract: "report",
|
|
2949
|
+
},
|
|
2950
|
+
"researcher",
|
|
2951
|
+
),
|
|
2952
|
+
).not.toContain("Output contract");
|
|
2953
|
+
});
|
|
2954
|
+
|
|
2955
|
+
test("the verdict contract reaches the child's system prompt", () => {
|
|
2956
|
+
const prompt = buildSubagentSystemPrompt(
|
|
2957
|
+
{
|
|
2958
|
+
id: "sub-verdict",
|
|
2959
|
+
parentConversationId: "conv-1",
|
|
2960
|
+
label: "Check it",
|
|
2961
|
+
objective: "Verify each acceptance criterion",
|
|
2962
|
+
outputContract: "verdict",
|
|
2963
|
+
},
|
|
2964
|
+
"researcher",
|
|
2965
|
+
);
|
|
2966
|
+
expect(prompt).toContain("- Output contract: ");
|
|
2967
|
+
expect(prompt).toContain("return PASS or FAIL plus the exact evidence");
|
|
2968
|
+
expect(prompt).toContain("CANNOT VERIFY");
|
|
2969
|
+
});
|
|
2970
|
+
|
|
2971
|
+
test("the artifact contract reaches the child's system prompt", () => {
|
|
2972
|
+
const prompt = buildSubagentSystemPrompt(
|
|
1600
2973
|
{
|
|
1601
|
-
|
|
1602
|
-
|
|
1603
|
-
|
|
2974
|
+
id: "sub-artifact",
|
|
2975
|
+
parentConversationId: "conv-1",
|
|
2976
|
+
label: "Write it",
|
|
2977
|
+
objective: "Produce the migration file",
|
|
2978
|
+
outputContract: "artifact",
|
|
1604
2979
|
},
|
|
1605
|
-
|
|
2980
|
+
"builder",
|
|
1606
2981
|
);
|
|
1607
|
-
expect(
|
|
1608
|
-
expect(result.content).toContain("Invalid subagent role");
|
|
1609
|
-
expect(result.content).toContain("nonexistent-role");
|
|
1610
|
-
expect(result.content).toContain("Must be one of");
|
|
1611
|
-
expect(result.content).toContain("general");
|
|
1612
|
-
expect(result.content).toContain("researcher");
|
|
2982
|
+
expect(prompt).toContain("Your deliverable is the artifact itself.");
|
|
1613
2983
|
});
|
|
1614
2984
|
|
|
1615
|
-
test("spawn tool definition
|
|
2985
|
+
test("spawn tool definition declares output_contract", () => {
|
|
1616
2986
|
const def = findTool("subagent_spawn");
|
|
1617
|
-
expect(def).toBeDefined();
|
|
1618
|
-
expect(def.input_schema.properties.
|
|
1619
|
-
expect(def.input_schema.properties.
|
|
1620
|
-
|
|
1621
|
-
"
|
|
1622
|
-
"
|
|
1623
|
-
"coder",
|
|
1624
|
-
"planner",
|
|
1625
|
-
"investigator",
|
|
1626
|
-
"advisor",
|
|
2987
|
+
expect(def.input_schema.properties.output_contract).toBeDefined();
|
|
2988
|
+
expect(def.input_schema.properties.output_contract.type).toBe("string");
|
|
2989
|
+
expect(def.input_schema.properties.output_contract.enum).toEqual([
|
|
2990
|
+
"report",
|
|
2991
|
+
"verdict",
|
|
2992
|
+
"artifact",
|
|
1627
2993
|
]);
|
|
1628
|
-
|
|
1629
|
-
expect(def.input_schema.required).not.toContain("role");
|
|
2994
|
+
expect(def.input_schema.required).not.toContain("output_contract");
|
|
1630
2995
|
});
|
|
1631
2996
|
});
|
|
1632
2997
|
|
|
@@ -1636,14 +3001,26 @@ describe("Subagent advisor-role consult", () => {
|
|
|
1636
3001
|
type Block = { type: string; [k: string]: unknown };
|
|
1637
3002
|
type CapturedAwait = {
|
|
1638
3003
|
config: Record<string, unknown>;
|
|
1639
|
-
opts?: {
|
|
3004
|
+
opts?: {
|
|
3005
|
+
signal?: AbortSignal;
|
|
3006
|
+
onText?: (chunk: string) => void;
|
|
3007
|
+
onProgress?: () => void;
|
|
3008
|
+
};
|
|
1640
3009
|
};
|
|
1641
3010
|
|
|
1642
3011
|
/**
|
|
1643
3012
|
* Stub `manager.spawnAndAwait` to capture the config + opts and resolve to
|
|
1644
3013
|
* `advice`. Restores the original on cleanup. Returns the captured-call ref.
|
|
3014
|
+
*
|
|
3015
|
+
* A function `advice` receives the sender the consult passed in, so a test
|
|
3016
|
+
* can drive the child's event stream (the tool activity the consult counts)
|
|
3017
|
+
* before deciding what the run returns.
|
|
1645
3018
|
*/
|
|
1646
|
-
function stubAwait(
|
|
3019
|
+
function stubAwait(
|
|
3020
|
+
advice:
|
|
3021
|
+
| string
|
|
3022
|
+
| ((send: (msg: Record<string, unknown>) => void) => Promise<string>),
|
|
3023
|
+
): {
|
|
1647
3024
|
captured: { current?: CapturedAwait };
|
|
1648
3025
|
restore: () => void;
|
|
1649
3026
|
} {
|
|
@@ -1652,12 +3029,12 @@ describe("Subagent advisor-role consult", () => {
|
|
|
1652
3029
|
const captured: { current?: CapturedAwait } = {};
|
|
1653
3030
|
manager.spawnAndAwait = (async (
|
|
1654
3031
|
config: Record<string, unknown>,
|
|
1655
|
-
|
|
3032
|
+
send: (msg: Record<string, unknown>) => void,
|
|
1656
3033
|
opts?: CapturedAwait["opts"],
|
|
1657
3034
|
) => {
|
|
1658
3035
|
captured.current = { config, opts };
|
|
1659
|
-
return typeof advice === "function" ? await advice() : advice;
|
|
1660
|
-
}) as typeof manager.spawnAndAwait;
|
|
3036
|
+
return typeof advice === "function" ? await advice(send) : advice;
|
|
3037
|
+
}) as unknown as typeof manager.spawnAndAwait;
|
|
1661
3038
|
return {
|
|
1662
3039
|
captured,
|
|
1663
3040
|
restore: () => {
|
|
@@ -1687,6 +3064,10 @@ describe("Subagent advisor-role consult", () => {
|
|
|
1687
3064
|
expect(captured.current).toBeDefined();
|
|
1688
3065
|
expect(captured.current!.config.fork).toBe(true);
|
|
1689
3066
|
expect(captured.current!.config.role).toBe("advisor");
|
|
3067
|
+
// The advisor is a ROLE, not an `LLMCallSiteEnum` value, so its usage
|
|
3068
|
+
// lands under `subagentSpawn` like any other subagent. The declared
|
|
3069
|
+
// spawn mode is what separates it from a plain fork in cost telemetry.
|
|
3070
|
+
expect(captured.current!.config.spawnMode).toBe("advisor_consult");
|
|
1690
3071
|
// Framing embeds the executor prompt as advisor system prompt context.
|
|
1691
3072
|
expect(captured.current!.config.systemPromptOverride).toContain(
|
|
1692
3073
|
"PARENT SYSTEM PROMPT",
|
|
@@ -1828,6 +3209,27 @@ describe("Subagent advisor-role consult", () => {
|
|
|
1828
3209
|
}
|
|
1829
3210
|
});
|
|
1830
3211
|
|
|
3212
|
+
test("advisor uses llm.advisorProfile, not the conversation pin", async () => {
|
|
3213
|
+
mockFindConversation = () => ({
|
|
3214
|
+
messages: [{ role: "user", content: [{ type: "text", text: "Hi" }] }],
|
|
3215
|
+
getCurrentSystemPrompt: () => "SYS",
|
|
3216
|
+
});
|
|
3217
|
+
mockConversationOverrideProfile = "quality-optimized";
|
|
3218
|
+
const { captured, restore } = stubAwait("advice");
|
|
3219
|
+
try {
|
|
3220
|
+
await executeSubagentSpawn(
|
|
3221
|
+
{ label: "Consult", objective: "x", role: "advisor" },
|
|
3222
|
+
makeContext("advisor-sess-isolation", { sendToClient: () => {} }),
|
|
3223
|
+
);
|
|
3224
|
+
expect(captured.current!.config.overrideProfile).toBe("frontier");
|
|
3225
|
+
expect(captured.current!.config.forceOverrideProfile).toBe(true);
|
|
3226
|
+
} finally {
|
|
3227
|
+
restore();
|
|
3228
|
+
mockConversationOverrideProfile = undefined;
|
|
3229
|
+
mockFindConversation = () => undefined;
|
|
3230
|
+
}
|
|
3231
|
+
});
|
|
3232
|
+
|
|
1831
3233
|
test("advisor respects an explicit inference_profile over advisorProfile", async () => {
|
|
1832
3234
|
mockFindConversation = () => ({
|
|
1833
3235
|
messages: [{ role: "user", content: [{ type: "text", text: "Hi" }] }],
|
|
@@ -1854,6 +3256,90 @@ describe("Subagent advisor-role consult", () => {
|
|
|
1854
3256
|
}
|
|
1855
3257
|
});
|
|
1856
3258
|
|
|
3259
|
+
test("an advisorProfile the catalog denies tools falls back with a note", async () => {
|
|
3260
|
+
// The advisor carries read tools now, so a profile that cannot call them
|
|
3261
|
+
// would answer from the transcript alone with nothing saying why.
|
|
3262
|
+
mockFindConversation = () => ({
|
|
3263
|
+
messages: [{ role: "user", content: [{ type: "text", text: "Hi" }] }],
|
|
3264
|
+
getCurrentSystemPrompt: () => "SYS",
|
|
3265
|
+
});
|
|
3266
|
+
setConfig("llm", { ...BASE_LLM_CONFIG, advisorProfile: "no-tool-model" });
|
|
3267
|
+
const { captured, restore } = stubAwait("Ship the data model first.");
|
|
3268
|
+
try {
|
|
3269
|
+
const result = await executeSubagentSpawn(
|
|
3270
|
+
{ label: "Consult", objective: "x", role: "advisor" },
|
|
3271
|
+
makeContext("advisor-sess-no-tools", { sendToClient: () => {} }),
|
|
3272
|
+
);
|
|
3273
|
+
// No override travels with the consult, which is what lands it on the
|
|
3274
|
+
// subagentSpawn call site's own profile while leaving usage attribution
|
|
3275
|
+
// on call_site instead of reporting a pin nobody set.
|
|
3276
|
+
expect(captured.current!.config.overrideProfile).toBeUndefined();
|
|
3277
|
+
expect(captured.current!.config.forceOverrideProfile).toBeUndefined();
|
|
3278
|
+
expect(result.content).toContain("Ship the data model first.");
|
|
3279
|
+
expect(result.content).toContain(
|
|
3280
|
+
'profile "no-tool-model" is not verified for tool calling',
|
|
3281
|
+
);
|
|
3282
|
+
} finally {
|
|
3283
|
+
restore();
|
|
3284
|
+
setConfig("llm", BASE_LLM_CONFIG);
|
|
3285
|
+
mockFindConversation = () => undefined;
|
|
3286
|
+
}
|
|
3287
|
+
});
|
|
3288
|
+
|
|
3289
|
+
test("an explicit inference_profile the catalog denies tools falls back too", async () => {
|
|
3290
|
+
mockFindConversation = () => ({
|
|
3291
|
+
messages: [{ role: "user", content: [{ type: "text", text: "Hi" }] }],
|
|
3292
|
+
getCurrentSystemPrompt: () => "SYS",
|
|
3293
|
+
});
|
|
3294
|
+
const { captured, restore } = stubAwait("advice");
|
|
3295
|
+
try {
|
|
3296
|
+
const result = await executeSubagentSpawn(
|
|
3297
|
+
{
|
|
3298
|
+
label: "Consult",
|
|
3299
|
+
objective: "x",
|
|
3300
|
+
role: "advisor",
|
|
3301
|
+
inference_profile: "no-tool-model",
|
|
3302
|
+
},
|
|
3303
|
+
makeContext("advisor-sess-no-tools-explicit", {
|
|
3304
|
+
sendToClient: () => {},
|
|
3305
|
+
}),
|
|
3306
|
+
);
|
|
3307
|
+
expect(captured.current!.config.overrideProfile).toBeUndefined();
|
|
3308
|
+
expect(result.content).toContain("is not verified for tool calling");
|
|
3309
|
+
} finally {
|
|
3310
|
+
restore();
|
|
3311
|
+
mockFindConversation = () => undefined;
|
|
3312
|
+
}
|
|
3313
|
+
});
|
|
3314
|
+
|
|
3315
|
+
test("a model the catalog does not list keeps the advisor on it", async () => {
|
|
3316
|
+
// Fail open: an unknown model is not evidence of anything, and BYOK
|
|
3317
|
+
// installs point profiles at models the catalog has never heard of.
|
|
3318
|
+
mockFindConversation = () => ({
|
|
3319
|
+
messages: [{ role: "user", content: [{ type: "text", text: "Hi" }] }],
|
|
3320
|
+
getCurrentSystemPrompt: () => "SYS",
|
|
3321
|
+
});
|
|
3322
|
+
const { captured, restore } = stubAwait("advice");
|
|
3323
|
+
try {
|
|
3324
|
+
const result = await executeSubagentSpawn(
|
|
3325
|
+
{
|
|
3326
|
+
label: "Consult",
|
|
3327
|
+
objective: "x",
|
|
3328
|
+
role: "advisor",
|
|
3329
|
+
inference_profile: "byok-unknown-model",
|
|
3330
|
+
},
|
|
3331
|
+
makeContext("advisor-sess-byok", { sendToClient: () => {} }),
|
|
3332
|
+
);
|
|
3333
|
+
expect(captured.current!.config.overrideProfile).toBe(
|
|
3334
|
+
"byok-unknown-model",
|
|
3335
|
+
);
|
|
3336
|
+
expect(result.content).toBe("advice");
|
|
3337
|
+
} finally {
|
|
3338
|
+
restore();
|
|
3339
|
+
mockFindConversation = () => undefined;
|
|
3340
|
+
}
|
|
3341
|
+
});
|
|
3342
|
+
|
|
1857
3343
|
test("advisor forwards streamed chunks to the tool's onOutput sink", async () => {
|
|
1858
3344
|
mockFindConversation = () => ({
|
|
1859
3345
|
messages: [{ role: "user", content: [{ type: "text", text: "Hi" }] }],
|
|
@@ -1867,8 +3353,8 @@ describe("Subagent advisor-role consult", () => {
|
|
|
1867
3353
|
{ label: "Consult", objective: "x", role: "advisor" },
|
|
1868
3354
|
makeContext("advisor-sess-6", { sendToClient: () => {}, onOutput }),
|
|
1869
3355
|
);
|
|
1870
|
-
// onText is a
|
|
1871
|
-
//
|
|
3356
|
+
// onText is a forwarding wrapper rather than onOutput itself, but
|
|
3357
|
+
// invoking it must still forward to onOutput.
|
|
1872
3358
|
expect(captured.current!.opts?.onText).toBeInstanceOf(Function);
|
|
1873
3359
|
captured.current!.opts?.onText?.("hello");
|
|
1874
3360
|
expect(chunks).toEqual(["hello"]);
|
|
@@ -1879,6 +3365,209 @@ describe("Subagent advisor-role consult", () => {
|
|
|
1879
3365
|
}
|
|
1880
3366
|
});
|
|
1881
3367
|
|
|
3368
|
+
test("advisor taps tool activity as deadline progress, separately from onText", async () => {
|
|
3369
|
+
mockFindConversation = () => ({
|
|
3370
|
+
messages: [{ role: "user", content: [{ type: "text", text: "Hi" }] }],
|
|
3371
|
+
getCurrentSystemPrompt: () => "SYS",
|
|
3372
|
+
});
|
|
3373
|
+
const { captured, restore } = stubAwait("advice");
|
|
3374
|
+
const chunks: string[] = [];
|
|
3375
|
+
const onOutput = (c: string) => chunks.push(c);
|
|
3376
|
+
try {
|
|
3377
|
+
await executeSubagentSpawn(
|
|
3378
|
+
{ label: "Consult", objective: "x", role: "advisor" },
|
|
3379
|
+
makeContext("advisor-sess-progress", {
|
|
3380
|
+
sendToClient: () => {},
|
|
3381
|
+
onOutput,
|
|
3382
|
+
}),
|
|
3383
|
+
);
|
|
3384
|
+
// The idle deadline is re-armed from onProgress, which the manager fires
|
|
3385
|
+
// for tool events as well as tokens. An advisor reading a file emits no
|
|
3386
|
+
// token, so without this tap the consult would be killed mid-read.
|
|
3387
|
+
expect(captured.current!.opts?.onProgress).toBeInstanceOf(Function);
|
|
3388
|
+
captured.current!.opts?.onProgress?.();
|
|
3389
|
+
// Progress is a liveness signal, not content: it must not reach the
|
|
3390
|
+
// caller's stream sink.
|
|
3391
|
+
expect(chunks).toEqual([]);
|
|
3392
|
+
expect(captured.current!.opts?.signal?.aborted).toBe(false);
|
|
3393
|
+
} finally {
|
|
3394
|
+
restore();
|
|
3395
|
+
mockFindConversation = () => undefined;
|
|
3396
|
+
}
|
|
3397
|
+
});
|
|
3398
|
+
|
|
3399
|
+
/** One child tool call, enveloped the way the manager sends it to a parent. */
|
|
3400
|
+
function toolCallEvent(toolUseId: string): Record<string, unknown> {
|
|
3401
|
+
return {
|
|
3402
|
+
type: "subagent_event",
|
|
3403
|
+
subagentId: "advisor-child",
|
|
3404
|
+
conversationId: "advisor-parent",
|
|
3405
|
+
event: {
|
|
3406
|
+
type: "tool_use_start",
|
|
3407
|
+
toolName: "file_read",
|
|
3408
|
+
toolUseId,
|
|
3409
|
+
conversationId: "advisor-child-conv",
|
|
3410
|
+
},
|
|
3411
|
+
};
|
|
3412
|
+
}
|
|
3413
|
+
|
|
3414
|
+
test("a consult that keeps reading is stopped and answers with what it has", async () => {
|
|
3415
|
+
// Tool events re-arm the idle window, so without a ceiling on tool rounds
|
|
3416
|
+
// a reading advisor blocks the user's turn until the absolute backstop.
|
|
3417
|
+
mockFindConversation = () => ({
|
|
3418
|
+
messages: [{ role: "user", content: [{ type: "text", text: "Hi" }] }],
|
|
3419
|
+
getCurrentSystemPrompt: () => "SYS",
|
|
3420
|
+
});
|
|
3421
|
+
let sawAbort = false;
|
|
3422
|
+
const { captured, restore } = stubAwait(async (send) => {
|
|
3423
|
+
for (let i = 0; i < 12; i++) {
|
|
3424
|
+
send(toolCallEvent(`tool-${i}`));
|
|
3425
|
+
}
|
|
3426
|
+
sawAbort = captured.current!.opts!.signal!.aborted;
|
|
3427
|
+
throw new SubagentAbortedError("Check the migration ordering first.");
|
|
3428
|
+
});
|
|
3429
|
+
const forwarded: unknown[] = [];
|
|
3430
|
+
try {
|
|
3431
|
+
const result = await executeSubagentSpawn(
|
|
3432
|
+
{ label: "Consult", objective: "x", role: "advisor" },
|
|
3433
|
+
makeContext("advisor-sess-tool-cap", {
|
|
3434
|
+
sendToClient: (msg: unknown) => forwarded.push(msg),
|
|
3435
|
+
}),
|
|
3436
|
+
);
|
|
3437
|
+
|
|
3438
|
+
expect(sawAbort).toBe(true);
|
|
3439
|
+
expect(result.isError).toBe(false);
|
|
3440
|
+
expect(result.content).toContain("Check the migration ordering first.");
|
|
3441
|
+
expect(result.content).toContain("used its full budget of 8 tool calls");
|
|
3442
|
+
// Counting must not swallow the child's events on their way to the client.
|
|
3443
|
+
expect(forwarded).toHaveLength(12);
|
|
3444
|
+
} finally {
|
|
3445
|
+
restore();
|
|
3446
|
+
mockFindConversation = () => undefined;
|
|
3447
|
+
}
|
|
3448
|
+
});
|
|
3449
|
+
|
|
3450
|
+
test("tool calls under the ceiling leave the consult running", async () => {
|
|
3451
|
+
mockFindConversation = () => ({
|
|
3452
|
+
messages: [{ role: "user", content: [{ type: "text", text: "Hi" }] }],
|
|
3453
|
+
getCurrentSystemPrompt: () => "SYS",
|
|
3454
|
+
});
|
|
3455
|
+
let sawAbort = true;
|
|
3456
|
+
const { captured, restore } = stubAwait(async (send) => {
|
|
3457
|
+
for (let i = 0; i < 8; i++) {
|
|
3458
|
+
send(toolCallEvent(`tool-${i}`));
|
|
3459
|
+
}
|
|
3460
|
+
// Non-tool traffic must not count against the ceiling.
|
|
3461
|
+
for (let i = 0; i < 20; i++) {
|
|
3462
|
+
send({
|
|
3463
|
+
type: "subagent_event",
|
|
3464
|
+
subagentId: "advisor-child",
|
|
3465
|
+
event: { type: "assistant_text_delta", text: "thinking" },
|
|
3466
|
+
});
|
|
3467
|
+
}
|
|
3468
|
+
sawAbort = captured.current!.opts!.signal!.aborted;
|
|
3469
|
+
return "Read enough. Ship the data model first.";
|
|
3470
|
+
});
|
|
3471
|
+
try {
|
|
3472
|
+
const result = await executeSubagentSpawn(
|
|
3473
|
+
{ label: "Consult", objective: "x", role: "advisor" },
|
|
3474
|
+
makeContext("advisor-sess-tool-cap-under", { sendToClient: () => {} }),
|
|
3475
|
+
);
|
|
3476
|
+
|
|
3477
|
+
expect(sawAbort).toBe(false);
|
|
3478
|
+
expect(result.content).toBe("Read enough. Ship the data model first.");
|
|
3479
|
+
} finally {
|
|
3480
|
+
restore();
|
|
3481
|
+
mockFindConversation = () => undefined;
|
|
3482
|
+
}
|
|
3483
|
+
});
|
|
3484
|
+
|
|
3485
|
+
test("a consult stopped at the ceiling with nothing written says so", async () => {
|
|
3486
|
+
mockFindConversation = () => ({
|
|
3487
|
+
messages: [{ role: "user", content: [{ type: "text", text: "Hi" }] }],
|
|
3488
|
+
getCurrentSystemPrompt: () => "SYS",
|
|
3489
|
+
});
|
|
3490
|
+
const { restore } = stubAwait(async (send) => {
|
|
3491
|
+
for (let i = 0; i < 9; i++) {
|
|
3492
|
+
send(toolCallEvent(`tool-${i}`));
|
|
3493
|
+
}
|
|
3494
|
+
throw new SubagentAbortedError(" ");
|
|
3495
|
+
});
|
|
3496
|
+
try {
|
|
3497
|
+
const result = await executeSubagentSpawn(
|
|
3498
|
+
{ label: "Consult", objective: "x", role: "advisor" },
|
|
3499
|
+
makeContext("advisor-sess-tool-cap-empty", { sendToClient: () => {} }),
|
|
3500
|
+
);
|
|
3501
|
+
|
|
3502
|
+
expect(result.isError).toBe(false);
|
|
3503
|
+
expect(result.content).toContain(
|
|
3504
|
+
"advisor used its full budget of 8 tool calls without writing any guidance",
|
|
3505
|
+
);
|
|
3506
|
+
// The generic degrade would say nothing about why it stopped.
|
|
3507
|
+
expect(result.content).not.toContain("advisor unavailable");
|
|
3508
|
+
} finally {
|
|
3509
|
+
restore();
|
|
3510
|
+
mockFindConversation = () => undefined;
|
|
3511
|
+
}
|
|
3512
|
+
});
|
|
3513
|
+
|
|
3514
|
+
test("a consult that both fell back on profile and hit the cap says both", async () => {
|
|
3515
|
+
// The profile note explains guidance that reads oddly, so the branch with
|
|
3516
|
+
// no guidance at all is exactly where it is most needed.
|
|
3517
|
+
mockFindConversation = () => ({
|
|
3518
|
+
messages: [{ role: "user", content: [{ type: "text", text: "Hi" }] }],
|
|
3519
|
+
getCurrentSystemPrompt: () => "SYS",
|
|
3520
|
+
});
|
|
3521
|
+
setConfig("llm", { ...BASE_LLM_CONFIG, advisorProfile: "no-tool-model" });
|
|
3522
|
+
const { restore } = stubAwait(async (send) => {
|
|
3523
|
+
for (let i = 0; i < 9; i++) {
|
|
3524
|
+
send(toolCallEvent(`tool-${i}`));
|
|
3525
|
+
}
|
|
3526
|
+
throw new SubagentAbortedError(" ");
|
|
3527
|
+
});
|
|
3528
|
+
try {
|
|
3529
|
+
const result = await executeSubagentSpawn(
|
|
3530
|
+
{ label: "Consult", objective: "x", role: "advisor" },
|
|
3531
|
+
makeContext("advisor-sess-tool-cap-note", { sendToClient: () => {} }),
|
|
3532
|
+
);
|
|
3533
|
+
|
|
3534
|
+
expect(result.content).toContain(
|
|
3535
|
+
"advisor used its full budget of 8 tool calls without writing any guidance",
|
|
3536
|
+
);
|
|
3537
|
+
expect(result.content).toContain(
|
|
3538
|
+
'profile "no-tool-model" is not verified for tool calling',
|
|
3539
|
+
);
|
|
3540
|
+
} finally {
|
|
3541
|
+
restore();
|
|
3542
|
+
setConfig("llm", BASE_LLM_CONFIG);
|
|
3543
|
+
mockFindConversation = () => undefined;
|
|
3544
|
+
}
|
|
3545
|
+
});
|
|
3546
|
+
|
|
3547
|
+
test("advisor consult runs under the owner-gated read-only guard", async () => {
|
|
3548
|
+
mockFindConversation = () => ({
|
|
3549
|
+
messages: [{ role: "user", content: [{ type: "text", text: "Hi" }] }],
|
|
3550
|
+
getCurrentSystemPrompt: () => "SYS",
|
|
3551
|
+
});
|
|
3552
|
+
const { captured, restore } = stubAwait("advice");
|
|
3553
|
+
try {
|
|
3554
|
+
await executeSubagentSpawn(
|
|
3555
|
+
{ label: "Consult", objective: "x", role: "advisor" },
|
|
3556
|
+
makeContext("advisor-sess-readonly", { sendToClient: () => {} }),
|
|
3557
|
+
);
|
|
3558
|
+
// Without this the advisor's read-only guarantee is a list of NAMES, and
|
|
3559
|
+
// a workspace tool registered as `file_read` would be handed to it to
|
|
3560
|
+
// execute. The owner check that names cannot express rides on the role,
|
|
3561
|
+
// so it reaches every path that projects the role rather than only the
|
|
3562
|
+
// spawn site that remembered to ask for it.
|
|
3563
|
+
expect(captured.current!.config.role).toBe("advisor");
|
|
3564
|
+
expect(SUBAGENT_ROLE_REGISTRY.advisor.denySideEffects).toBe(true);
|
|
3565
|
+
} finally {
|
|
3566
|
+
restore();
|
|
3567
|
+
mockFindConversation = () => undefined;
|
|
3568
|
+
}
|
|
3569
|
+
});
|
|
3570
|
+
|
|
1882
3571
|
test("advisor degrades benignly when the consult throws (incl. depth limit)", async () => {
|
|
1883
3572
|
mockFindConversation = () => ({
|
|
1884
3573
|
messages: [{ role: "user", content: [{ type: "text", text: "Hi" }] }],
|
|
@@ -1968,31 +3657,38 @@ describe("Subagent advisor-role consult", () => {
|
|
|
1968
3657
|
});
|
|
1969
3658
|
});
|
|
1970
3659
|
|
|
1971
|
-
// ── Advisor role
|
|
3660
|
+
// ── Advisor role read-only enforcement ──────────────────────────────
|
|
1972
3661
|
|
|
1973
|
-
describe("Advisor role is
|
|
1974
|
-
test("the advisor
|
|
1975
|
-
//
|
|
1976
|
-
//
|
|
1977
|
-
//
|
|
3662
|
+
describe("Advisor role is read-only", () => {
|
|
3663
|
+
test("the advisor allowlist admits its read tools and nothing write-capable", async () => {
|
|
3664
|
+
// The allowlist is a ceiling, not a hint: build a resolveTools callback
|
|
3665
|
+
// with the advisor's allowlist and confirm a write-capable tool is dropped
|
|
3666
|
+
// while its read tools survive.
|
|
1978
3667
|
const { createResolveToolsCallback } =
|
|
1979
3668
|
await import("../daemon/conversation-tool-setup.js");
|
|
1980
3669
|
const { SUBAGENT_ROLE_REGISTRY } = await import("../subagent/types.js");
|
|
1981
3670
|
const advisorAllowed = SUBAGENT_ROLE_REGISTRY.advisor.allowedTools;
|
|
1982
|
-
expect(advisorAllowed).toEqual([]);
|
|
3671
|
+
expect(advisorAllowed).toEqual(["file_read", "file_list", "code_search"]);
|
|
1983
3672
|
|
|
1984
3673
|
const toolDefs = [
|
|
1985
|
-
|
|
1986
|
-
|
|
1987
|
-
|
|
3674
|
+
"bash",
|
|
3675
|
+
"file_write",
|
|
3676
|
+
"file_read",
|
|
3677
|
+
"file_list",
|
|
3678
|
+
"code_search",
|
|
3679
|
+
"recall",
|
|
3680
|
+
].map((name) => ({
|
|
3681
|
+
name,
|
|
3682
|
+
description: "",
|
|
3683
|
+
input_schema: { type: "object" },
|
|
3684
|
+
}));
|
|
1988
3685
|
const ctx = {
|
|
1989
3686
|
skillProjectionState: new Map<string, string>(),
|
|
1990
3687
|
skillProjectionCache: new Map(),
|
|
1991
3688
|
toolsDisabledDepth: 0,
|
|
1992
|
-
// The advisor role applies `new Set(allowedTools)` — empty Set here.
|
|
1993
3689
|
subagentAllowedTools: new Set<string>(advisorAllowed),
|
|
1994
3690
|
// Default (absent) gate mode is "wire": the allowlist filters the wire
|
|
1995
|
-
// tool list
|
|
3691
|
+
// tool list.
|
|
1996
3692
|
isSubagent: true,
|
|
1997
3693
|
} as unknown as Parameters<typeof createResolveToolsCallback>[1];
|
|
1998
3694
|
|
|
@@ -2001,13 +3697,20 @@ describe("Advisor role is tool-less", () => {
|
|
|
2001
3697
|
ctx,
|
|
2002
3698
|
);
|
|
2003
3699
|
expect(resolve).toBeDefined();
|
|
2004
|
-
const
|
|
2005
|
-
expect(
|
|
2006
|
-
|
|
2007
|
-
|
|
2008
|
-
|
|
2009
|
-
|
|
2010
|
-
|
|
3700
|
+
const resolvedNames = resolve!([]).map((t) => t.name);
|
|
3701
|
+
expect(resolvedNames.sort()).toEqual([
|
|
3702
|
+
"code_search",
|
|
3703
|
+
"file_list",
|
|
3704
|
+
"file_read",
|
|
3705
|
+
]);
|
|
3706
|
+
// The per-turn execution gate matches the wire list.
|
|
3707
|
+
const gate = (ctx as unknown as { allowedToolNames?: Set<string> })
|
|
3708
|
+
.allowedToolNames;
|
|
3709
|
+
expect(gate?.has("file_read")).toBe(true);
|
|
3710
|
+
expect(gate?.has("bash")).toBe(false);
|
|
3711
|
+
// `recall` reaches memory and prior conversations, which the consult's
|
|
3712
|
+
// contract excludes, so the allowlist does not admit it either.
|
|
3713
|
+
expect(gate?.has("recall")).toBe(false);
|
|
2011
3714
|
});
|
|
2012
3715
|
});
|
|
2013
3716
|
|
|
@@ -2075,6 +3778,58 @@ describe("subagent tools — model-input schema validation", () => {
|
|
|
2075
3778
|
});
|
|
2076
3779
|
});
|
|
2077
3780
|
|
|
3781
|
+
// ── Read tool misuse redirects ──────────────────────────────────────
|
|
3782
|
+
|
|
3783
|
+
describe("Subagent read tool misuse", () => {
|
|
3784
|
+
for (const key of ["path", "file", "filename"]) {
|
|
3785
|
+
test(`read redirects a "${key}" param to file_read`, async () => {
|
|
3786
|
+
const result = await executeSubagentRead(
|
|
3787
|
+
{ [key]: "/tmp/notes.md" },
|
|
3788
|
+
makeContext("misuse-sess"),
|
|
3789
|
+
);
|
|
3790
|
+
expect(result.isError).toBe(true);
|
|
3791
|
+
expect(result.content).toContain("it does not read files");
|
|
3792
|
+
expect(result.content).toContain("Use file_read for files");
|
|
3793
|
+
expect(result.content).toContain("Pass subagent_id or label here");
|
|
3794
|
+
});
|
|
3795
|
+
}
|
|
3796
|
+
|
|
3797
|
+
for (const key of ["subagentId", "agent_id"]) {
|
|
3798
|
+
test(`read names "${key}" as an unknown parameter`, async () => {
|
|
3799
|
+
const result = await executeSubagentRead(
|
|
3800
|
+
{ [key]: "some-id" },
|
|
3801
|
+
makeContext("misuse-sess"),
|
|
3802
|
+
);
|
|
3803
|
+
expect(result.isError).toBe(true);
|
|
3804
|
+
expect(result.content).toBe(
|
|
3805
|
+
"Unknown parameter. Use subagent_id (snake_case) or label.",
|
|
3806
|
+
);
|
|
3807
|
+
});
|
|
3808
|
+
}
|
|
3809
|
+
|
|
3810
|
+
test("a file-reader key wins over the misnamed-id message", async () => {
|
|
3811
|
+
const result = await executeSubagentRead(
|
|
3812
|
+
{ path: "/tmp/notes.md", subagentId: "some-id" },
|
|
3813
|
+
makeContext("misuse-sess"),
|
|
3814
|
+
);
|
|
3815
|
+
expect(result.isError).toBe(true);
|
|
3816
|
+
expect(result.content).toContain("it does not read files");
|
|
3817
|
+
});
|
|
3818
|
+
|
|
3819
|
+
test("a correctly named read is untouched by the misuse checks", async () => {
|
|
3820
|
+
const manager = getSubagentManager();
|
|
3821
|
+
const subagentId = "misuse-untouched-1";
|
|
3822
|
+
injectSubagent(manager, subagentId, "misuse-sess", "running");
|
|
3823
|
+
|
|
3824
|
+
const result = await executeSubagentRead(
|
|
3825
|
+
{ subagent_id: subagentId },
|
|
3826
|
+
makeContext("misuse-sess"),
|
|
3827
|
+
);
|
|
3828
|
+
expect(result.isError).toBe(false);
|
|
3829
|
+
expect(result.content).toContain("still running");
|
|
3830
|
+
});
|
|
3831
|
+
});
|
|
3832
|
+
|
|
2078
3833
|
// ── Durable fallback past the rehydration bound ─────────────────────
|
|
2079
3834
|
|
|
2080
3835
|
/** Mirrors `MAX_REHYDRATED_TERMINAL_RECORDS` in `subagent/manager.ts`. */
|
|
@@ -2091,7 +3846,7 @@ function terminalRecord(over: Partial<SubagentRecord>): SubagentRecord {
|
|
|
2091
3846
|
conversationId: "conv-seed",
|
|
2092
3847
|
label: "seed",
|
|
2093
3848
|
objective: "seeded objective",
|
|
2094
|
-
role: "
|
|
3849
|
+
role: "builder",
|
|
2095
3850
|
isFork: false,
|
|
2096
3851
|
sendResultToUser: true,
|
|
2097
3852
|
parentToolUseId: null,
|
|
@@ -2194,7 +3949,45 @@ describe("Subagent tools past the startup rehydration bound", () => {
|
|
|
2194
3949
|
makeContext(beyondCapParent),
|
|
2195
3950
|
);
|
|
2196
3951
|
expect(result.isError).toBe(false);
|
|
2197
|
-
|
|
3952
|
+
// The counters live with the in-memory entry, which the bound dropped, so
|
|
3953
|
+
// the footer says unavailable rather than reporting zero tool calls.
|
|
3954
|
+
expect(result.content).toBe(
|
|
3955
|
+
"Output from beyond the bound\n\n[stats: unavailable (tool counters are not retained for this subagent)]",
|
|
3956
|
+
);
|
|
3957
|
+
} finally {
|
|
3958
|
+
mockGetMessages = () => null;
|
|
3959
|
+
}
|
|
3960
|
+
});
|
|
3961
|
+
|
|
3962
|
+
test("read reports unavailable for a subagent the rehydration DID load", async () => {
|
|
3963
|
+
// This entry is inside the rehydration bound, so the restart put it back
|
|
3964
|
+
// in the manager, but its counters died with the process that ran it, and
|
|
3965
|
+
// no rehydrated entry can ever have them. Manager membership is therefore
|
|
3966
|
+
// not the signal; `rehydrated` is.
|
|
3967
|
+
const filler = `cap-filler-${REHYDRATION_CAP - 1}`;
|
|
3968
|
+
expect(manager.getState(filler)).toBeDefined();
|
|
3969
|
+
|
|
3970
|
+
mockGetMessages = (convId: string) =>
|
|
3971
|
+
convId === `conv-${filler}`
|
|
3972
|
+
? [
|
|
3973
|
+
{
|
|
3974
|
+
role: "assistant",
|
|
3975
|
+
content: [
|
|
3976
|
+
{ type: "text", text: "Output from before the restart" },
|
|
3977
|
+
],
|
|
3978
|
+
},
|
|
3979
|
+
]
|
|
3980
|
+
: null;
|
|
3981
|
+
|
|
3982
|
+
try {
|
|
3983
|
+
const result = await executeSubagentRead(
|
|
3984
|
+
{ subagent_id: filler },
|
|
3985
|
+
makeContext(beyondCapParent),
|
|
3986
|
+
);
|
|
3987
|
+
expect(result.isError).toBe(false);
|
|
3988
|
+
expect(result.content).toBe(
|
|
3989
|
+
"Output from before the restart\n\n[stats: unavailable (tool counters are not retained for this subagent)]",
|
|
3990
|
+
);
|
|
2198
3991
|
} finally {
|
|
2199
3992
|
mockGetMessages = () => null;
|
|
2200
3993
|
}
|