@vellumai/assistant 0.8.10 → 0.8.11-staging.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/bun.lock +62 -1
- package/docs/workspace-tools.md +196 -0
- package/examples/plugins/echo/README.md +3 -3
- package/knip.json +1 -0
- package/openapi.yaml +460 -128
- package/package.json +2 -1
- package/scripts/build-plugin-api.ts +299 -0
- package/src/__tests__/agent-loop-callsite-precedence.test.ts +7 -0
- package/src/__tests__/agent-loop-compaction-events.test.ts +197 -0
- package/src/__tests__/agent-loop-exit-reason.test.ts +93 -96
- package/src/__tests__/agent-loop-mutable-latest-user-message.test.ts +2 -0
- package/src/__tests__/agent-loop-output-hooks.test.ts +274 -1
- package/src/__tests__/agent-loop-override-profile.test.ts +3 -0
- package/src/__tests__/agent-loop-provider-error-recording.test.ts +4 -0
- package/src/__tests__/agent-loop-thinking.test.ts +4 -0
- package/src/__tests__/agent-loop.test.ts +578 -5
- package/src/__tests__/agent-wake-disk-pressure-callsite.test.ts +0 -1
- package/src/__tests__/approval-cascade.test.ts +1 -0
- package/src/__tests__/background-workers-disk-pressure.test.ts +0 -2
- package/src/__tests__/btw-routes.test.ts +0 -1
- package/src/__tests__/build-persisted-content.test.ts +75 -1
- package/src/__tests__/catalog-install-normalize.test.ts +141 -0
- package/src/__tests__/ces-startup-timeout.test.ts +60 -0
- package/src/__tests__/compaction-events.test.ts +1 -0
- package/src/__tests__/config-managed-gemini-defaults.test.ts +2 -46
- package/src/__tests__/context-overflow-reducer.test.ts +264 -124
- package/src/__tests__/context-window-manager-overflow-rung.test.ts +351 -0
- package/src/__tests__/conversation-abort-tool-results.test.ts +1 -1
- package/src/__tests__/conversation-agent-loop-inference-profile.test.ts +13 -5
- package/src/__tests__/conversation-agent-loop-overflow.test.ts +284 -455
- package/src/__tests__/conversation-agent-loop.test.ts +131 -551
- package/src/__tests__/conversation-app-control-instantiation.test.ts +13 -0
- package/src/__tests__/conversation-confirmation-signals.test.ts +1 -0
- package/src/__tests__/conversation-fork-crud.test.ts +259 -0
- package/src/__tests__/conversation-history-web-search.test.ts +1 -1
- package/src/__tests__/conversation-lifecycle.test.ts +257 -1
- package/src/__tests__/conversation-process-callsite.test.ts +1 -0
- package/src/__tests__/conversation-provider-retry-repair.test.ts +38 -377
- package/src/__tests__/conversation-queue.test.ts +1 -39
- package/src/__tests__/conversation-runtime-assembly.test.ts +119 -8
- package/src/__tests__/conversation-skill-tools.test.ts +491 -5
- package/src/__tests__/conversation-slash-queue.test.ts +1 -1
- package/src/__tests__/conversation-slash-unknown.test.ts +1 -0
- package/src/__tests__/conversation-speed-override.test.ts +1 -0
- package/src/__tests__/conversation-store.test.ts +74 -0
- package/src/__tests__/conversation-surfaces-app-control.test.ts +4 -1
- package/src/__tests__/conversation-tool-setup-attribution.test.ts +323 -0
- package/src/__tests__/conversation-tool-setup-tools-disabled.test.ts +34 -0
- package/src/__tests__/conversation-workspace-cache-state.test.ts +1 -0
- package/src/__tests__/conversation-workspace-injection.test.ts +1 -1
- package/src/__tests__/conversation-workspace-tool-tracking.test.ts +1 -0
- package/src/__tests__/corrected-target.test.ts +93 -0
- package/src/__tests__/credential-execution-feature-gates.test.ts +3 -5
- package/src/__tests__/credential-execution-tools.test.ts +23 -11
- package/src/__tests__/credential-security-invariants.test.ts +6 -1
- package/src/__tests__/db-schedule-syntax-migration.test.ts +80 -0
- package/src/__tests__/device-id.test.ts +70 -1
- package/src/__tests__/embedding-managed-proxy-selection.test.ts +6 -40
- package/src/__tests__/empty-response-hook.test.ts +242 -66
- package/src/__tests__/external-plugin-loader.test.ts +0 -31
- package/src/__tests__/get-skill-detail-audit.test.ts +43 -1
- package/src/__tests__/guardian-routing-invariants.test.ts +91 -0
- package/src/__tests__/history-repair-hook.test.ts +228 -3
- package/src/__tests__/host-app-control-proxy.test.ts +45 -0
- package/src/__tests__/host-browser-proxy.test.ts +254 -9
- package/src/__tests__/identity-routes.test.ts +1 -0
- package/src/__tests__/image-recovery-hook.test.ts +387 -0
- package/src/__tests__/injector-chain.test.ts +5 -4
- package/src/__tests__/injector-v3-suppression.test.ts +373 -47
- package/src/__tests__/intent-routing.test.ts +7 -0
- package/src/__tests__/memory-retrieval-hook.test.ts +117 -15
- package/src/__tests__/notification-decision-strategy.test.ts +3 -3
- package/src/__tests__/oauth-store.test.ts +0 -85
- package/src/__tests__/{context-overflow-policy.test.ts → overflow-policy.test.ts} +1 -1
- package/src/__tests__/parallel-tool.benchmark.test.ts +4 -0
- package/src/__tests__/persist-unsendable-image-downscale.test.ts +29 -9
- package/src/__tests__/persist-unsendable-image.test.ts +4 -4
- package/src/__tests__/persistence-secret-redaction.test.ts +78 -0
- package/src/__tests__/plugin-bootstrap.test.ts +82 -73
- package/src/__tests__/plugin-tool-contribution.test.ts +7 -4
- package/src/__tests__/plugin-types.test.ts +0 -8
- package/src/__tests__/provider-catalog-visibility.test.ts +1 -9
- package/src/__tests__/prune-old-conversations-job.test.ts +99 -0
- package/src/__tests__/registry.test.ts +240 -1
- package/src/__tests__/require-fresh-approval.test.ts +3 -0
- package/src/__tests__/schedule-routes.test.ts +116 -1
- package/src/__tests__/schedule-store.test.ts +28 -0
- package/src/__tests__/schedule-tools.test.ts +94 -1
- package/src/__tests__/server-history-render.test.ts +39 -0
- package/src/__tests__/skill-projection-feature-flag.test.ts +13 -0
- package/src/__tests__/skill-projection.benchmark.test.ts +25 -7
- package/src/__tests__/skills.test.ts +202 -0
- package/src/__tests__/slim-skill-category.test.ts +195 -0
- package/src/__tests__/strip-memory-injections.test.ts +33 -39
- package/src/__tests__/test-support/tool-invocation-seed.ts +79 -0
- package/src/__tests__/title-generate-hook.test.ts +9 -7
- package/src/__tests__/tool-audit-listener.test.ts +264 -1
- package/src/__tests__/tool-error-hook.test.ts +4 -3
- package/src/__tests__/tool-execution-pipeline.benchmark.test.ts +1 -0
- package/src/__tests__/tool-executor-lifecycle-events.test.ts +273 -0
- package/src/__tests__/tool-result-truncate-hook.test.ts +1 -0
- package/src/__tests__/tool-start-timestamp.test.ts +218 -0
- package/src/__tests__/tools-get-route.test.ts +202 -0
- package/src/__tests__/workspace-tool-loader.test.ts +319 -0
- package/src/__tests__/workspace-tools-watcher-flag.test.ts +70 -0
- package/src/agent/loop.ts +569 -319
- package/src/api/events/tool-result.ts +9 -0
- package/src/api/events/tool-use-start.ts +7 -0
- package/src/api/index.ts +10 -0
- package/src/api/responses/conversation-message.ts +135 -27
- package/src/api/responses/memory-v3-selection-log.ts +4 -4
- package/src/approvals/guardian-request-resolvers.ts +26 -0
- package/src/browser-session/backends/host-bridge.ts +29 -0
- package/src/browser-session/index.ts +1 -0
- package/src/browser-session/types.ts +5 -1
- package/src/cli/commands/__tests__/schedules.test.ts +62 -4
- package/src/cli/commands/__tests__/skills.test.ts +53 -0
- package/src/cli/commands/channel-verification-sessions.ts +6 -6
- package/src/cli/commands/inference-providers.ts +0 -8
- package/src/cli/commands/plugins.ts +2 -2
- package/src/cli/commands/schedules.ts +27 -4
- package/src/cli/commands/skills.ts +187 -146
- package/src/cli/commands/tools.ts +106 -0
- package/src/cli/lib/__tests__/install-from-github.test.ts +256 -328
- package/src/cli/lib/__tests__/plugin-catalog-cache.test.ts +6 -2
- package/src/cli/lib/__tests__/plugin-details.test.ts +10 -16
- package/src/cli/lib/__tests__/plugin-marketplace.test.ts +2 -2
- package/src/cli/lib/__tests__/search-plugins.test.ts +145 -240
- package/src/cli/lib/install-from-github.ts +187 -117
- package/src/cli/lib/plugin-catalog-cache.ts +9 -9
- package/src/cli/lib/plugin-details.ts +38 -68
- package/src/cli/lib/plugin-marketplace.ts +42 -14
- package/src/cli/lib/search-plugins.ts +29 -129
- package/src/cli/program.ts +2 -0
- package/src/config/bundled-skills/acp/SKILL.md +1 -0
- package/src/config/bundled-skills/app-builder/SKILL.md +1 -0
- package/src/config/bundled-skills/app-control/SKILL.md +1 -0
- package/src/config/bundled-skills/computer-use/SKILL.md +1 -0
- package/src/config/bundled-skills/contacts/SKILL.md +1 -0
- package/src/config/bundled-skills/document-editor/SKILL.md +1 -0
- package/src/config/bundled-skills/followups/SKILL.md +1 -0
- package/src/config/bundled-skills/image-studio/SKILL.md +1 -0
- package/src/config/bundled-skills/media-processing/SKILL.md +1 -0
- package/src/config/bundled-skills/messaging/SKILL.md +1 -0
- package/src/config/bundled-skills/phone-calls/SKILL.md +1 -0
- package/src/config/bundled-skills/playbooks/SKILL.md +1 -0
- package/src/config/bundled-skills/schedule/SKILL.md +1 -0
- package/src/config/bundled-skills/schedule/TOOLS.json +11 -3
- package/src/config/bundled-skills/sequences/SKILL.md +1 -0
- package/src/config/bundled-skills/settings/SKILL.md +1 -0
- package/src/config/bundled-skills/skill-management/SKILL.md +101 -1
- package/src/config/bundled-skills/subagent/SKILL.md +1 -0
- package/src/config/bundled-skills/transcribe/SKILL.md +1 -0
- package/src/config/env-registry.ts +23 -0
- package/src/config/feature-flag-registry.json +17 -81
- package/src/config/loader.ts +5 -22
- package/src/config/schema.ts +2 -0
- package/src/config/schemas/__tests__/compaction-logs.test.ts +56 -0
- package/src/config/schemas/__tests__/memory-v2.test.ts +0 -1
- package/src/config/schemas/__tests__/memory-v3.test.ts +61 -1
- package/src/config/schemas/compaction-logs.ts +79 -0
- package/src/config/schemas/memory-v2.ts +0 -8
- package/src/config/schemas/memory-v3.ts +104 -33
- package/src/config/seed-inference-profiles.ts +1 -1
- package/src/config/skills.ts +117 -47
- package/src/context/compactor.ts +11 -0
- package/src/context/strip-injections.ts +38 -4
- package/src/credential-execution/feature-gates.ts +0 -21
- package/src/credential-execution/startup-timeout.ts +32 -4
- package/src/daemon/__tests__/conversation-tool-setup-exclude.test.ts +18 -0
- package/src/daemon/conversation-agent-loop-handlers.ts +140 -91
- package/src/daemon/conversation-agent-loop.ts +123 -658
- package/src/daemon/conversation-error.ts +6 -33
- package/src/daemon/conversation-lifecycle.ts +1 -1
- package/src/daemon/conversation-runtime-assembly.ts +183 -22
- package/src/daemon/conversation-skill-tools.ts +137 -8
- package/src/daemon/conversation-slash.ts +0 -14
- package/src/daemon/conversation-store.ts +2 -19
- package/src/daemon/conversation-tool-setup.ts +87 -1
- package/src/daemon/conversation.ts +119 -50
- package/src/daemon/external-plugins-bootstrap.ts +36 -95
- package/src/daemon/handlers/config-channels.ts +11 -2
- package/src/daemon/handlers/shared.ts +9 -1
- package/src/daemon/handlers/skills.ts +10 -4
- package/src/daemon/host-app-control-proxy.ts +72 -57
- package/src/daemon/host-browser-proxy.ts +117 -22
- package/src/daemon/lifecycle.ts +34 -5
- package/src/daemon/message-protocol.ts +0 -7
- package/src/daemon/message-types/schedules.ts +1 -0
- package/src/daemon/message-types/skills.ts +17 -0
- package/src/daemon/providers-setup.ts +3 -0
- package/src/daemon/server.ts +3 -3
- package/src/daemon/tool-setup-types.ts +9 -3
- package/src/daemon/trust-context.ts +23 -0
- package/src/daemon/workspace-tools-watcher.ts +324 -0
- package/src/events/tool-audit-listener.ts +78 -16
- package/src/events/tool-metrics-listener.ts +2 -5
- package/src/memory/__tests__/compaction-log-writer-clickhouse.test.ts +227 -0
- package/src/memory/__tests__/conversation-queries.test.ts +176 -0
- package/src/memory/__tests__/jobs-worker-v2-schedule.test.ts +20 -32
- package/src/memory/compaction-log-writer-clickhouse.ts +418 -0
- package/src/memory/conversation-crud.ts +134 -3
- package/src/memory/conversation-queries.ts +64 -7
- package/src/memory/db-init.ts +14 -0
- package/src/memory/embedding-backend.test.ts +130 -1
- package/src/memory/embedding-backend.ts +79 -106
- package/src/memory/embedding-gemini.ts +5 -0
- package/src/memory/graph/__tests__/conversation-graph-memory-v2-routing.test.ts +12 -0
- package/src/memory/graph/__tests__/handle-remember-v2.test.ts +19 -0
- package/src/memory/graph/conversation-graph-memory.ts +36 -25
- package/src/memory/graph/tool-handlers.ts +3 -0
- package/src/memory/job-handlers/cleanup.ts +3 -1
- package/src/memory/jobs-store.ts +0 -28
- package/src/memory/jobs-worker.ts +10 -22
- package/src/memory/memory-marker.ts +29 -0
- package/src/memory/memory-retrospective-startup-cleanup.ts +1 -1
- package/src/memory/migrations/268-add-memory-v3-selections.ts +6 -0
- package/src/memory/migrations/270-schedule-description.ts +36 -0
- package/src/memory/migrations/275-tool-invocations-add-skill-id.test.ts +81 -0
- package/src/memory/migrations/275-tool-invocations-add-skill-id.ts +20 -0
- package/src/memory/migrations/276-tool-invocations-created-at-id-index.test.ts +68 -0
- package/src/memory/migrations/276-tool-invocations-created-at-id-index.ts +20 -0
- package/src/memory/migrations/277-add-memory-v3-ever-injected.ts +29 -0
- package/src/memory/migrations/278-tool-invocations-telemetry-columns.test.ts +96 -0
- package/src/memory/migrations/278-tool-invocations-telemetry-columns.ts +39 -0
- package/src/memory/migrations/279-create-skill-loaded-events.test.ts +84 -0
- package/src/memory/migrations/279-create-skill-loaded-events.ts +26 -0
- package/src/memory/migrations/280-conversations-surfaced-at.test.ts +88 -0
- package/src/memory/migrations/280-conversations-surfaced-at.ts +24 -0
- package/src/memory/migrations/index.ts +10 -0
- package/src/memory/migrations/registry.ts +8 -0
- package/src/memory/schema/conversations.ts +16 -0
- package/src/memory/schema/infrastructure.ts +26 -0
- package/src/memory/skill-loaded-events-store.test.ts +160 -0
- package/src/memory/skill-loaded-events-store.ts +95 -0
- package/src/memory/tool-executed-events-store.test.ts +219 -0
- package/src/memory/tool-executed-events-store.ts +102 -0
- package/src/memory/tool-usage-store.ts +15 -3
- package/src/memory/v2/__tests__/consolidation-job.test.ts +117 -12
- package/src/memory/v2/__tests__/consolidation-prompt-flag-gating-guard.test.ts +189 -0
- package/src/memory/v2/__tests__/injected-block-slugs.test.ts +90 -0
- package/src/memory/v2/__tests__/page-store.test.ts +33 -0
- package/src/memory/v2/__tests__/prompts-consolidation.test.ts +88 -15
- package/src/memory/v2/activation-store.ts +50 -1
- package/src/memory/v2/consolidation-job.ts +113 -29
- package/src/memory/v2/injected-block-slugs.ts +79 -0
- package/src/memory/v2/injection.ts +6 -1
- package/src/memory/v2/prompts/consolidation.ts +414 -13
- package/src/memory/v2/router.ts +2 -28
- package/src/memory/v2/static-context.ts +1 -1
- package/src/memory/v2/types.ts +16 -0
- package/src/notifications/__tests__/copy-composer.test.ts +244 -0
- package/src/notifications/access-request-copy.ts +298 -0
- package/src/notifications/adapters/slack.ts +3 -3
- package/src/notifications/adapters/telegram.ts +2 -1
- package/src/notifications/copy-composer.ts +49 -267
- package/src/notifications/decision-engine.ts +16 -35
- package/src/notifications/home-feed-side-effect.ts +1 -6
- package/src/oauth/oauth-store.ts +0 -9
- package/src/permissions/checker.test.ts +83 -1
- package/src/permissions/checker.ts +25 -2
- package/src/permissions/gateway-threshold-reader.test.ts +182 -0
- package/src/permissions/gateway-threshold-reader.ts +80 -0
- package/src/platform/client.ts +1 -3
- package/src/platform/feature-gate.ts +3 -12
- package/src/plugin-api/constants.ts +4 -2
- package/src/plugin-api/index.ts +63 -11
- package/src/plugin-api/types.ts +236 -71
- package/src/plugins/defaults/compaction/compact.ts +66 -2
- package/src/plugins/defaults/compaction/context-overflow-reducer.ts +240 -32
- package/src/plugins/defaults/compaction/corrected-target.ts +53 -0
- package/src/{daemon/context-overflow-policy.ts → plugins/defaults/compaction/overflow-policy.ts} +1 -1
- package/src/plugins/defaults/compaction/window-manager.ts +303 -1
- package/src/plugins/defaults/empty-response/hooks/post-model-call.ts +173 -0
- package/src/plugins/defaults/empty-response/hooks/stop.ts +11 -115
- package/src/plugins/defaults/empty-response/nudge-state-store.ts +46 -0
- package/src/plugins/defaults/history-repair/hooks/post-model-call.ts +50 -0
- package/src/plugins/defaults/history-repair/hooks/stop.ts +22 -0
- package/src/plugins/defaults/history-repair/repair-state-store.ts +51 -0
- package/src/plugins/defaults/history-repair/terminal.ts +39 -2
- package/src/plugins/defaults/image-recovery/detect.ts +25 -0
- package/src/plugins/defaults/image-recovery/hooks/post-model-call.ts +73 -0
- package/src/plugins/defaults/image-recovery/hooks/stop.ts +22 -0
- package/src/plugins/defaults/image-recovery/image-recovery-state-store.ts +48 -0
- package/src/plugins/defaults/image-recovery/package.json +14 -0
- package/src/{daemon/persist-unsendable-image.ts → plugins/defaults/image-recovery/recover.ts} +67 -14
- package/src/plugins/defaults/index.ts +71 -5
- package/src/plugins/defaults/memory-retrieval/hooks/post-compact.ts +76 -112
- package/src/plugins/defaults/memory-retrieval/hooks/{user-prompt-submit-temp.ts → user-prompt-submit.ts} +83 -74
- package/src/plugins/defaults/memory-retrieval/injector-chain.ts +14 -8
- package/src/plugins/defaults/memory-retrieval/injectors.ts +2 -18
- package/src/plugins/defaults/memory-retrieval/package.json +14 -0
- package/src/plugins/defaults/memory-v3-shadow/__tests__/carry-integration.test.ts +1157 -0
- package/src/plugins/defaults/memory-v3-shadow/__tests__/injection.test.ts +683 -0
- package/src/plugins/defaults/memory-v3-shadow/__tests__/live-integration.test.ts +161 -140
- package/src/plugins/defaults/memory-v3-shadow/__tests__/maintain-job.test.ts +160 -0
- package/src/plugins/defaults/memory-v3-shadow/__tests__/orchestrate.test.ts +335 -316
- package/src/plugins/defaults/memory-v3-shadow/__tests__/pool-select.test.ts +145 -53
- package/src/plugins/defaults/memory-v3-shadow/__tests__/render-injection.test.ts +38 -1
- package/src/plugins/defaults/memory-v3-shadow/__tests__/selection-log-store.test.ts +18 -8
- package/src/plugins/defaults/memory-v3-shadow/__tests__/shadow-integration.test.ts +112 -71
- package/src/plugins/defaults/memory-v3-shadow/__tests__/shadow-plugin.test.ts +192 -51
- package/src/plugins/defaults/memory-v3-shadow/__tests__/types.test.ts +4 -16
- package/src/plugins/defaults/memory-v3-shadow/card.test.ts +173 -0
- package/src/plugins/defaults/memory-v3-shadow/card.ts +116 -0
- package/src/plugins/defaults/memory-v3-shadow/core-set.test.ts +104 -0
- package/src/plugins/defaults/memory-v3-shadow/core-set.ts +59 -0
- package/src/plugins/defaults/memory-v3-shadow/ever-injected-store.test.ts +305 -0
- package/src/plugins/defaults/memory-v3-shadow/ever-injected-store.ts +278 -0
- package/src/plugins/defaults/memory-v3-shadow/hot-set.test.ts +138 -0
- package/src/plugins/defaults/memory-v3-shadow/hot-set.ts +85 -0
- package/src/plugins/defaults/memory-v3-shadow/injector.ts +331 -24
- package/src/plugins/defaults/memory-v3-shadow/maintain-job.ts +119 -13
- package/src/plugins/defaults/memory-v3-shadow/orchestrate.ts +169 -114
- package/src/plugins/defaults/memory-v3-shadow/page-content.ts +47 -16
- package/src/plugins/defaults/memory-v3-shadow/pool-select.ts +144 -66
- package/src/plugins/defaults/memory-v3-shadow/prune.test.ts +758 -0
- package/src/plugins/defaults/memory-v3-shadow/prune.ts +471 -0
- package/src/plugins/defaults/memory-v3-shadow/render-injection.ts +68 -16
- package/src/plugins/defaults/memory-v3-shadow/selection-log-store.ts +19 -12
- package/src/plugins/defaults/memory-v3-shadow/shadow-plugin.ts +96 -43
- package/src/plugins/defaults/memory-v3-shadow/types.ts +34 -17
- package/src/plugins/defaults/title-generate/hooks/stop.ts +9 -11
- package/src/plugins/pipeline.ts +8 -5
- package/src/plugins/types.ts +5 -51
- package/src/providers/cache-control.ts +26 -0
- package/src/providers/inference/__tests__/base-url-route-validation.test.ts +1 -2
- package/src/providers/model-catalog.ts +13 -1
- package/src/providers/openai/__tests__/tool-choice-mapping.test.ts +147 -0
- package/src/providers/openai/chat-completions-provider.ts +46 -0
- package/src/providers/openai/responses-provider.ts +45 -0
- package/src/providers/registry.ts +0 -8
- package/src/runtime/__tests__/agent-wake.test.ts +0 -1
- package/src/runtime/agent-wake.ts +13 -0
- package/src/runtime/routes/__tests__/acp-routes.test.ts +151 -0
- package/src/runtime/routes/__tests__/consolidation-routes.test.ts +12 -50
- package/src/runtime/routes/__tests__/conversation-surface-routes.test.ts +322 -0
- package/src/runtime/routes/__tests__/inference-provider-connection-routes.test.ts +0 -62
- package/src/runtime/routes/__tests__/plugins-routes.test.ts +31 -11
- package/src/runtime/routes/acp-routes.test.ts +106 -0
- package/src/runtime/routes/acp-routes.ts +248 -2
- package/src/runtime/routes/browser-tabs-routes.ts +1 -1
- package/src/runtime/routes/channel-verification-routes.ts +14 -5
- package/src/runtime/routes/chatgpt-subscription-auth-routes.ts +0 -14
- package/src/runtime/routes/consolidation-routes.ts +6 -82
- package/src/runtime/routes/conversation-list-routes.ts +6 -0
- package/src/runtime/routes/conversation-management-routes.ts +71 -0
- package/src/runtime/routes/identity-routes.ts +8 -0
- package/src/runtime/routes/inbound-stages/acl-enforcement.ts +33 -34
- package/src/runtime/routes/inference-provider-connection-routes.ts +0 -45
- package/src/runtime/routes/plugins-routes.ts +34 -45
- package/src/runtime/routes/schedule-routes.ts +43 -5
- package/src/runtime/routes/settings-routes.ts +140 -15
- package/src/runtime/routes/skills-routes.ts +18 -6
- package/src/runtime/services/__tests__/conversation-serializer.test.ts +140 -0
- package/src/runtime/services/conversation-serializer.ts +38 -1
- package/src/runtime/verification-outbound-actions.ts +147 -2
- package/src/runtime/verification-templates.ts +29 -3
- package/src/schedule/schedule-store.ts +19 -0
- package/src/skills/catalog-install.ts +77 -13
- package/src/tasks/task-scheduler.ts +1 -0
- package/src/telemetry/types.ts +66 -1
- package/src/telemetry/usage-telemetry-reporter.test.ts +542 -13
- package/src/telemetry/usage-telemetry-reporter.ts +213 -20
- package/src/tools/browser/__tests__/browser-execution-acquire.test.ts +49 -2
- package/src/tools/browser/__tests__/browser-status.test.ts +29 -5
- package/src/tools/browser/browser-execution.ts +27 -9
- package/src/tools/browser/cdp-client/__tests__/factory.test.ts +380 -4
- package/src/tools/browser/cdp-client/__tests__/host-bridge-cdp-client.test.ts +107 -0
- package/src/tools/browser/cdp-client/__tests__/types.test.ts +6 -1
- package/src/tools/browser/cdp-client/factory.ts +217 -17
- package/src/tools/browser/cdp-client/host-bridge-cdp-client.ts +67 -0
- package/src/tools/browser/cdp-client/types.ts +22 -2
- package/src/tools/credential-execution/make-authenticated-request.ts +2 -1
- package/src/tools/credential-execution/manage-secure-command-tool.ts +169 -164
- package/src/tools/credential-execution/run-authenticated-command.ts +2 -1
- package/src/tools/executor.ts +39 -7
- package/src/tools/registry.ts +387 -5
- package/src/tools/schedule/create.ts +16 -0
- package/src/tools/schedule/list.ts +12 -4
- package/src/tools/schedule/update.ts +12 -0
- package/src/tools/skills/load.ts +11 -6
- package/src/tools/terminal/safe-env.ts +2 -0
- package/src/tools/types.ts +65 -9
- package/src/tools/workspace-tools/loader.ts +673 -0
- package/src/usage/attribution.ts +28 -0
- package/src/util/device-id.ts +17 -3
- package/src/util/platform.ts +16 -0
- package/tsconfig.plugin-api.json +13 -0
- package/src/__tests__/plugin-external-api.test.ts +0 -68
- package/src/__tests__/plugin-skill-contribution.test.ts +0 -355
- package/src/daemon/message-types/browser.ts +0 -10
- package/src/notifications/__tests__/emit-signal-home-feed.test.ts +0 -187
- package/src/plugins/defaults/memory-v3-shadow/__tests__/fixtures/eval-turns.json +0 -36
- package/src/plugins/defaults/memory-v3-shadow/__tests__/fixtures/live-turns.json +0 -37
- package/src/plugins/defaults/memory-v3-shadow/__tests__/working-set-eviction.test.ts +0 -106
- package/src/plugins/defaults/memory-v3-shadow/__tests__/working-set-skeleton.test.ts +0 -44
- package/src/plugins/defaults/memory-v3-shadow/working-set.ts +0 -91
- package/src/plugins/external-api.ts +0 -114
- package/src/plugins/plugin-skill-contributions.ts +0 -292
|
@@ -6,7 +6,10 @@ import type {
|
|
|
6
6
|
CheckpointInfo,
|
|
7
7
|
} from "../agent/loop.js";
|
|
8
8
|
import { AgentLoop } from "../agent/loop.js";
|
|
9
|
+
import type { StopContext } from "../plugin-api/types.js";
|
|
10
|
+
import { REFUSAL_FALLBACK_TEXT } from "../plugins/defaults/empty-response/hooks/post-model-call.js";
|
|
9
11
|
import { resetPluginRegistryAndRegisterDefaults } from "../plugins/defaults/index.js";
|
|
12
|
+
import { registerPlugin } from "../plugins/registry.js";
|
|
10
13
|
import type {
|
|
11
14
|
ContentBlock,
|
|
12
15
|
Message,
|
|
@@ -14,6 +17,7 @@ import type {
|
|
|
14
17
|
ProviderResponse,
|
|
15
18
|
ToolDefinition,
|
|
16
19
|
} from "../providers/types.js";
|
|
20
|
+
import { ContextOverflowError } from "../providers/types.js";
|
|
17
21
|
import {
|
|
18
22
|
createMockProvider,
|
|
19
23
|
textResponse,
|
|
@@ -64,6 +68,7 @@ describe("AgentLoop", () => {
|
|
|
64
68
|
|
|
65
69
|
const events: AgentEvent[] = [];
|
|
66
70
|
const { history } = await loop.run({
|
|
71
|
+
requestId: "test-request",
|
|
67
72
|
messages: [userMessage],
|
|
68
73
|
onEvent: collectEvents(events),
|
|
69
74
|
trust: { sourceChannel: "vellum", trustClass: "unknown" },
|
|
@@ -102,6 +107,7 @@ describe("AgentLoop", () => {
|
|
|
102
107
|
});
|
|
103
108
|
const events: AgentEvent[] = [];
|
|
104
109
|
const { history } = await loop.run({
|
|
110
|
+
requestId: "test-request",
|
|
105
111
|
messages: [userMessage],
|
|
106
112
|
onEvent: collectEvents(events),
|
|
107
113
|
trust: { sourceChannel: "vellum", trustClass: "unknown" },
|
|
@@ -239,6 +245,7 @@ describe("AgentLoop", () => {
|
|
|
239
245
|
toolExecutor: toolExecutor,
|
|
240
246
|
});
|
|
241
247
|
const { history } = await loop.run({
|
|
248
|
+
requestId: "test-request",
|
|
242
249
|
messages: [userMessage],
|
|
243
250
|
onEvent: () => {},
|
|
244
251
|
trust: { sourceChannel: "vellum", trustClass: "unknown" },
|
|
@@ -268,6 +275,7 @@ describe("AgentLoop", () => {
|
|
|
268
275
|
tools: dummyTools,
|
|
269
276
|
});
|
|
270
277
|
const { history } = await loop.run({
|
|
278
|
+
requestId: "test-request",
|
|
271
279
|
messages: [userMessage],
|
|
272
280
|
onEvent: () => {},
|
|
273
281
|
trust: { sourceChannel: "vellum", trustClass: "unknown" },
|
|
@@ -295,6 +303,7 @@ describe("AgentLoop", () => {
|
|
|
295
303
|
});
|
|
296
304
|
const events: AgentEvent[] = [];
|
|
297
305
|
const { history } = await loop.run({
|
|
306
|
+
requestId: "test-request",
|
|
298
307
|
messages: [userMessage],
|
|
299
308
|
onEvent: collectEvents(events),
|
|
300
309
|
trust: { sourceChannel: "vellum", trustClass: "unknown" },
|
|
@@ -311,6 +320,333 @@ describe("AgentLoop", () => {
|
|
|
311
320
|
).toBe("API rate limit exceeded");
|
|
312
321
|
});
|
|
313
322
|
|
|
323
|
+
// 5b. Reactive context-overflow recovery
|
|
324
|
+
//
|
|
325
|
+
// When recovery is enabled the loop calibrates the estimator from the
|
|
326
|
+
// rejection, loops back, and the budget gate forwards the overflow signal
|
|
327
|
+
// into the compaction plugin's reduction ladder. That manager-owned path —
|
|
328
|
+
// single-rung recovery, multi-rung escalation, and the terminal
|
|
329
|
+
// `context_too_large` / `budget_yield_unrecovered` exit on exhaustion — is
|
|
330
|
+
// exercised end-to-end against a manager harness in
|
|
331
|
+
// `conversation-agent-loop-overflow.test.ts`. Loop-level coverage against a
|
|
332
|
+
// manager-store stub lands with the suite migration.
|
|
333
|
+
test.todo(
|
|
334
|
+
"drives the reduction ladder on a context-overflow rejection",
|
|
335
|
+
() => {},
|
|
336
|
+
);
|
|
337
|
+
|
|
338
|
+
test("surfaces a context-overflow error when recovery is unavailable", async () => {
|
|
339
|
+
/**
|
|
340
|
+
* Overflow recovery is gated on the budget gate being active. When it is
|
|
341
|
+
* disabled (e.g. agent wakes pass `overflowRecovery.enabled = false`, or no
|
|
342
|
+
* context window resolves) there is no ladder to drive, so a context
|
|
343
|
+
* overflow surfaces as an error on the first call rather than looping.
|
|
344
|
+
*/
|
|
345
|
+
|
|
346
|
+
// GIVEN a provider that rejects every call as too large and a loop with no
|
|
347
|
+
// overflow-recovery budget gate
|
|
348
|
+
const { provider, calls } = createMockProvider([
|
|
349
|
+
new ContextOverflowError("prompt is too long", "mock", {
|
|
350
|
+
actualTokens: 250_000,
|
|
351
|
+
}),
|
|
352
|
+
]);
|
|
353
|
+
const loop = new AgentLoop({
|
|
354
|
+
provider,
|
|
355
|
+
systemPrompt: "system",
|
|
356
|
+
conversationId: "test-conversation",
|
|
357
|
+
});
|
|
358
|
+
const events: AgentEvent[] = [];
|
|
359
|
+
|
|
360
|
+
// WHEN the loop runs
|
|
361
|
+
await loop.run({
|
|
362
|
+
requestId: "test-request",
|
|
363
|
+
messages: [userMessage],
|
|
364
|
+
onEvent: collectEvents(events),
|
|
365
|
+
trust: { sourceChannel: "vellum", trustClass: "unknown" },
|
|
366
|
+
});
|
|
367
|
+
|
|
368
|
+
// THEN it surfaced the error on the first call without retrying
|
|
369
|
+
expect(calls).toHaveLength(1);
|
|
370
|
+
expect(events.filter((e) => e.type === "error")).toHaveLength(1);
|
|
371
|
+
});
|
|
372
|
+
|
|
373
|
+
// 5c. Reactive ordering-error recovery
|
|
374
|
+
//
|
|
375
|
+
// When the provider rejects a call because the history violates
|
|
376
|
+
// tool-use/tool-result pairing or role-alternation rules, the loop runs
|
|
377
|
+
// `deepRepairHistory` directly and re-issues the call. It is bounded to a
|
|
378
|
+
// single repair pass per provider call, so a second consecutive ordering
|
|
379
|
+
// rejection falls through to the error path instead of looping forever.
|
|
380
|
+
test("repairs the history and re-issues on a provider ordering rejection", async () => {
|
|
381
|
+
// GIVEN a provider that rejects the first call with an ordering error and
|
|
382
|
+
// succeeds on the retry
|
|
383
|
+
const { provider, calls } = createMockProvider([
|
|
384
|
+
new Error(
|
|
385
|
+
"tool_result blocks that are not immediately after a tool_use block",
|
|
386
|
+
),
|
|
387
|
+
textResponse("recovered"),
|
|
388
|
+
]);
|
|
389
|
+
const loop = new AgentLoop({
|
|
390
|
+
provider,
|
|
391
|
+
systemPrompt: "system",
|
|
392
|
+
conversationId: "test-conversation",
|
|
393
|
+
});
|
|
394
|
+
const events: AgentEvent[] = [];
|
|
395
|
+
|
|
396
|
+
// WHEN the loop runs
|
|
397
|
+
const { history } = await loop.run({
|
|
398
|
+
messages: [userMessage],
|
|
399
|
+
onEvent: collectEvents(events),
|
|
400
|
+
trust: { sourceChannel: "vellum", trustClass: "unknown" },
|
|
401
|
+
requestId: "test-request",
|
|
402
|
+
});
|
|
403
|
+
|
|
404
|
+
// THEN it re-issued the call after repairing, surfaced no error, and
|
|
405
|
+
// appended the recovered assistant response
|
|
406
|
+
expect(calls).toHaveLength(2);
|
|
407
|
+
expect(events.filter((e) => e.type === "error")).toHaveLength(0);
|
|
408
|
+
expect(history[history.length - 1]).toEqual({
|
|
409
|
+
role: "assistant",
|
|
410
|
+
content: [{ type: "text", text: "recovered" }],
|
|
411
|
+
});
|
|
412
|
+
});
|
|
413
|
+
|
|
414
|
+
test("surfaces an ordering error after a single repair fails to recover", async () => {
|
|
415
|
+
// GIVEN a provider that rejects every call with an ordering error
|
|
416
|
+
const { provider, calls } = createMockProvider([
|
|
417
|
+
new Error("tool_use_id provided without a matching tool_result"),
|
|
418
|
+
]);
|
|
419
|
+
const loop = new AgentLoop({
|
|
420
|
+
provider,
|
|
421
|
+
systemPrompt: "system",
|
|
422
|
+
conversationId: "test-conversation",
|
|
423
|
+
});
|
|
424
|
+
const events: AgentEvent[] = [];
|
|
425
|
+
|
|
426
|
+
// WHEN the loop runs
|
|
427
|
+
await loop.run({
|
|
428
|
+
messages: [userMessage],
|
|
429
|
+
onEvent: collectEvents(events),
|
|
430
|
+
trust: { sourceChannel: "vellum", trustClass: "unknown" },
|
|
431
|
+
requestId: "test-request",
|
|
432
|
+
});
|
|
433
|
+
|
|
434
|
+
// THEN it repaired once, re-issued once, then surfaced the error rather
|
|
435
|
+
// than retrying again
|
|
436
|
+
expect(calls).toHaveLength(2);
|
|
437
|
+
expect(events.filter((e) => e.type === "error")).toHaveLength(1);
|
|
438
|
+
});
|
|
439
|
+
|
|
440
|
+
test("reports only the retry's output as new messages when deep-repair shrinks the base", async () => {
|
|
441
|
+
// The wrapper reconstructs the persisted transcript from the loop's
|
|
442
|
+
// `newMessages` tail, so after `deepRepairHistory` removes a base message
|
|
443
|
+
// the new-message boundary must track the re-normalized history — otherwise
|
|
444
|
+
// the recovered assistant response is sliced off and lost.
|
|
445
|
+
|
|
446
|
+
// GIVEN an input whose leading assistant message deep-repair strips, a
|
|
447
|
+
// provider that rejects the first call on ordering grounds, then succeeds
|
|
448
|
+
const leadingAssistantMessage: Message = {
|
|
449
|
+
role: "assistant",
|
|
450
|
+
content: [{ type: "text", text: "stale leading turn" }],
|
|
451
|
+
};
|
|
452
|
+
const { provider } = createMockProvider([
|
|
453
|
+
new Error(
|
|
454
|
+
"tool_result blocks that are not immediately after a tool_use block",
|
|
455
|
+
),
|
|
456
|
+
textResponse("recovered"),
|
|
457
|
+
]);
|
|
458
|
+
const loop = new AgentLoop({
|
|
459
|
+
provider,
|
|
460
|
+
systemPrompt: "system",
|
|
461
|
+
conversationId: "test-conversation",
|
|
462
|
+
});
|
|
463
|
+
|
|
464
|
+
// WHEN the loop runs
|
|
465
|
+
const { history, newMessages } = await loop.run({
|
|
466
|
+
messages: [leadingAssistantMessage, userMessage],
|
|
467
|
+
onEvent: () => {},
|
|
468
|
+
trust: { sourceChannel: "vellum", trustClass: "unknown" },
|
|
469
|
+
requestId: "test-request",
|
|
470
|
+
});
|
|
471
|
+
|
|
472
|
+
// THEN deep-repair dropped the leading assistant from the base, and
|
|
473
|
+
// `newMessages` is exactly the retry's appended response
|
|
474
|
+
expect(history).toEqual([
|
|
475
|
+
userMessage,
|
|
476
|
+
{ role: "assistant", content: [{ type: "text", text: "recovered" }] },
|
|
477
|
+
]);
|
|
478
|
+
expect(newMessages).toEqual([
|
|
479
|
+
{ role: "assistant", content: [{ type: "text", text: "recovered" }] },
|
|
480
|
+
]);
|
|
481
|
+
});
|
|
482
|
+
|
|
483
|
+
test("repairs a fresh ordering rejection on a later run after a prior run left the bound spent", async () => {
|
|
484
|
+
// The ordering-repair bound is owned by the history-repair plugin's
|
|
485
|
+
// per-conversation state. A run scopes that bound, so the loop clears it on
|
|
486
|
+
// entry — otherwise a direct `run` caller that bypasses the daemon
|
|
487
|
+
// orchestrator (e.g. an agent wake) inherits a spent bound from a prior run
|
|
488
|
+
// on the same conversation and surfaces a repairable rejection instead of
|
|
489
|
+
// repairing it.
|
|
490
|
+
|
|
491
|
+
// GIVEN a first run that rejects on ordering grounds, repairs once, then
|
|
492
|
+
// rejects again — ending with the bound spent and the error surfaced
|
|
493
|
+
const { provider: firstProvider, calls: firstCalls } = createMockProvider([
|
|
494
|
+
new Error("tool_use_id provided without a matching tool_result"),
|
|
495
|
+
new Error("tool_use_id provided without a matching tool_result"),
|
|
496
|
+
]);
|
|
497
|
+
const conversationId = "ordering-bound-across-runs";
|
|
498
|
+
const firstLoop = new AgentLoop({
|
|
499
|
+
provider: firstProvider,
|
|
500
|
+
systemPrompt: "system",
|
|
501
|
+
conversationId,
|
|
502
|
+
});
|
|
503
|
+
const firstEvents: AgentEvent[] = [];
|
|
504
|
+
await firstLoop.run({
|
|
505
|
+
messages: [userMessage],
|
|
506
|
+
onEvent: collectEvents(firstEvents),
|
|
507
|
+
trust: { sourceChannel: "vellum", trustClass: "unknown" },
|
|
508
|
+
requestId: "test-request",
|
|
509
|
+
});
|
|
510
|
+
expect(firstCalls).toHaveLength(2);
|
|
511
|
+
expect(firstEvents.filter((e) => e.type === "error")).toHaveLength(1);
|
|
512
|
+
|
|
513
|
+
// AND a second run on the same conversation whose first call rejects on
|
|
514
|
+
// ordering grounds, then succeeds on the retry
|
|
515
|
+
const { provider: secondProvider, calls: secondCalls } = createMockProvider(
|
|
516
|
+
[
|
|
517
|
+
new Error(
|
|
518
|
+
"tool_result blocks that are not immediately after a tool_use block",
|
|
519
|
+
),
|
|
520
|
+
textResponse("recovered"),
|
|
521
|
+
],
|
|
522
|
+
);
|
|
523
|
+
const secondLoop = new AgentLoop({
|
|
524
|
+
provider: secondProvider,
|
|
525
|
+
systemPrompt: "system",
|
|
526
|
+
conversationId,
|
|
527
|
+
});
|
|
528
|
+
const secondEvents: AgentEvent[] = [];
|
|
529
|
+
|
|
530
|
+
// WHEN the second run executes
|
|
531
|
+
const { history } = await secondLoop.run({
|
|
532
|
+
messages: [userMessage],
|
|
533
|
+
onEvent: collectEvents(secondEvents),
|
|
534
|
+
trust: { sourceChannel: "vellum", trustClass: "unknown" },
|
|
535
|
+
requestId: "test-request",
|
|
536
|
+
});
|
|
537
|
+
|
|
538
|
+
// THEN it repaired and re-issued (the bound did not leak across runs),
|
|
539
|
+
// surfaced no error, and appended the recovered response
|
|
540
|
+
expect(secondCalls).toHaveLength(2);
|
|
541
|
+
expect(secondEvents.filter((e) => e.type === "error")).toHaveLength(0);
|
|
542
|
+
expect(history[history.length - 1]).toEqual({
|
|
543
|
+
role: "assistant",
|
|
544
|
+
content: [{ type: "text", text: "recovered" }],
|
|
545
|
+
});
|
|
546
|
+
});
|
|
547
|
+
|
|
548
|
+
test("isolates a throwing stop hook on a successful stop and still emits the terminal exit", async () => {
|
|
549
|
+
// The `stop` chain is the loop's terminal teardown: a throwing teardown
|
|
550
|
+
// hook (e.g. a third-party plugin) is logged and contained, never escalated
|
|
551
|
+
// into a turn error. The turn's real outcome stands — a successful no-tool
|
|
552
|
+
// stop — and the terminal `agent_loop_exit` still fires so the run stays
|
|
553
|
+
// observable.
|
|
554
|
+
|
|
555
|
+
// GIVEN a registered stop hook that always throws, and a provider that
|
|
556
|
+
// returns a successful no-tool response (a successful stop)
|
|
557
|
+
registerPlugin({
|
|
558
|
+
manifest: { name: "throwing-stop", version: "0.0.1" },
|
|
559
|
+
hooks: {
|
|
560
|
+
stop: async () => {
|
|
561
|
+
throw new Error("stop hook boom");
|
|
562
|
+
},
|
|
563
|
+
},
|
|
564
|
+
});
|
|
565
|
+
const { provider, calls } = createMockProvider([textResponse("done")]);
|
|
566
|
+
const loop = new AgentLoop({
|
|
567
|
+
provider,
|
|
568
|
+
systemPrompt: "system",
|
|
569
|
+
conversationId: "test-conversation",
|
|
570
|
+
});
|
|
571
|
+
const events: AgentEvent[] = [];
|
|
572
|
+
|
|
573
|
+
// WHEN the loop runs
|
|
574
|
+
let rejected = false;
|
|
575
|
+
await loop
|
|
576
|
+
.run({
|
|
577
|
+
messages: [userMessage],
|
|
578
|
+
onEvent: collectEvents(events),
|
|
579
|
+
trust: { sourceChannel: "vellum", trustClass: "unknown" },
|
|
580
|
+
requestId: "test-request",
|
|
581
|
+
})
|
|
582
|
+
.catch(() => {
|
|
583
|
+
rejected = true;
|
|
584
|
+
});
|
|
585
|
+
|
|
586
|
+
// THEN the loop did not reject, did not re-issue the call, surfaced no
|
|
587
|
+
// error event, and still emitted the terminal exit with the real reason
|
|
588
|
+
expect(rejected).toBe(false);
|
|
589
|
+
expect(calls).toHaveLength(1);
|
|
590
|
+
expect(events.filter((e) => e.type === "error")).toHaveLength(0);
|
|
591
|
+
const exitEvents = events.filter((e) => e.type === "agent_loop_exit");
|
|
592
|
+
expect(exitEvents).toHaveLength(1);
|
|
593
|
+
expect(exitEvents[0]).toMatchObject({ reason: "no_tool_calls" });
|
|
594
|
+
});
|
|
595
|
+
|
|
596
|
+
test("fires the stop teardown chain on a checkpoint handoff without emitting agent_loop_exit", async () => {
|
|
597
|
+
// A handoff pauses this run so the orchestrator can drain a queued message,
|
|
598
|
+
// then re-enters with a fresh run. It still ends *this* turn, so the
|
|
599
|
+
// terminal `stop` chain must fire — that is what runs per-turn teardown
|
|
600
|
+
// (e.g. clearing the recovery bounds the post-model-call hooks set) before
|
|
601
|
+
// the queued message is processed. But `agent_loop_exit` must NOT be
|
|
602
|
+
// emitted, because the handoff is a control transfer, not a terminal exit.
|
|
603
|
+
|
|
604
|
+
// GIVEN a stop hook recording every exit reason it observes, and a provider
|
|
605
|
+
// whose first reply requests a tool so the loop reaches a checkpoint
|
|
606
|
+
const stopReasons: string[] = [];
|
|
607
|
+
registerPlugin({
|
|
608
|
+
manifest: { name: "recording-stop", version: "0.0.1" },
|
|
609
|
+
hooks: {
|
|
610
|
+
stop: async (ctx: StopContext) => {
|
|
611
|
+
stopReasons.push(ctx.exitReason);
|
|
612
|
+
},
|
|
613
|
+
},
|
|
614
|
+
});
|
|
615
|
+
const { provider } = createMockProvider([
|
|
616
|
+
toolUseResponse("t1", "read_file", { path: "/a.txt" }),
|
|
617
|
+
textResponse("never reached"),
|
|
618
|
+
]);
|
|
619
|
+
const toolExecutor = async () => ({ content: "ok", isError: false });
|
|
620
|
+
const loop = new AgentLoop({
|
|
621
|
+
provider,
|
|
622
|
+
systemPrompt: "system",
|
|
623
|
+
conversationId: "test-conversation",
|
|
624
|
+
tools: dummyTools,
|
|
625
|
+
toolExecutor,
|
|
626
|
+
});
|
|
627
|
+
const events: AgentEvent[] = [];
|
|
628
|
+
|
|
629
|
+
// AND a checkpoint callback that hands off at the first opportunity
|
|
630
|
+
const onCheckpoint = (_info: CheckpointInfo): CheckpointDecision =>
|
|
631
|
+
"handoff";
|
|
632
|
+
|
|
633
|
+
// WHEN the loop runs and yields control at the checkpoint
|
|
634
|
+
const { exitReason } = await loop.run({
|
|
635
|
+
requestId: "test-request",
|
|
636
|
+
messages: [userMessage],
|
|
637
|
+
onEvent: collectEvents(events),
|
|
638
|
+
trust: { sourceChannel: "vellum", trustClass: "unknown" },
|
|
639
|
+
onCheckpoint,
|
|
640
|
+
});
|
|
641
|
+
|
|
642
|
+
// THEN the terminal stop chain fired exactly once with the handoff reason
|
|
643
|
+
// (teardown ran), the run reported the handoff, and no agent_loop_exit was
|
|
644
|
+
// emitted
|
|
645
|
+
expect(stopReasons).toEqual(["checkpoint_handoff"]);
|
|
646
|
+
expect(exitReason).toBe("handoff");
|
|
647
|
+
expect(events.filter((e) => e.type === "agent_loop_exit")).toHaveLength(0);
|
|
648
|
+
});
|
|
649
|
+
|
|
314
650
|
// 6. Abort signal — verify the loop respects AbortSignal
|
|
315
651
|
test("stops when abort signal is triggered before provider call", async () => {
|
|
316
652
|
const controller = new AbortController();
|
|
@@ -323,6 +659,7 @@ describe("AgentLoop", () => {
|
|
|
323
659
|
conversationId: "test-conversation",
|
|
324
660
|
});
|
|
325
661
|
const { history } = await loop.run({
|
|
662
|
+
requestId: "test-request",
|
|
326
663
|
messages: [userMessage],
|
|
327
664
|
onEvent: () => {},
|
|
328
665
|
trust: { sourceChannel: "vellum", trustClass: "unknown" },
|
|
@@ -360,6 +697,7 @@ describe("AgentLoop", () => {
|
|
|
360
697
|
toolExecutor: toolExecutor,
|
|
361
698
|
});
|
|
362
699
|
const { history } = await loop.run({
|
|
700
|
+
requestId: "test-request",
|
|
363
701
|
messages: [userMessage],
|
|
364
702
|
onEvent: () => {},
|
|
365
703
|
trust: { sourceChannel: "vellum", trustClass: "unknown" },
|
|
@@ -413,6 +751,7 @@ describe("AgentLoop", () => {
|
|
|
413
751
|
});
|
|
414
752
|
const start = Date.now();
|
|
415
753
|
const { history } = await loop.run({
|
|
754
|
+
requestId: "test-request",
|
|
416
755
|
messages: [userMessage],
|
|
417
756
|
onEvent: () => {},
|
|
418
757
|
trust: { sourceChannel: "vellum", trustClass: "unknown" },
|
|
@@ -462,6 +801,7 @@ describe("AgentLoop", () => {
|
|
|
462
801
|
|
|
463
802
|
const events: AgentEvent[] = [];
|
|
464
803
|
await loop.run({
|
|
804
|
+
requestId: "test-request",
|
|
465
805
|
messages: [userMessage],
|
|
466
806
|
onEvent: collectEvents(events),
|
|
467
807
|
trust: { sourceChannel: "vellum", trustClass: "unknown" },
|
|
@@ -484,6 +824,7 @@ describe("AgentLoop", () => {
|
|
|
484
824
|
|
|
485
825
|
const events: AgentEvent[] = [];
|
|
486
826
|
await loop.run({
|
|
827
|
+
requestId: "test-request",
|
|
487
828
|
messages: [userMessage],
|
|
488
829
|
onEvent: collectEvents(events),
|
|
489
830
|
trust: { sourceChannel: "vellum", trustClass: "unknown" },
|
|
@@ -510,6 +851,7 @@ describe("AgentLoop", () => {
|
|
|
510
851
|
|
|
511
852
|
const events: AgentEvent[] = [];
|
|
512
853
|
await loop.run({
|
|
854
|
+
requestId: "test-request",
|
|
513
855
|
messages: [userMessage],
|
|
514
856
|
onEvent: collectEvents(events),
|
|
515
857
|
trust: { sourceChannel: "vellum", trustClass: "unknown" },
|
|
@@ -540,6 +882,7 @@ describe("AgentLoop", () => {
|
|
|
540
882
|
|
|
541
883
|
const events: AgentEvent[] = [];
|
|
542
884
|
await loop.run({
|
|
885
|
+
requestId: "test-request",
|
|
543
886
|
messages: [userMessage],
|
|
544
887
|
onEvent: collectEvents(events),
|
|
545
888
|
trust: { sourceChannel: "vellum", trustClass: "unknown" },
|
|
@@ -590,6 +933,7 @@ describe("AgentLoop", () => {
|
|
|
590
933
|
});
|
|
591
934
|
|
|
592
935
|
await loop.run({
|
|
936
|
+
requestId: "test-request",
|
|
593
937
|
messages: [userMessage],
|
|
594
938
|
onEvent: () => {},
|
|
595
939
|
trust: { sourceChannel: "vellum", trustClass: "unknown" },
|
|
@@ -632,6 +976,7 @@ describe("AgentLoop", () => {
|
|
|
632
976
|
});
|
|
633
977
|
const events: AgentEvent[] = [];
|
|
634
978
|
await loop.run({
|
|
979
|
+
requestId: "test-request",
|
|
635
980
|
messages: [userMessage],
|
|
636
981
|
onEvent: collectEvents(events),
|
|
637
982
|
trust: { sourceChannel: "vellum", trustClass: "unknown" },
|
|
@@ -660,6 +1005,7 @@ describe("AgentLoop", () => {
|
|
|
660
1005
|
});
|
|
661
1006
|
|
|
662
1007
|
await loop.run({
|
|
1008
|
+
requestId: "test-request",
|
|
663
1009
|
messages: [userMessage],
|
|
664
1010
|
onEvent: () => {},
|
|
665
1011
|
trust: { sourceChannel: "vellum", trustClass: "unknown" },
|
|
@@ -679,6 +1025,7 @@ describe("AgentLoop", () => {
|
|
|
679
1025
|
});
|
|
680
1026
|
|
|
681
1027
|
await loop.run({
|
|
1028
|
+
requestId: "test-request",
|
|
682
1029
|
messages: [userMessage],
|
|
683
1030
|
onEvent: () => {},
|
|
684
1031
|
trust: { sourceChannel: "vellum", trustClass: "unknown" },
|
|
@@ -744,6 +1091,7 @@ describe("AgentLoop", () => {
|
|
|
744
1091
|
});
|
|
745
1092
|
const events: AgentEvent[] = [];
|
|
746
1093
|
const { history } = await loop.run({
|
|
1094
|
+
requestId: "test-request",
|
|
747
1095
|
messages: [userMessage],
|
|
748
1096
|
onEvent: collectEvents(events),
|
|
749
1097
|
trust: { sourceChannel: "vellum", trustClass: "unknown" },
|
|
@@ -845,6 +1193,7 @@ describe("AgentLoop", () => {
|
|
|
845
1193
|
});
|
|
846
1194
|
const events: AgentEvent[] = [];
|
|
847
1195
|
const { history } = await loop.run({
|
|
1196
|
+
requestId: "test-request",
|
|
848
1197
|
messages: [userMessage],
|
|
849
1198
|
onEvent: collectEvents(events),
|
|
850
1199
|
trust: { sourceChannel: "vellum", trustClass: "unknown" },
|
|
@@ -925,6 +1274,7 @@ describe("AgentLoop", () => {
|
|
|
925
1274
|
});
|
|
926
1275
|
const events: AgentEvent[] = [];
|
|
927
1276
|
await loop.run({
|
|
1277
|
+
requestId: "test-request",
|
|
928
1278
|
messages: [userMessage],
|
|
929
1279
|
onEvent: collectEvents(events),
|
|
930
1280
|
trust: { sourceChannel: "vellum", trustClass: "unknown" },
|
|
@@ -970,6 +1320,7 @@ describe("AgentLoop", () => {
|
|
|
970
1320
|
};
|
|
971
1321
|
|
|
972
1322
|
await loop.run({
|
|
1323
|
+
requestId: "test-request",
|
|
973
1324
|
messages: [userMessage],
|
|
974
1325
|
onEvent: () => {},
|
|
975
1326
|
trust: { sourceChannel: "vellum", trustClass: "unknown" },
|
|
@@ -1006,6 +1357,7 @@ describe("AgentLoop", () => {
|
|
|
1006
1357
|
const onCheckpoint = (): CheckpointDecision => "continue";
|
|
1007
1358
|
|
|
1008
1359
|
const { history } = await loop.run({
|
|
1360
|
+
requestId: "test-request",
|
|
1009
1361
|
messages: [userMessage],
|
|
1010
1362
|
onEvent: () => {},
|
|
1011
1363
|
trust: { sourceChannel: "vellum", trustClass: "unknown" },
|
|
@@ -1036,9 +1388,10 @@ describe("AgentLoop", () => {
|
|
|
1036
1388
|
toolExecutor: toolExecutor,
|
|
1037
1389
|
});
|
|
1038
1390
|
|
|
1039
|
-
const onCheckpoint = (): CheckpointDecision => "
|
|
1391
|
+
const onCheckpoint = (): CheckpointDecision => "handoff";
|
|
1040
1392
|
|
|
1041
1393
|
const { history } = await loop.run({
|
|
1394
|
+
requestId: "test-request",
|
|
1042
1395
|
messages: [userMessage],
|
|
1043
1396
|
onEvent: () => {},
|
|
1044
1397
|
trust: { sourceChannel: "vellum", trustClass: "unknown" },
|
|
@@ -1070,6 +1423,7 @@ describe("AgentLoop", () => {
|
|
|
1070
1423
|
});
|
|
1071
1424
|
|
|
1072
1425
|
const { history } = await loop.run({
|
|
1426
|
+
requestId: "test-request",
|
|
1073
1427
|
messages: [userMessage],
|
|
1074
1428
|
onEvent: () => {},
|
|
1075
1429
|
trust: { sourceChannel: "vellum", trustClass: "unknown" },
|
|
@@ -1106,6 +1460,7 @@ describe("AgentLoop", () => {
|
|
|
1106
1460
|
};
|
|
1107
1461
|
|
|
1108
1462
|
await loop.run({
|
|
1463
|
+
requestId: "test-request",
|
|
1109
1464
|
messages: [userMessage],
|
|
1110
1465
|
onEvent: () => {},
|
|
1111
1466
|
trust: { sourceChannel: "vellum", trustClass: "unknown" },
|
|
@@ -1137,6 +1492,7 @@ describe("AgentLoop", () => {
|
|
|
1137
1492
|
};
|
|
1138
1493
|
|
|
1139
1494
|
const { history } = await loop.run({
|
|
1495
|
+
requestId: "test-request",
|
|
1140
1496
|
messages: [userMessage],
|
|
1141
1497
|
onEvent: () => {},
|
|
1142
1498
|
trust: { sourceChannel: "vellum", trustClass: "unknown" },
|
|
@@ -1199,6 +1555,7 @@ describe("AgentLoop", () => {
|
|
|
1199
1555
|
};
|
|
1200
1556
|
|
|
1201
1557
|
await loop.run({
|
|
1558
|
+
requestId: "test-request",
|
|
1202
1559
|
messages: [userMessage],
|
|
1203
1560
|
onEvent: () => {},
|
|
1204
1561
|
trust: { sourceChannel: "vellum", trustClass: "unknown" },
|
|
@@ -1235,11 +1592,12 @@ describe("AgentLoop", () => {
|
|
|
1235
1592
|
const onCheckpoint = (checkpoint: CheckpointInfo): CheckpointDecision => {
|
|
1236
1593
|
checkpoints.push(checkpoint);
|
|
1237
1594
|
// Yield on turn 3 (0-indexed)
|
|
1238
|
-
return checkpoint.turnIndex === 3 ? "
|
|
1595
|
+
return checkpoint.turnIndex === 3 ? "handoff" : "continue";
|
|
1239
1596
|
};
|
|
1240
1597
|
|
|
1241
1598
|
const events: AgentEvent[] = [];
|
|
1242
1599
|
const { history } = await loop.run({
|
|
1600
|
+
requestId: "test-request",
|
|
1243
1601
|
messages: [userMessage],
|
|
1244
1602
|
onEvent: collectEvents(events),
|
|
1245
1603
|
trust: { sourceChannel: "vellum", trustClass: "unknown" },
|
|
@@ -1308,10 +1666,11 @@ describe("AgentLoop", () => {
|
|
|
1308
1666
|
|
|
1309
1667
|
const onCheckpoint = (checkpoint: CheckpointInfo): CheckpointDecision => {
|
|
1310
1668
|
// Yield on the second turn (turnIndex 1)
|
|
1311
|
-
return checkpoint.turnIndex === 1 ? "
|
|
1669
|
+
return checkpoint.turnIndex === 1 ? "handoff" : "continue";
|
|
1312
1670
|
};
|
|
1313
1671
|
|
|
1314
1672
|
const { history } = await loop.run({
|
|
1673
|
+
requestId: "test-request",
|
|
1315
1674
|
messages: [userMessage],
|
|
1316
1675
|
onEvent: () => {},
|
|
1317
1676
|
trust: { sourceChannel: "vellum", trustClass: "unknown" },
|
|
@@ -1339,6 +1698,7 @@ describe("AgentLoop", () => {
|
|
|
1339
1698
|
});
|
|
1340
1699
|
|
|
1341
1700
|
await loop.run({
|
|
1701
|
+
requestId: "test-request",
|
|
1342
1702
|
messages: [userMessage],
|
|
1343
1703
|
onEvent: () => {},
|
|
1344
1704
|
trust: { sourceChannel: "vellum", trustClass: "unknown" },
|
|
@@ -1381,6 +1741,7 @@ describe("AgentLoop", () => {
|
|
|
1381
1741
|
resolveTools: resolveTools,
|
|
1382
1742
|
});
|
|
1383
1743
|
await loop.run({
|
|
1744
|
+
requestId: "test-request",
|
|
1384
1745
|
messages: [userMessage],
|
|
1385
1746
|
onEvent: () => {},
|
|
1386
1747
|
trust: { sourceChannel: "vellum", trustClass: "unknown" },
|
|
@@ -1420,6 +1781,7 @@ describe("AgentLoop", () => {
|
|
|
1420
1781
|
resolveTools: resolveTools,
|
|
1421
1782
|
});
|
|
1422
1783
|
await loop.run({
|
|
1784
|
+
requestId: "test-request",
|
|
1423
1785
|
messages: [userMessage],
|
|
1424
1786
|
onEvent: () => {},
|
|
1425
1787
|
trust: { sourceChannel: "vellum", trustClass: "unknown" },
|
|
@@ -1484,6 +1846,7 @@ describe("AgentLoop", () => {
|
|
|
1484
1846
|
resolveTools: resolveTools,
|
|
1485
1847
|
});
|
|
1486
1848
|
await loop.run({
|
|
1849
|
+
requestId: "test-request",
|
|
1487
1850
|
messages: [userMessage],
|
|
1488
1851
|
onEvent: () => {},
|
|
1489
1852
|
trust: { sourceChannel: "vellum", trustClass: "unknown" },
|
|
@@ -1514,6 +1877,7 @@ describe("AgentLoop", () => {
|
|
|
1514
1877
|
resolveTools: resolveTools,
|
|
1515
1878
|
});
|
|
1516
1879
|
await loop.run({
|
|
1880
|
+
requestId: "test-request",
|
|
1517
1881
|
messages: [userMessage],
|
|
1518
1882
|
onEvent: () => {},
|
|
1519
1883
|
trust: { sourceChannel: "vellum", trustClass: "unknown" },
|
|
@@ -1551,6 +1915,7 @@ describe("AgentLoop", () => {
|
|
|
1551
1915
|
});
|
|
1552
1916
|
const events: AgentEvent[] = [];
|
|
1553
1917
|
const { history } = await loop.run({
|
|
1918
|
+
requestId: "test-request",
|
|
1554
1919
|
messages: [userMessage],
|
|
1555
1920
|
onEvent: collectEvents(events),
|
|
1556
1921
|
trust: { sourceChannel: "vellum", trustClass: "unknown" },
|
|
@@ -1607,6 +1972,7 @@ describe("AgentLoop", () => {
|
|
|
1607
1972
|
});
|
|
1608
1973
|
const events: AgentEvent[] = [];
|
|
1609
1974
|
const { history } = await loop.run({
|
|
1975
|
+
requestId: "test-request",
|
|
1610
1976
|
messages: [userMessage],
|
|
1611
1977
|
onEvent: collectEvents(events),
|
|
1612
1978
|
trust: { sourceChannel: "vellum", trustClass: "unknown" },
|
|
@@ -1671,6 +2037,7 @@ describe("AgentLoop", () => {
|
|
|
1671
2037
|
});
|
|
1672
2038
|
const events: AgentEvent[] = [];
|
|
1673
2039
|
const { history } = await loop.run({
|
|
2040
|
+
requestId: "test-request",
|
|
1674
2041
|
messages: [userMessage],
|
|
1675
2042
|
onEvent: collectEvents(events),
|
|
1676
2043
|
trust: { sourceChannel: "vellum", trustClass: "unknown" },
|
|
@@ -1744,6 +2111,7 @@ describe("AgentLoop", () => {
|
|
|
1744
2111
|
});
|
|
1745
2112
|
const events: AgentEvent[] = [];
|
|
1746
2113
|
await loop.run({
|
|
2114
|
+
requestId: "test-request",
|
|
1747
2115
|
messages: [userMessage],
|
|
1748
2116
|
onEvent: collectEvents(events),
|
|
1749
2117
|
trust: { sourceChannel: "vellum", trustClass: "unknown" },
|
|
@@ -1783,6 +2151,7 @@ describe("AgentLoop", () => {
|
|
|
1783
2151
|
});
|
|
1784
2152
|
const events: AgentEvent[] = [];
|
|
1785
2153
|
const { history } = await loop.run({
|
|
2154
|
+
requestId: "test-request",
|
|
1786
2155
|
messages: [userMessage],
|
|
1787
2156
|
onEvent: collectEvents(events),
|
|
1788
2157
|
trust: { sourceChannel: "vellum", trustClass: "unknown" },
|
|
@@ -1833,6 +2202,7 @@ describe("AgentLoop", () => {
|
|
|
1833
2202
|
toolExecutor: toolExecutor,
|
|
1834
2203
|
});
|
|
1835
2204
|
await loop.run({
|
|
2205
|
+
requestId: "test-request",
|
|
1836
2206
|
messages: [userMessage],
|
|
1837
2207
|
onEvent: () => {},
|
|
1838
2208
|
trust: { sourceChannel: "vellum", trustClass: "unknown" },
|
|
@@ -1898,6 +2268,7 @@ describe("AgentLoop", () => {
|
|
|
1898
2268
|
toolExecutor: toolExecutor,
|
|
1899
2269
|
});
|
|
1900
2270
|
await loop.run({
|
|
2271
|
+
requestId: "test-request",
|
|
1901
2272
|
messages: [userMessage],
|
|
1902
2273
|
onEvent: () => {},
|
|
1903
2274
|
trust: { sourceChannel: "vellum", trustClass: "unknown" },
|
|
@@ -1952,6 +2323,7 @@ describe("AgentLoop", () => {
|
|
|
1952
2323
|
});
|
|
1953
2324
|
const events: AgentEvent[] = [];
|
|
1954
2325
|
const { history } = await loop.run({
|
|
2326
|
+
requestId: "test-request",
|
|
1955
2327
|
messages: [userMessage],
|
|
1956
2328
|
onEvent: collectEvents(events),
|
|
1957
2329
|
trust: { sourceChannel: "vellum", trustClass: "unknown" },
|
|
@@ -2034,6 +2406,7 @@ describe("AgentLoop", () => {
|
|
|
2034
2406
|
});
|
|
2035
2407
|
const events: AgentEvent[] = [];
|
|
2036
2408
|
const { history } = await loop.run({
|
|
2409
|
+
requestId: "test-request",
|
|
2037
2410
|
messages: [userMessage],
|
|
2038
2411
|
onEvent: collectEvents(events),
|
|
2039
2412
|
trust: { sourceChannel: "vellum", trustClass: "unknown" },
|
|
@@ -2100,6 +2473,7 @@ describe("AgentLoop", () => {
|
|
|
2100
2473
|
});
|
|
2101
2474
|
const events: AgentEvent[] = [];
|
|
2102
2475
|
const { history } = await loop.run({
|
|
2476
|
+
requestId: "test-request",
|
|
2103
2477
|
messages: [userMessage],
|
|
2104
2478
|
onEvent: collectEvents(events),
|
|
2105
2479
|
trust: { sourceChannel: "vellum", trustClass: "unknown" },
|
|
@@ -2114,12 +2488,23 @@ describe("AgentLoop", () => {
|
|
|
2114
2488
|
);
|
|
2115
2489
|
expect(messageCompletes).toHaveLength(2);
|
|
2116
2490
|
|
|
2117
|
-
//
|
|
2491
|
+
// An organic empty `end_turn` (not a refusal) is left as-is — the fallback
|
|
2492
|
+
// is refusal-specific, so the exhausted empty turn stays empty.
|
|
2118
2493
|
const lastAssistant = [...history]
|
|
2119
2494
|
.reverse()
|
|
2120
2495
|
.find((m) => m.role === "assistant");
|
|
2121
2496
|
expect(lastAssistant).toBeDefined();
|
|
2122
2497
|
expect(lastAssistant!.content).toEqual([]);
|
|
2498
|
+
|
|
2499
|
+
// AND no fallback text is streamed for a non-refusal empty turn.
|
|
2500
|
+
const textDeltas = events.filter((e) => e.type === "text_delta");
|
|
2501
|
+
expect(
|
|
2502
|
+
textDeltas.some(
|
|
2503
|
+
(e) =>
|
|
2504
|
+
(e as { type: "text_delta"; text: string }).text ===
|
|
2505
|
+
REFUSAL_FALLBACK_TEXT,
|
|
2506
|
+
),
|
|
2507
|
+
).toBe(false);
|
|
2123
2508
|
});
|
|
2124
2509
|
|
|
2125
2510
|
test("does not retry empty response on first turn (no prior tool use)", async () => {
|
|
@@ -2139,6 +2524,7 @@ describe("AgentLoop", () => {
|
|
|
2139
2524
|
});
|
|
2140
2525
|
const events: AgentEvent[] = [];
|
|
2141
2526
|
const { history } = await loop.run({
|
|
2527
|
+
requestId: "test-request",
|
|
2142
2528
|
messages: [userMessage],
|
|
2143
2529
|
onEvent: collectEvents(events),
|
|
2144
2530
|
trust: { sourceChannel: "vellum", trustClass: "unknown" },
|
|
@@ -2146,7 +2532,192 @@ describe("AgentLoop", () => {
|
|
|
2146
2532
|
|
|
2147
2533
|
// Should NOT retry — this is the first turn with no tool use history
|
|
2148
2534
|
expect(calls).toHaveLength(1);
|
|
2149
|
-
|
|
2535
|
+
// user + assistant; an organic empty `end_turn` is left as-is (the fallback
|
|
2536
|
+
// is refusal-specific), so the empty turn stays empty.
|
|
2537
|
+
expect(history).toHaveLength(2);
|
|
2538
|
+
expect(history[1].content).toEqual([]);
|
|
2539
|
+
});
|
|
2540
|
+
|
|
2541
|
+
// Refusal stop boundary — the provider zeroed the response with
|
|
2542
|
+
// `stopReason: "refusal"`. The default `stop` hook rewrites the empty turn
|
|
2543
|
+
// into a user-facing fallback (no retry — a safety-classifier refusal
|
|
2544
|
+
// re-fires on a re-query), and the loop persists + streams it rather than the
|
|
2545
|
+
// empty assistant bubble the user would otherwise see.
|
|
2546
|
+
test("rewrites a refusal into a user-facing fallback without retrying", async () => {
|
|
2547
|
+
// GIVEN a provider that refuses (empty content, stopReason "refusal").
|
|
2548
|
+
const refusalResponse: ProviderResponse = {
|
|
2549
|
+
content: [],
|
|
2550
|
+
model: "mock-model",
|
|
2551
|
+
usage: { inputTokens: 10, outputTokens: 0 },
|
|
2552
|
+
stopReason: "refusal",
|
|
2553
|
+
};
|
|
2554
|
+
const { provider, calls } = createMockProvider([
|
|
2555
|
+
refusalResponse,
|
|
2556
|
+
refusalResponse,
|
|
2557
|
+
]);
|
|
2558
|
+
const loop = new AgentLoop({
|
|
2559
|
+
provider: provider,
|
|
2560
|
+
systemPrompt: "system",
|
|
2561
|
+
conversationId: "test-conversation",
|
|
2562
|
+
});
|
|
2563
|
+
|
|
2564
|
+
// WHEN the loop runs.
|
|
2565
|
+
const events: AgentEvent[] = [];
|
|
2566
|
+
const { history } = await loop.run({
|
|
2567
|
+
requestId: "test-request",
|
|
2568
|
+
messages: [userMessage],
|
|
2569
|
+
onEvent: collectEvents(events),
|
|
2570
|
+
trust: { sourceChannel: "vellum", trustClass: "unknown" },
|
|
2571
|
+
});
|
|
2572
|
+
|
|
2573
|
+
// THEN the loop calls the provider once: the refusal is rewritten in place
|
|
2574
|
+
// with no nudge-driven retry.
|
|
2575
|
+
expect(calls).toHaveLength(1);
|
|
2576
|
+
|
|
2577
|
+
// AND the user-visible assistant turn is the fallback, not an empty bubble.
|
|
2578
|
+
const lastAssistant = [...history]
|
|
2579
|
+
.reverse()
|
|
2580
|
+
.find((m) => m.role === "assistant");
|
|
2581
|
+
expect(lastAssistant!.content).toEqual([
|
|
2582
|
+
{ type: "text", text: REFUSAL_FALLBACK_TEXT },
|
|
2583
|
+
]);
|
|
2584
|
+
|
|
2585
|
+
// AND the fallback is streamed so the client renders it.
|
|
2586
|
+
const textDeltas = events.filter((e) => e.type === "text_delta");
|
|
2587
|
+
expect(
|
|
2588
|
+
textDeltas.some(
|
|
2589
|
+
(e) =>
|
|
2590
|
+
(e as { type: "text_delta"; text: string }).text ===
|
|
2591
|
+
REFUSAL_FALLBACK_TEXT,
|
|
2592
|
+
),
|
|
2593
|
+
).toBe(true);
|
|
2594
|
+
});
|
|
2595
|
+
|
|
2596
|
+
// The fallback must not pile a spurious apology beneath a real answer: when
|
|
2597
|
+
// an earlier turn this run already produced visible text (text alongside a
|
|
2598
|
+
// tool call), a refusal on the trailing turn after the tool result must not
|
|
2599
|
+
// rewrite the turn even though a refusal would normally trigger it.
|
|
2600
|
+
test("does not substitute a fallback when the run already produced visible text", async () => {
|
|
2601
|
+
// GIVEN a first turn with text + a tool call, then a refusal after the
|
|
2602
|
+
// tool result.
|
|
2603
|
+
const textPlusToolUse: ProviderResponse = {
|
|
2604
|
+
content: [
|
|
2605
|
+
{ type: "text", text: "Here is your answer." },
|
|
2606
|
+
{
|
|
2607
|
+
type: "tool_use",
|
|
2608
|
+
id: "t1",
|
|
2609
|
+
name: "read_file",
|
|
2610
|
+
input: { path: "/a" },
|
|
2611
|
+
},
|
|
2612
|
+
],
|
|
2613
|
+
model: "mock-model",
|
|
2614
|
+
usage: { inputTokens: 10, outputTokens: 5 },
|
|
2615
|
+
stopReason: "tool_use",
|
|
2616
|
+
};
|
|
2617
|
+
const refusalResponse: ProviderResponse = {
|
|
2618
|
+
content: [],
|
|
2619
|
+
model: "mock-model",
|
|
2620
|
+
usage: { inputTokens: 10, outputTokens: 0 },
|
|
2621
|
+
stopReason: "refusal",
|
|
2622
|
+
};
|
|
2623
|
+
const { provider } = createMockProvider([textPlusToolUse, refusalResponse]);
|
|
2624
|
+
const loop = new AgentLoop({
|
|
2625
|
+
provider: provider,
|
|
2626
|
+
systemPrompt: "system",
|
|
2627
|
+
conversationId: "test-conversation",
|
|
2628
|
+
tools: dummyTools,
|
|
2629
|
+
toolExecutor: async () => ({ content: "data", isError: false }),
|
|
2630
|
+
});
|
|
2631
|
+
|
|
2632
|
+
// WHEN the loop runs.
|
|
2633
|
+
const events: AgentEvent[] = [];
|
|
2634
|
+
const { history } = await loop.run({
|
|
2635
|
+
requestId: "test-request",
|
|
2636
|
+
messages: [userMessage],
|
|
2637
|
+
onEvent: collectEvents(events),
|
|
2638
|
+
trust: { sourceChannel: "vellum", trustClass: "unknown" },
|
|
2639
|
+
});
|
|
2640
|
+
|
|
2641
|
+
// THEN no fallback text is emitted or persisted — the real answer stands
|
|
2642
|
+
// alone and the trailing empty turn stays empty.
|
|
2643
|
+
const fallbackEmitted = events.some(
|
|
2644
|
+
(e) =>
|
|
2645
|
+
e.type === "text_delta" &&
|
|
2646
|
+
(e as { type: "text_delta"; text: string }).text ===
|
|
2647
|
+
REFUSAL_FALLBACK_TEXT,
|
|
2648
|
+
);
|
|
2649
|
+
expect(fallbackEmitted).toBe(false);
|
|
2650
|
+
const fallbackInHistory = history.some(
|
|
2651
|
+
(m) =>
|
|
2652
|
+
m.role === "assistant" &&
|
|
2653
|
+
m.content.some(
|
|
2654
|
+
(b) =>
|
|
2655
|
+
b.type === "text" &&
|
|
2656
|
+
"text" in b &&
|
|
2657
|
+
(b as { text: string }).text === REFUSAL_FALLBACK_TEXT,
|
|
2658
|
+
),
|
|
2659
|
+
);
|
|
2660
|
+
expect(fallbackInHistory).toBe(false);
|
|
2661
|
+
});
|
|
2662
|
+
|
|
2663
|
+
// A native web-search turn lands at the stop boundary with no `tool_use` and
|
|
2664
|
+
// no visible text, but with `stopReason: "end_turn"` (not a refusal) and
|
|
2665
|
+
// `server_tool_use`/`web_search_tool_result` blocks that render the search
|
|
2666
|
+
// card. Because the fallback is refusal-specific, it must not fire here and
|
|
2667
|
+
// the server-tool blocks must persist untouched.
|
|
2668
|
+
test("does not substitute a fallback for a server-tool turn with no text", async () => {
|
|
2669
|
+
// GIVEN a turn carrying only native web-search blocks (no text, no
|
|
2670
|
+
// client-side tool_use).
|
|
2671
|
+
const serverToolBlocks: ContentBlock[] = [
|
|
2672
|
+
{
|
|
2673
|
+
type: "server_tool_use",
|
|
2674
|
+
id: "srv1",
|
|
2675
|
+
name: "web_search",
|
|
2676
|
+
input: { query: "weather" },
|
|
2677
|
+
},
|
|
2678
|
+
{
|
|
2679
|
+
type: "web_search_tool_result",
|
|
2680
|
+
tool_use_id: "srv1",
|
|
2681
|
+
content: [{ title: "result" }],
|
|
2682
|
+
},
|
|
2683
|
+
];
|
|
2684
|
+
const webSearchResponse: ProviderResponse = {
|
|
2685
|
+
content: serverToolBlocks,
|
|
2686
|
+
model: "mock-model",
|
|
2687
|
+
usage: { inputTokens: 10, outputTokens: 5 },
|
|
2688
|
+
stopReason: "end_turn",
|
|
2689
|
+
};
|
|
2690
|
+
const { provider, calls } = createMockProvider([webSearchResponse]);
|
|
2691
|
+
const loop = new AgentLoop({
|
|
2692
|
+
provider: provider,
|
|
2693
|
+
systemPrompt: "system",
|
|
2694
|
+
conversationId: "test-conversation",
|
|
2695
|
+
});
|
|
2696
|
+
|
|
2697
|
+
// WHEN the loop runs.
|
|
2698
|
+
const events: AgentEvent[] = [];
|
|
2699
|
+
const { history } = await loop.run({
|
|
2700
|
+
requestId: "test-request",
|
|
2701
|
+
messages: [userMessage],
|
|
2702
|
+
onEvent: collectEvents(events),
|
|
2703
|
+
trust: { sourceChannel: "vellum", trustClass: "unknown" },
|
|
2704
|
+
});
|
|
2705
|
+
|
|
2706
|
+
// THEN the provider is called once (no nudge for a non-refusal turn) and
|
|
2707
|
+
// the server-tool blocks are persisted untouched, with no fallback added.
|
|
2708
|
+
expect(calls).toHaveLength(1);
|
|
2709
|
+
const lastAssistant = [...history]
|
|
2710
|
+
.reverse()
|
|
2711
|
+
.find((m) => m.role === "assistant");
|
|
2712
|
+
expect(lastAssistant!.content).toEqual(serverToolBlocks);
|
|
2713
|
+
|
|
2714
|
+
const fallbackEmitted = events.some(
|
|
2715
|
+
(e) =>
|
|
2716
|
+
e.type === "text_delta" &&
|
|
2717
|
+
(e as { type: "text_delta"; text: string }).text ===
|
|
2718
|
+
REFUSAL_FALLBACK_TEXT,
|
|
2719
|
+
);
|
|
2720
|
+
expect(fallbackEmitted).toBe(false);
|
|
2150
2721
|
});
|
|
2151
2722
|
|
|
2152
2723
|
// PR 6: callSite threading from AgentLoop.run() into provider config.
|
|
@@ -2161,6 +2732,7 @@ describe("AgentLoop", () => {
|
|
|
2161
2732
|
conversationId: "test-conversation",
|
|
2162
2733
|
});
|
|
2163
2734
|
await loop.run({
|
|
2735
|
+
requestId: "test-request",
|
|
2164
2736
|
messages: [userMessage],
|
|
2165
2737
|
onEvent: () => {},
|
|
2166
2738
|
trust: { sourceChannel: "vellum", trustClass: "unknown" },
|
|
@@ -2180,6 +2752,7 @@ describe("AgentLoop", () => {
|
|
|
2180
2752
|
conversationId: "test-conversation",
|
|
2181
2753
|
});
|
|
2182
2754
|
await loop.run({
|
|
2755
|
+
requestId: "test-request",
|
|
2183
2756
|
messages: [userMessage],
|
|
2184
2757
|
onEvent: () => {},
|
|
2185
2758
|
trust: { sourceChannel: "vellum", trustClass: "unknown" },
|