@vellumai/assistant 0.8.12 → 0.9.0-staging.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/AGENTS.md +0 -14
- package/ARCHITECTURE.md +45 -45
- package/README.md +1 -1
- package/bun.lock +200 -154
- package/docs/architecture/integrations.md +3 -3
- package/docs/architecture/memory.md +2 -2
- package/docs/architecture/security.md +10 -10
- package/docs/runbook-trusted-contacts.md +12 -12
- package/docs/skills.md +6 -6
- package/docs/workflows-testing.md +221 -0
- package/docs/workflows.md +510 -0
- package/examples/plugins/echo/README.md +5 -5
- package/knip.json +2 -0
- package/node_modules/@vellumai/gateway-client/src/inbound-contract.ts +105 -0
- package/node_modules/@vellumai/gateway-client/src/index.ts +12 -0
- package/openapi.yaml +7197 -5708
- package/package.json +8 -4
- package/scripts/generate-openapi.ts +66 -114
- package/src/__tests__/access-request-seed-content-blocks.test.ts +213 -0
- package/src/__tests__/adaptive-thinking-repair.test.ts +32 -3
- package/src/__tests__/agent-loop-output-hooks.test.ts +183 -0
- package/src/__tests__/agent-loop-regrowth-guard.test.ts +506 -0
- package/src/__tests__/agent-wake-disk-pressure-callsite.test.ts +2 -0
- package/src/__tests__/agent-wake-override-profile.test.ts +77 -0
- package/src/__tests__/app-compiler.test.ts +7 -1
- package/src/__tests__/app-dir-path-guard.test.ts +27 -3
- package/src/__tests__/app-executors.test.ts +43 -0
- package/src/__tests__/approval-cascade.test.ts +0 -5
- package/src/__tests__/approval-routes-http.test.ts +91 -0
- package/src/__tests__/assistant-stream-state.test.ts +107 -0
- package/src/__tests__/browser-fill-credential.test.ts +3 -3
- package/src/__tests__/bundled-skill-retrieval-guard.test.ts +1 -1
- package/src/__tests__/compaction-events.test.ts +63 -7
- package/src/__tests__/compaction-trail-store.test.ts +74 -1
- package/src/__tests__/compaction.benchmark.test.ts +63 -41
- package/src/__tests__/compactor-low-watermark-cut.test.ts +349 -0
- package/src/__tests__/context-window-manager-compact-retry.test.ts +64 -0
- package/src/__tests__/conversation-abort-tool-results.test.ts +0 -5
- package/src/__tests__/conversation-confirmation-signals.test.ts +0 -5
- package/src/__tests__/conversation-history-web-search.test.ts +7 -0
- package/src/__tests__/conversation-process-callsite.test.ts +0 -5
- package/src/__tests__/conversation-provider-retry-repair.test.ts +0 -5
- package/src/__tests__/conversation-queue.test.ts +0 -5
- package/src/__tests__/conversation-slash-queue.test.ts +0 -5
- package/src/__tests__/conversation-slash-unknown.test.ts +0 -5
- package/src/__tests__/conversation-speed-override.test.ts +0 -5
- package/src/__tests__/conversation-surfaces-data-persist.test.ts +97 -0
- package/src/__tests__/conversation-surfaces-task-progress.test.ts +67 -0
- package/src/__tests__/conversation-usage.test.ts +2 -0
- package/src/__tests__/conversation-workspace-injection.test.ts +0 -5
- package/src/__tests__/conversation-workspace-tool-tracking.test.ts +0 -5
- package/src/__tests__/credential-broker-browser-fill.test.ts +2 -2
- package/src/__tests__/credential-broker-server-use.test.ts +2 -2
- package/src/__tests__/credential-broker.test.ts +1 -1
- package/src/__tests__/credential-prompt-route.test.ts +417 -0
- package/src/__tests__/credential-security-invariants.test.ts +1 -0
- package/src/__tests__/credential-vault.test.ts +37 -0
- package/src/__tests__/db-schedule-syntax-migration.test.ts +24 -0
- package/src/__tests__/dynamic-page-surface.test.ts +219 -0
- package/src/__tests__/empty-state-greeting-cache.test.ts +94 -0
- package/src/__tests__/gateway-flag-listener.test.ts +24 -7
- package/src/__tests__/guardian-action-sweep.test.ts +56 -219
- package/src/__tests__/guardian-routing-invariants.test.ts +138 -0
- package/src/__tests__/helpers/channel-test-adapter.ts +0 -2
- package/src/__tests__/list-messages-hidden-metadata.test.ts +99 -0
- package/src/__tests__/llm-request-log-source-clickhouse.test.ts +87 -1
- package/src/__tests__/llm-resolver.test.ts +115 -0
- package/src/__tests__/managed-profile-guard.test.ts +6 -5
- package/src/__tests__/max-tokens-continue-hook.test.ts +184 -0
- package/src/__tests__/media-generate-image.test.ts +20 -9
- package/src/__tests__/mock-gateway-ipc.ts +23 -0
- package/src/__tests__/model-intents.test.ts +1 -1
- package/src/__tests__/normalize-onboarding.test.ts +26 -0
- package/src/__tests__/notification-decision-strategy.test.ts +4 -2
- package/src/__tests__/notification-telegram-adapter.test.ts +21 -3
- package/src/__tests__/pending-interactions-resolved-event.test.ts +62 -0
- package/src/__tests__/post-turn-tool-result-truncation.test.ts +72 -18
- package/src/__tests__/require-fresh-approval.test.ts +425 -1
- package/src/__tests__/resolve-app-id.test.ts +56 -0
- package/src/__tests__/runtime-events-sse-parity.test.ts +2 -0
- package/src/__tests__/schedule-routes-workflow-validation.test.ts +408 -0
- package/src/__tests__/schedule-routes.test.ts +257 -4
- package/src/__tests__/schedule-store.test.ts +60 -0
- package/src/__tests__/schedule-tools.test.ts +247 -2
- package/src/__tests__/skill-execute-input.test.ts +85 -0
- package/src/__tests__/skill-secret-handling-guard.test.ts +21 -20
- package/src/__tests__/skills.test.ts +3 -3
- package/src/__tests__/slack-app-setup-skill-regression.test.ts +1 -1
- package/src/__tests__/subagent-tool-filtering.test.ts +50 -0
- package/src/__tests__/subagent-tool-gate-mode.test.ts +547 -0
- package/src/__tests__/system-prompt.test.ts +1 -1
- package/src/__tests__/task-progress-nudge-hook.test.ts +372 -0
- package/src/__tests__/task-scheduler.test.ts +299 -0
- package/src/__tests__/tool-approval-seed-content-blocks.test.ts +209 -0
- package/src/__tests__/tool-result-spool.test.ts +3 -1
- package/src/__tests__/workspace-migration-102-preserve-heartbeat-enabled-for-existing-workspaces.test.ts +181 -0
- package/src/__tests__/workspace-migration-103-upgrade-quality-profile-to-opus-4-8.test.ts +174 -0
- package/src/agent/compaction-circuit.ts +11 -0
- package/src/agent/loop.ts +181 -12
- package/src/api/constants/call-sites.ts +12 -0
- package/src/api/events/assistant-thinking-delta.ts +10 -0
- package/src/api/events/usage-update.ts +7 -0
- package/src/api/index.ts +4 -1
- package/src/api/responses/memory-v3-selection-log.ts +18 -11
- package/src/approvals/approval-primitive.ts +2 -2
- package/src/background-wake/background-wake-routes.test.ts +5 -2
- package/src/bundler/compiler-tools.ts +1 -1
- package/src/bundler/package-resolver.ts +0 -1
- package/src/calls/call-domain.ts +1 -1
- package/src/calls/guardian-action-sweep.ts +16 -93
- package/src/cli/AGENTS.md +4 -0
- package/src/cli/commands/__tests__/schedules.test.ts +430 -1
- package/src/cli/commands/credentials.ts +28 -24
- package/src/cli/commands/image-generation.ts +23 -9
- package/src/cli/commands/notifications.ts +1 -1
- package/src/cli/commands/plugins.ts +89 -46
- package/src/cli/commands/schedules.ts +384 -11
- package/src/cli/lib/__tests__/inspect-plugin.test.ts +69 -5
- package/src/cli/lib/__tests__/install-from-github.test.ts +15 -0
- package/src/cli/lib/__tests__/upgrade-plugin.test.ts +81 -4
- package/src/cli/lib/inspect-plugin.ts +62 -1
- package/src/cli/lib/install-from-github.ts +52 -5
- package/src/cli/lib/upgrade-plugin.ts +18 -0
- package/src/config/__tests__/workflows-schema.test.ts +60 -0
- package/src/config/bundled-skills/acp/SKILL.md +2 -2
- package/src/config/bundled-skills/app-builder/SKILL.md +1 -1
- package/src/config/bundled-skills/app-builder/tools/app-create.ts +6 -1
- package/src/config/bundled-skills/app-builder/tools/app-generate-icon.ts +7 -1
- package/src/config/bundled-skills/app-builder/tools/app-refresh.ts +7 -1
- package/src/config/bundled-skills/app-builder/tools/app-update.ts +10 -1
- package/src/config/bundled-skills/image-studio/SKILL.md +66 -19
- package/src/config/bundled-skills/image-studio/TOOLS.json +1 -6
- package/src/config/bundled-skills/image-studio/tools/media-generate-image.ts +22 -3
- package/src/config/bundled-skills/personal-page/SKILL.md +57 -0
- package/src/config/bundled-skills/personal-page/TOOLS.json +27 -0
- package/src/config/bundled-skills/personal-page/tools/app-refresh.ts +17 -0
- package/src/config/bundled-skills/schedule/SKILL.md +7 -2
- package/src/config/bundled-skills/schedule/TOOLS.json +48 -4
- package/src/config/bundled-skills/workflows/SKILL.md +214 -0
- package/src/config/bundled-skills/workflows/TOOLS.json +84 -0
- package/src/config/bundled-skills/workflows/tools/manage-workflows.ts +12 -0
- package/src/config/bundled-skills/workflows/tools/run-workflow.ts +12 -0
- package/src/config/bundled-tool-registry.ts +14 -2
- package/src/config/call-site-defaults.ts +5 -0
- package/src/config/feature-flag-registry.json +12 -4
- package/src/config/llm-context-resolution.ts +8 -0
- package/src/config/llm-resolver.ts +30 -0
- package/src/config/preloaded-apps/personal-page/src/components/About.tsx +22 -0
- package/src/config/preloaded-apps/personal-page/src/components/App.tsx +16 -0
- package/src/config/preloaded-apps/personal-page/src/components/Features.tsx +77 -0
- package/src/config/preloaded-apps/personal-page/src/components/Hero.tsx +57 -0
- package/src/config/preloaded-apps/personal-page/src/components/Pending.tsx +28 -0
- package/src/config/preloaded-apps/personal-page/src/components/animations.tsx +234 -0
- package/src/config/preloaded-apps/personal-page/src/components/icons.tsx +48 -0
- package/src/config/preloaded-apps/personal-page/src/components/media.ts +16 -0
- package/src/config/preloaded-apps/personal-page/src/index.html +20 -0
- package/src/config/preloaded-apps/personal-page/src/main.tsx +7 -0
- package/src/config/preloaded-apps/personal-page/src/profile-data.ts +82 -0
- package/src/config/preloaded-apps/personal-page/src/styles.css +759 -0
- package/src/config/schema.ts +2 -0
- package/src/config/schemas/call-site-catalog.ts +7 -0
- package/src/config/schemas/heartbeat.ts +4 -1
- package/src/config/schemas/llm.ts +33 -26
- package/src/config/schemas/memory-retrospective.ts +19 -0
- package/src/config/schemas/platform.ts +8 -0
- package/src/config/schemas/services.ts +5 -2
- package/src/config/schemas/workflows.ts +42 -0
- package/src/config/skills.ts +3 -3
- package/src/context/compactor.ts +273 -39
- package/src/context/post-turn-tool-result-truncation.ts +23 -6
- package/src/context/tool-result-spool.ts +12 -17
- package/src/credential-execution/executable-discovery.ts +1 -1
- package/src/credential-execution/process-manager.ts +37 -3
- package/src/credential-execution/prompted-credential.ts +205 -0
- package/src/daemon/conversation-agent-loop-handlers.ts +14 -0
- package/src/daemon/conversation-process.ts +11 -2
- package/src/daemon/conversation-surfaces.ts +167 -3
- package/src/daemon/conversation-tool-setup.ts +103 -26
- package/src/daemon/conversation-usage.ts +2 -0
- package/src/daemon/conversation.ts +115 -11
- package/src/daemon/handlers/shared.ts +26 -14
- package/src/daemon/host-cu-proxy.ts +15 -12
- package/src/daemon/host-file-proxy.ts +15 -12
- package/src/daemon/host-transfer-proxy.ts +30 -24
- package/src/daemon/lifecycle.ts +40 -3
- package/src/daemon/message-protocol.ts +3 -0
- package/src/daemon/message-types/messages.ts +2 -10
- package/src/daemon/message-types/workflows.ts +49 -0
- package/src/daemon/parse-actual-tokens-from-error.test.ts +62 -1
- package/src/daemon/parse-actual-tokens-from-error.ts +43 -4
- package/src/daemon/process-message.ts +6 -0
- package/src/daemon/tool-setup-types.ts +57 -0
- package/src/daemon/wake-conversation-ops.ts +18 -0
- package/src/heartbeat/heartbeat-run-store.ts +8 -2
- package/src/home/feed-types.ts +1 -1
- package/src/ipc/gateway-flag-listener.ts +28 -6
- package/src/mcp/mcp-auth-state.ts +8 -20
- package/src/media/__tests__/image-models.test.ts +57 -0
- package/src/media/image-models.ts +66 -0
- package/src/memory/__tests__/auto-analysis-enqueue.test.ts +38 -0
- package/src/memory/__tests__/find-most-recent-retrospective-for.test.ts +12 -2
- package/src/memory/__tests__/memory-retrospective-job.test.ts +911 -34
- package/src/memory/__tests__/memory-retrospective-startup-cleanup.test.ts +227 -5
- package/src/memory/__tests__/memory-retrospective-state.test.ts +195 -0
- package/src/memory/__tests__/preloaded-apps.test.ts +85 -0
- package/src/memory/auto-analysis-enqueue.ts +14 -1
- package/src/memory/compaction-log-store-clickhouse.ts +6 -4
- package/src/memory/conversation-crud.ts +9 -2
- package/src/memory/conversation-disk-view.ts +1 -1
- package/src/memory/conversation-queries.ts +22 -7
- package/src/memory/db-init.ts +20 -0
- package/src/memory/db-maintenance.ts +16 -0
- package/src/memory/embedding-runtime-manager.ts +1 -1
- package/src/memory/llm-request-log-source-clickhouse.ts +112 -14
- package/src/memory/llm-request-log-source-local.ts +19 -1
- package/src/memory/llm-request-log-source.ts +35 -6
- package/src/memory/llm-request-log-store.ts +90 -2
- package/src/memory/memory-retrospective-constants.ts +9 -0
- package/src/memory/memory-retrospective-enqueue.ts +3 -6
- package/src/memory/memory-retrospective-fork-boundary.ts +94 -0
- package/src/memory/memory-retrospective-job.ts +500 -208
- package/src/memory/memory-retrospective-startup-cleanup.ts +97 -19
- package/src/memory/memory-retrospective-state.ts +85 -2
- package/src/memory/migrations/281-memory-retrospective-remembered-log.ts +40 -0
- package/src/memory/migrations/282-schedule-inference-profile.test.ts +77 -0
- package/src/memory/migrations/282-schedule-inference-profile.ts +26 -0
- package/src/memory/migrations/283-memory-v3-selections-message-id-and-sections.test.ts +102 -0
- package/src/memory/migrations/283-memory-v3-selections-message-id-and-sections.ts +53 -0
- package/src/memory/migrations/284-workflow-runs.ts +51 -0
- package/src/memory/migrations/285-schedule-workflow-mode.ts +26 -0
- package/src/memory/migrations/286-workflow-run-trust.ts +27 -0
- package/src/memory/migrations/287-conversation-origin-channel-index.ts +15 -0
- package/src/memory/migrations/288-backfill-origin-channel-from-bindings.ts +43 -0
- package/src/memory/migrations/289-contact-channels-unique-ext-user.ts +115 -0
- package/src/memory/migrations/290-schedule-capabilities.test.ts +77 -0
- package/src/memory/migrations/290-schedule-capabilities.ts +25 -0
- package/src/memory/migrations/__tests__/281-memory-retrospective-remembered-log.test.ts +96 -0
- package/src/memory/migrations/__tests__/289-contact-channels-unique-ext-user.test.ts +571 -0
- package/src/memory/migrations/index.ts +10 -0
- package/src/memory/preloaded-apps.ts +116 -0
- package/src/memory/schema/infrastructure.ts +4 -0
- package/src/memory/schema/memory-core.ts +4 -0
- package/src/memory/v2/__tests__/concept-page-frontmatter-schema.test.ts +45 -0
- package/src/memory/v2/__tests__/frontmatter-sweep.test.ts +11 -7
- package/src/memory/v2/__tests__/page-store.test.ts +13 -2
- package/src/memory/v2/__tests__/qdrant.test.ts +24 -0
- package/src/memory/v2/frontmatter-sweep.ts +7 -6
- package/src/memory/v2/page-store.ts +4 -3
- package/src/memory/v2/qdrant.ts +42 -3
- package/src/memory/v2/types.ts +16 -10
- package/src/messaging/draft-store.ts +1 -1
- package/src/notifications/access-request-copy.ts +200 -113
- package/src/notifications/adapters/slack.ts +250 -111
- package/src/notifications/adapters/telegram.ts +7 -44
- package/src/notifications/approval-card-builder.ts +93 -0
- package/src/notifications/broadcaster.ts +74 -0
- package/src/notifications/conversation-pairing.ts +8 -6
- package/src/notifications/copy-composer.ts +32 -26
- package/src/notifications/decision-engine.ts +59 -7
- package/src/notifications/guardian-question-mode.ts +145 -155
- package/src/notifications/home-feed-side-effect.ts +28 -11
- package/src/notifications/notification-utils.ts +66 -0
- package/src/notifications/signal.ts +6 -0
- package/src/notifications/tool-approval-copy.ts +142 -0
- package/src/notifications/types.ts +19 -0
- package/src/permissions/threshold.ts +11 -0
- package/src/plugin-api/types.ts +16 -4
- package/src/plugins/defaults/compaction/window-manager.ts +44 -0
- package/src/plugins/defaults/index.ts +46 -0
- package/src/plugins/defaults/max-tokens-continue/continue-state-store.ts +53 -0
- package/src/plugins/defaults/max-tokens-continue/hooks/post-model-call.ts +80 -0
- package/src/plugins/defaults/max-tokens-continue/hooks/stop.ts +20 -0
- package/src/plugins/defaults/max-tokens-continue/package.json +14 -0
- package/src/plugins/defaults/memory-retrieval/hooks/__tests__/user-prompt-submit.test.ts +37 -0
- package/src/plugins/defaults/memory-retrieval/hooks/user-prompt-submit.ts +29 -1
- package/src/plugins/defaults/memory-v3-shadow/__tests__/carry-integration.test.ts +8 -3
- package/src/plugins/defaults/memory-v3-shadow/__tests__/injection.test.ts +4 -2
- package/src/plugins/defaults/memory-v3-shadow/__tests__/pool-select.test.ts +28 -18
- package/src/plugins/defaults/memory-v3-shadow/__tests__/section-dense-store.test.ts +67 -0
- package/src/plugins/defaults/memory-v3-shadow/__tests__/selection-log-store.test.ts +122 -22
- package/src/plugins/defaults/memory-v3-shadow/__tests__/shadow-integration.test.ts +2 -0
- package/src/plugins/defaults/memory-v3-shadow/__tests__/shadow-plugin.test.ts +63 -1
- package/src/plugins/defaults/memory-v3-shadow/injector.ts +61 -18
- package/src/plugins/defaults/memory-v3-shadow/orchestrate.ts +1 -1
- package/src/plugins/defaults/memory-v3-shadow/pool-select.ts +39 -10
- package/src/plugins/defaults/memory-v3-shadow/section-dense-store.ts +34 -1
- package/src/plugins/defaults/memory-v3-shadow/selection-log-store.ts +112 -47
- package/src/plugins/defaults/memory-v3-shadow/shadow-plugin.ts +78 -15
- package/src/plugins/defaults/task-progress-nudge/hooks/post-tool-use.ts +206 -0
- package/src/plugins/defaults/task-progress-nudge/package.json +15 -0
- package/src/prompts/__tests__/system-prompt.test.ts +100 -1
- package/src/prompts/__tests__/task-progress-hint-section.test.ts +5 -7
- package/src/prompts/normalize-onboarding.ts +2 -0
- package/src/prompts/persona-resolver.ts +3 -0
- package/src/prompts/system-prompt.ts +51 -2
- package/src/prompts/templates/BOOTSTRAP-ACTIVATION-RAIL.md +3 -1
- package/src/prompts/templates/system-sections.ts +8 -3
- package/src/providers/call-site-routing.ts +6 -3
- package/src/providers/fireworks/client.ts +3 -0
- package/src/providers/inference/auth.ts +52 -46
- package/src/providers/model-intents.ts +2 -2
- package/src/providers/openai/__tests__/coerce-object-args.test.ts +105 -0
- package/src/providers/openai/chat-completions-provider.ts +47 -9
- package/src/providers/openai/coerce-object-args.ts +104 -0
- package/src/providers/retry.ts +8 -5
- package/src/providers/types.ts +10 -0
- package/src/runtime/__tests__/agent-wake.test.ts +629 -7
- package/src/runtime/access-request-helper.ts +16 -9
- package/src/runtime/agent-wake.ts +302 -51
- package/src/runtime/assistant-stream-state.ts +141 -8
- package/src/runtime/background-job-runner.ts +9 -0
- package/src/runtime/channel-approval-types.ts +1 -0
- package/src/runtime/channel-invite-transports/telegram.ts +6 -5
- package/src/runtime/channel-invite-transports/voice.ts +2 -2
- package/src/runtime/channel-invite-types.ts +4 -2
- package/src/runtime/channel-retry-sweep.ts +19 -41
- package/src/runtime/finalize-event-delivery.ts +72 -0
- package/src/runtime/guardian-action-message-composer.ts +0 -54
- package/src/runtime/http-server.ts +6 -14
- package/src/runtime/http-types.ts +0 -1
- package/src/runtime/message-composer-types.ts +0 -9
- package/src/runtime/middleware/__tests__/rate-limiter.test.ts +63 -0
- package/src/runtime/middleware/auth.ts +27 -3
- package/src/runtime/middleware/rate-limiter.ts +28 -1
- package/src/runtime/migrations/vbundle-builder.ts +6 -5
- package/src/runtime/pending-interactions.ts +20 -1
- package/src/runtime/routes/__tests__/conversation-compaction-routes.test.ts +232 -173
- package/src/runtime/routes/__tests__/plugins-routes.test.ts +18 -0
- package/src/runtime/routes/__tests__/retrospective-routes.test.ts +436 -0
- package/src/runtime/routes/__tests__/surface-action-routes.test.ts +11 -0
- package/src/runtime/routes/approval-routes.ts +35 -8
- package/src/runtime/routes/approval-strategies/guardian-callback-strategy.ts +2 -1
- package/src/runtime/routes/btw-routes.ts +37 -1
- package/src/runtime/routes/channel-delivery-routes.ts +11 -7
- package/src/runtime/routes/channel-route-definitions.ts +3 -0
- package/src/runtime/routes/channel-route-shared.ts +3 -1
- package/src/runtime/routes/consolidation-routes.ts +17 -13
- package/src/runtime/routes/conversation-compaction-routes.ts +159 -119
- package/src/runtime/routes/conversation-list-routes.ts +41 -4
- package/src/runtime/routes/conversation-query-routes.ts +196 -11
- package/src/runtime/routes/conversation-routes.ts +15 -1
- package/src/runtime/routes/credential-prompt-routes.ts +39 -17
- package/src/runtime/routes/empty-state-greeting-cache.ts +65 -0
- package/src/runtime/routes/heartbeat-routes.ts +18 -13
- package/src/runtime/routes/image-generation-routes.ts +20 -2
- package/src/runtime/routes/inbound-message-handler.ts +13 -12
- package/src/runtime/routes/inbound-stages/acl-enforcement.ts +32 -31
- package/src/runtime/routes/inbound-stages/background-dispatch.test.ts +7 -5
- package/src/runtime/routes/inbound-stages/background-dispatch.ts +7 -21
- package/src/runtime/routes/inbound-stages/escalation-intercept.ts +5 -5
- package/src/runtime/routes/inbound-stages/guardian-activation-intercept.ts +6 -15
- package/src/runtime/routes/inbound-stages/secret-ingress-check.ts +1 -1
- package/src/runtime/routes/index.ts +6 -0
- package/src/runtime/routes/log-export/AGENTS.md +1 -1
- package/src/runtime/routes/log-export/workspace-allowlist.ts +1 -1
- package/src/runtime/routes/migration-routes.ts +5 -9
- package/src/runtime/routes/plugins-routes.ts +26 -0
- package/src/runtime/routes/ps-routes.ts +10 -8
- package/src/runtime/routes/retrospective-routes.ts +235 -0
- package/src/runtime/routes/runs-pagination.ts +75 -0
- package/src/runtime/routes/schedule-routes.ts +367 -59
- package/src/runtime/routes/sounds-config-routes.ts +239 -0
- package/src/runtime/routes/surface-action-routes.ts +84 -4
- package/src/runtime/routes/workflow-routes.test.ts +372 -0
- package/src/runtime/routes/workflow-routes.ts +363 -0
- package/src/runtime/routes/workspace-routes.test.ts +61 -1
- package/src/runtime/routes/workspace-routes.ts +26 -1
- package/src/runtime/services/__tests__/analyze-conversation.test.ts +38 -0
- package/src/runtime/services/analyze-conversation.ts +26 -13
- package/src/schedule/inference-profile.ts +28 -0
- package/src/schedule/schedule-store.ts +152 -4
- package/src/schedule/scheduler-types.ts +6 -0
- package/src/schedule/scheduler.ts +96 -0
- package/src/security/secret-allowlist.ts +1 -1
- package/src/skills/path-classifier.ts +1 -1
- package/src/tools/apps/executors.ts +23 -0
- package/src/tools/apps/resolve-app-id.ts +42 -0
- package/src/tools/browser/browser-execution.ts +9 -11
- package/src/tools/credentials/broker.ts +4 -4
- package/src/tools/credentials/vault.ts +26 -137
- package/src/tools/executor.ts +69 -0
- package/src/tools/flag-gated-tools.test.ts +76 -0
- package/src/tools/permission-checker.ts +8 -1
- package/src/tools/registry.ts +51 -0
- package/src/tools/schedule/create.ts +77 -1
- package/src/tools/schedule/list.ts +1 -0
- package/src/tools/schedule/update.ts +73 -1
- package/src/tools/skills/execute.ts +56 -0
- package/src/tools/terminal/shell.ts +1 -1
- package/src/tools/ui-surface/definitions.ts +91 -2
- package/src/tools/workflows/manage-workflows.ts +183 -0
- package/src/tools/workflows/run-workflow.test.ts +442 -0
- package/src/tools/workflows/run-workflow.ts +88 -0
- package/src/types/onboarding-context.ts +2 -0
- package/src/usage/attribution.ts +24 -0
- package/src/util/canonicalize-identity.ts +12 -3
- package/src/util/platform.ts +17 -17
- package/src/watcher/__tests__/engine.test.ts +24 -0
- package/src/watcher/__tests__/telemetry.test.ts +135 -0
- package/src/watcher/engine.ts +7 -0
- package/src/watcher/telemetry.ts +74 -0
- package/src/workflows/capabilities.test.ts +365 -0
- package/src/workflows/capabilities.ts +359 -0
- package/src/workflows/deterministic-stringify.ts +27 -0
- package/src/workflows/engine-integration.test.ts +656 -0
- package/src/workflows/engine.test.ts +1144 -0
- package/src/workflows/engine.ts +1078 -0
- package/src/workflows/fanout-load.test.ts +168 -0
- package/src/workflows/journal-store.test.ts +369 -0
- package/src/workflows/journal-store.ts +470 -0
- package/src/workflows/leaf-runner.test.ts +704 -0
- package/src/workflows/leaf-runner.ts +589 -0
- package/src/workflows/library.test.ts +134 -0
- package/src/workflows/library.ts +124 -0
- package/src/workflows/run-manager.test.ts +711 -0
- package/src/workflows/run-manager.ts +593 -0
- package/src/workflows/sandbox-escape.test.ts +339 -0
- package/src/workflows/sandbox.test.ts +251 -0
- package/src/workflows/sandbox.ts +447 -0
- package/src/workspace/adaptive-thinking-repair.ts +33 -11
- package/src/workspace/migrations/021-move-signals-to-workspace.ts +1 -1
- package/src/workspace/migrations/022-move-hooks-to-workspace.ts +1 -1
- package/src/workspace/migrations/026-backfill-install-meta.ts +1 -1
- package/src/workspace/migrations/030-seed-pkb-autoinject.ts +2 -1
- package/src/workspace/migrations/031-drop-user-md.ts +1 -4
- package/src/workspace/migrations/048-remove-workspace-hooks.ts +1 -1
- package/src/workspace/migrations/056-release-notes-inference-profile-reordering.ts +5 -2
- package/src/workspace/migrations/061-move-backup-key-to-workspace.ts +1 -1
- package/src/workspace/migrations/082-backfill-managed-profile-labels.ts +8 -2
- package/src/workspace/migrations/097-enable-adaptive-thinking-managed-profiles.ts +41 -14
- package/src/workspace/migrations/102-preserve-heartbeat-enabled-for-existing-workspaces.ts +69 -0
- package/src/workspace/migrations/103-upgrade-quality-profile-to-opus-4-8.ts +83 -0
- package/src/workspace/migrations/104-recheck-adaptive-thinking-model-implied-anthropic.ts +133 -0
- package/src/workspace/migrations/registry.ts +6 -0
- package/src/workspace/migrations/runner.ts +1 -1
- package/tsconfig.json +1 -1
- package/src/__tests__/guardian-action-copy-generator.test.ts +0 -200
- package/src/__tests__/guardian-action-grant-mint-consume.test.ts +0 -579
- package/src/__tests__/guardian-action-store.test.ts +0 -106
- package/src/daemon/guardian-action-generators.ts +0 -71
- package/src/memory/guardian-action-store.ts +0 -484
- package/src/runtime/guardian-action-grant-minter.ts +0 -150
|
@@ -0,0 +1,206 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Default `post-tool-use` hook: when an interactive turn accumulates several
|
|
3
|
+
* tool-call rounds without the model having shown a `task_progress` card,
|
|
4
|
+
* surface a soft notice via `additionalContext` reminding it to show progress.
|
|
5
|
+
*
|
|
6
|
+
* Motivation: the system prompt and the `ui_show` tool description both ask
|
|
7
|
+
* the model to show a `task_progress` card on long multi-step turns, but
|
|
8
|
+
* weaker models (e.g. MiniMax M3) disregard the static instruction and run
|
|
9
|
+
* 30+ tool calls with no user-visible progress. A reminder injected mid-turn,
|
|
10
|
+
* right after a tool result, is far more salient than static prompt text.
|
|
11
|
+
*
|
|
12
|
+
* The nudge is strictly best-effort:
|
|
13
|
+
* - It fires at most once per turn (no nagging).
|
|
14
|
+
* - It is advisory — the model may decline if it is about to finish or judges
|
|
15
|
+
* a card unnecessary.
|
|
16
|
+
* - A failed or ignored card never blocks the turn.
|
|
17
|
+
*
|
|
18
|
+
* It is scoped to weaker open models (Kimi, DeepSeek, MiniMax) that disregard
|
|
19
|
+
* the static progress-card instruction; capable models follow the prompt and
|
|
20
|
+
* never need the reminder. It also self-targets: a model that already showed a
|
|
21
|
+
* card this turn is never nudged.
|
|
22
|
+
*
|
|
23
|
+
* The turn state is derived from conversation history on every call (mirroring
|
|
24
|
+
* the tool-error and exploration-drift plugins) so the signal survives
|
|
25
|
+
* mid-turn compaction rewriting the array. The current turn is the trailing
|
|
26
|
+
* window bounded by the last genuine user message (a user row with no
|
|
27
|
+
* tool_result blocks). Within it we count tool-use rounds and detect whether a
|
|
28
|
+
* `ui_show` task_progress card was issued.
|
|
29
|
+
*
|
|
30
|
+
* Gating:
|
|
31
|
+
* - weaker open models only (Kimi, DeepSeek, MiniMax) — checked first on the
|
|
32
|
+
* hot path so capable-model turns short-circuit immediately.
|
|
33
|
+
* - mainAgent call site only — background turns (wake, title-gen, memory) and
|
|
34
|
+
* subagents have no live user watching, so a progress card is pointless.
|
|
35
|
+
* - the client must support dynamic UI — otherwise `ui_show` is blocked
|
|
36
|
+
* client-side and the nudge would only provoke a wasted, erroring call.
|
|
37
|
+
*
|
|
38
|
+
* The call-site, capability, and subagent gates are resolved lazily on the
|
|
39
|
+
* would-nudge path; the model gate is a cheap regex test run up front.
|
|
40
|
+
*
|
|
41
|
+
* Dedup uses a per-conversation high-water mark of the round count at the last
|
|
42
|
+
* nudge: a non-zero mark means "already nudged this turn", which also dedupes
|
|
43
|
+
* the parallel tool results of one batch (they observe identical history and
|
|
44
|
+
* compute the same round count). The mark resets when the round count drops
|
|
45
|
+
* below it (a new turn restarts counting low).
|
|
46
|
+
*/
|
|
47
|
+
|
|
48
|
+
import type { PluginHookFn, PostToolUseContext } from "@vellumai/plugin-api";
|
|
49
|
+
|
|
50
|
+
import type { ContentBlock, Message } from "../../../../providers/types.js";
|
|
51
|
+
|
|
52
|
+
/**
|
|
53
|
+
* Canonical nudge notice. Module-level constant so tests and wrapping plugins
|
|
54
|
+
* can match it without duplicating the string. Shown to the model as
|
|
55
|
+
* provider-only context, never to the user. Deliberately soft: coarse steps
|
|
56
|
+
* are fine and the model may skip it when wrapping up.
|
|
57
|
+
*/
|
|
58
|
+
export const TASK_PROGRESS_NUDGE_TEXT =
|
|
59
|
+
'<system_notice>You are several tool calls into this turn and have not shown the user a progress card. If you are doing multi-step work, call ui_show now with surface_type "card" and template "task_progress" (coarse steps are fine — a rough "Working on X" beats no signal) so the user can see what is happening, and keep it updated with ui_update as you go. Skip this if you are about to finish; never let it interrupt the actual work.</system_notice>';
|
|
60
|
+
|
|
61
|
+
/**
|
|
62
|
+
* Number of tool-use rounds in a turn, with no task_progress card shown, that
|
|
63
|
+
* triggers the nudge. Tuned conservatively: a turn of one or two rounds (the
|
|
64
|
+
* common simple case) is never touched; only a clearly multi-step turn is
|
|
65
|
+
* nudged, and only once. Lower from telemetry if cards still arrive too late.
|
|
66
|
+
*/
|
|
67
|
+
export const TASK_PROGRESS_NUDGE_ROUND_THRESHOLD = 3;
|
|
68
|
+
|
|
69
|
+
/**
|
|
70
|
+
* Weaker open models that disregard the static progress-card instruction and
|
|
71
|
+
* so get the mid-turn nudge: Kimi, DeepSeek, and MiniMax. Family-level matching
|
|
72
|
+
* spans provider naming conventions (OpenRouter `moonshotai/kimi-k2.6`,
|
|
73
|
+
* `deepseek/deepseek-chat`, `minimax/minimax-m3`; Fireworks
|
|
74
|
+
* `accounts/fireworks/models/minimax-m3`, `kimi-k2p6`). Extend as other models
|
|
75
|
+
* show the same gap. Capable models (Claude, GPT) follow the prompt and are
|
|
76
|
+
* intentionally excluded.
|
|
77
|
+
*/
|
|
78
|
+
const WEAK_MODEL_PATTERN = /kimi|deepseek|minimax/i;
|
|
79
|
+
|
|
80
|
+
/**
|
|
81
|
+
* Round count at the last nudge, per conversation. A non-zero entry means the
|
|
82
|
+
* turn has already been nudged; it resets when the round count drops below the
|
|
83
|
+
* mark (a new turn). Mirrors the exploration-drift high-water mark so parallel
|
|
84
|
+
* results of one batch dedupe to a single notice.
|
|
85
|
+
*/
|
|
86
|
+
const lastNudgedRoundsByConversation = new Map<string, number>();
|
|
87
|
+
|
|
88
|
+
/** Test-only: clear the per-conversation nudge high-water marks. */
|
|
89
|
+
export function resetTaskProgressNudgeStateForTests(): void {
|
|
90
|
+
lastNudgedRoundsByConversation.clear();
|
|
91
|
+
}
|
|
92
|
+
|
|
93
|
+
/**
|
|
94
|
+
* True when a `ui_show` tool input shows a `task_progress` card — accepting the
|
|
95
|
+
* template either at the top level or nested under `data`, mirroring the
|
|
96
|
+
* server-side normalization tolerance.
|
|
97
|
+
*/
|
|
98
|
+
function isTaskProgressShowInput(input: unknown): boolean {
|
|
99
|
+
if (input === null || typeof input !== "object") return false;
|
|
100
|
+
const record = input as Record<string, unknown>;
|
|
101
|
+
if (record.template === "task_progress") return true;
|
|
102
|
+
const data = record.data;
|
|
103
|
+
return (
|
|
104
|
+
data !== null &&
|
|
105
|
+
typeof data === "object" &&
|
|
106
|
+
(data as Record<string, unknown>).template === "task_progress"
|
|
107
|
+
);
|
|
108
|
+
}
|
|
109
|
+
|
|
110
|
+
/**
|
|
111
|
+
* Scan the trailing turn (walking back to the last genuine user message) for
|
|
112
|
+
* the number of tool-use rounds and whether a task_progress card was shown.
|
|
113
|
+
* The current round's assistant tool_use is already in history; its result is
|
|
114
|
+
* not, so the count includes the current round.
|
|
115
|
+
*/
|
|
116
|
+
function scanTurn(messages: ReadonlyArray<Message>): {
|
|
117
|
+
rounds: number;
|
|
118
|
+
taskProgressShown: boolean;
|
|
119
|
+
} {
|
|
120
|
+
let rounds = 0;
|
|
121
|
+
let taskProgressShown = false;
|
|
122
|
+
for (let i = messages.length - 1; i >= 0; i--) {
|
|
123
|
+
const message = messages[i];
|
|
124
|
+
if (message.role === "user") {
|
|
125
|
+
const carriesToolResult = message.content.some(
|
|
126
|
+
(block: ContentBlock) =>
|
|
127
|
+
block.type === "tool_result" ||
|
|
128
|
+
block.type === "web_search_tool_result",
|
|
129
|
+
);
|
|
130
|
+
if (!carriesToolResult) break; // genuine user prompt — turn boundary
|
|
131
|
+
continue;
|
|
132
|
+
}
|
|
133
|
+
if (message.role !== "assistant") continue;
|
|
134
|
+
let hasToolUse = false;
|
|
135
|
+
for (const block of message.content) {
|
|
136
|
+
if (block.type !== "tool_use") continue;
|
|
137
|
+
hasToolUse = true;
|
|
138
|
+
if (block.name === "ui_show" && isTaskProgressShowInput(block.input)) {
|
|
139
|
+
taskProgressShown = true;
|
|
140
|
+
}
|
|
141
|
+
}
|
|
142
|
+
if (hasToolUse) rounds++;
|
|
143
|
+
}
|
|
144
|
+
return { rounds, taskProgressShown };
|
|
145
|
+
}
|
|
146
|
+
|
|
147
|
+
const postToolUse: PluginHookFn<PostToolUseContext> = async (ctx) => {
|
|
148
|
+
if (!WEAK_MODEL_PATTERN.test(ctx.model)) return;
|
|
149
|
+
|
|
150
|
+
const { rounds, taskProgressShown } = scanTurn(ctx.messages);
|
|
151
|
+
|
|
152
|
+
let lastNudged = lastNudgedRoundsByConversation.get(ctx.conversationId) ?? 0;
|
|
153
|
+
if (rounds < lastNudged) {
|
|
154
|
+
// New turn (round count restarted low) — drop the stale mark.
|
|
155
|
+
lastNudgedRoundsByConversation.delete(ctx.conversationId);
|
|
156
|
+
lastNudged = 0;
|
|
157
|
+
}
|
|
158
|
+
|
|
159
|
+
// A card now exists this turn: clear any stale mark and never nudge.
|
|
160
|
+
if (taskProgressShown) {
|
|
161
|
+
if (lastNudged !== 0) {
|
|
162
|
+
lastNudgedRoundsByConversation.delete(ctx.conversationId);
|
|
163
|
+
}
|
|
164
|
+
return;
|
|
165
|
+
}
|
|
166
|
+
|
|
167
|
+
if (rounds < TASK_PROGRESS_NUDGE_ROUND_THRESHOLD) return;
|
|
168
|
+
if (lastNudged !== 0) return; // already nudged this turn
|
|
169
|
+
|
|
170
|
+
// Resolve gating lazily on the rare would-nudge path to keep the hot path
|
|
171
|
+
// free of the conversation-registry and subagent module graphs.
|
|
172
|
+
const { findConversation } =
|
|
173
|
+
await import("../../../../daemon/conversation-registry.js");
|
|
174
|
+
const conversation = findConversation(ctx.conversationId);
|
|
175
|
+
if (!conversation) return;
|
|
176
|
+
|
|
177
|
+
if (
|
|
178
|
+
conversation.currentCallSite &&
|
|
179
|
+
conversation.currentCallSite !== "mainAgent"
|
|
180
|
+
) {
|
|
181
|
+
return; // background call site — no live user watching
|
|
182
|
+
}
|
|
183
|
+
|
|
184
|
+
const capabilities =
|
|
185
|
+
conversation.currentTurnChannelCapabilities ??
|
|
186
|
+
conversation.channelCapabilities;
|
|
187
|
+
if (capabilities && capabilities.supportsDynamicUi === false) {
|
|
188
|
+
return; // client cannot render surfaces — a nudge would only error
|
|
189
|
+
}
|
|
190
|
+
|
|
191
|
+
const { getSubagentManager } = await import("../../../../subagent/index.js");
|
|
192
|
+
if (getSubagentManager().getParentInfo(ctx.conversationId) !== undefined) {
|
|
193
|
+
return; // subagents have no live user
|
|
194
|
+
}
|
|
195
|
+
|
|
196
|
+
lastNudgedRoundsByConversation.set(ctx.conversationId, rounds);
|
|
197
|
+
ctx.logger.info(
|
|
198
|
+
{ plugin: "task-progress-nudge", rounds },
|
|
199
|
+
"Multi-step turn with no task_progress card — nudging the model to show progress",
|
|
200
|
+
);
|
|
201
|
+
ctx.additionalContext = ctx.additionalContext
|
|
202
|
+
? `${ctx.additionalContext}\n${TASK_PROGRESS_NUDGE_TEXT}`
|
|
203
|
+
: TASK_PROGRESS_NUDGE_TEXT;
|
|
204
|
+
};
|
|
205
|
+
|
|
206
|
+
export default postToolUse;
|
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
{
|
|
2
|
+
"name": "default-task-progress-nudge",
|
|
3
|
+
"version": "1.0.0",
|
|
4
|
+
"description": "First-party default plugin contributing a post-tool-use hook that nudges the model to show a task_progress card when an interactive turn accumulates several tool-call rounds without one.",
|
|
5
|
+
"private": true,
|
|
6
|
+
"license": "MIT",
|
|
7
|
+
"type": "module",
|
|
8
|
+
"main": "./register.ts",
|
|
9
|
+
"engines": {
|
|
10
|
+
"node": ">=20.12.0"
|
|
11
|
+
},
|
|
12
|
+
"peerDependencies": {
|
|
13
|
+
"@vellumai/plugin-api": "^0.8.0"
|
|
14
|
+
}
|
|
15
|
+
}
|
|
@@ -6,7 +6,7 @@
|
|
|
6
6
|
* user-message injection that replaced it.
|
|
7
7
|
*/
|
|
8
8
|
|
|
9
|
-
import { copyFileSync, mkdirSync, readFileSync } from "node:fs";
|
|
9
|
+
import { copyFileSync, mkdirSync, readFileSync, writeFileSync } from "node:fs";
|
|
10
10
|
import { join } from "node:path";
|
|
11
11
|
import { beforeEach, describe, expect, mock, test } from "bun:test";
|
|
12
12
|
|
|
@@ -73,6 +73,105 @@ describe("buildSystemPrompt — tool routing guidance", () => {
|
|
|
73
73
|
});
|
|
74
74
|
});
|
|
75
75
|
|
|
76
|
+
describe("buildSystemPrompt — persona override", () => {
|
|
77
|
+
const DEFAULT_SENTINEL = "Sentinel: default persona body.";
|
|
78
|
+
const ALICE_SENTINEL = "Sentinel: alice persona body.";
|
|
79
|
+
const TELEGRAM_SENTINEL = "Sentinel: telegram channel persona body.";
|
|
80
|
+
|
|
81
|
+
beforeEach(() => {
|
|
82
|
+
mkdirSync(join(TEST_DIR, "users"), { recursive: true });
|
|
83
|
+
mkdirSync(join(TEST_DIR, "channels"), { recursive: true });
|
|
84
|
+
writeFileSync(join(TEST_DIR, "users", "default.md"), DEFAULT_SENTINEL);
|
|
85
|
+
writeFileSync(join(TEST_DIR, "users", "alice.md"), ALICE_SENTINEL);
|
|
86
|
+
writeFileSync(join(TEST_DIR, "channels", "telegram.md"), TELEGRAM_SENTINEL);
|
|
87
|
+
});
|
|
88
|
+
|
|
89
|
+
test("personaOverride renders the given user + channel persona sections", () => {
|
|
90
|
+
const result = buildSystemPrompt({
|
|
91
|
+
personaOverride: { userSlug: "alice", channelSlug: "telegram" },
|
|
92
|
+
});
|
|
93
|
+
|
|
94
|
+
expect(result).toContain(ALICE_SENTINEL);
|
|
95
|
+
expect(result).toContain(TELEGRAM_SENTINEL);
|
|
96
|
+
expect(result).not.toContain(DEFAULT_SENTINEL);
|
|
97
|
+
});
|
|
98
|
+
|
|
99
|
+
test("no override → trust-context-derived resolution (default persona, vellum channel)", () => {
|
|
100
|
+
// No trust context and no resolvable guardian contact in this test env,
|
|
101
|
+
// so the user persona falls back to users/default.md and the channel
|
|
102
|
+
// section resolves channels/vellum.md (absent → omitted).
|
|
103
|
+
const result = buildSystemPrompt({});
|
|
104
|
+
|
|
105
|
+
expect(result).toContain(DEFAULT_SENTINEL);
|
|
106
|
+
expect(result).not.toContain(ALICE_SENTINEL);
|
|
107
|
+
expect(result).not.toContain(TELEGRAM_SENTINEL);
|
|
108
|
+
});
|
|
109
|
+
|
|
110
|
+
test("partial override: userSlug alone leaves channel resolution untouched", () => {
|
|
111
|
+
const result = buildSystemPrompt({
|
|
112
|
+
personaOverride: { userSlug: "alice" },
|
|
113
|
+
});
|
|
114
|
+
|
|
115
|
+
expect(result).toContain(ALICE_SENTINEL);
|
|
116
|
+
expect(result).not.toContain(TELEGRAM_SENTINEL);
|
|
117
|
+
});
|
|
118
|
+
|
|
119
|
+
test("override userSlug with no matching file falls back to users/default.md", () => {
|
|
120
|
+
const result = buildSystemPrompt({
|
|
121
|
+
personaOverride: { userSlug: "missing-user" },
|
|
122
|
+
});
|
|
123
|
+
|
|
124
|
+
expect(result).toContain(DEFAULT_SENTINEL);
|
|
125
|
+
});
|
|
126
|
+
});
|
|
127
|
+
|
|
128
|
+
describe("buildSystemPrompt — hasNoClient pin", () => {
|
|
129
|
+
// Marker unique to the `{{^hasNoClient}}` branch of 05-access-preference:
|
|
130
|
+
// host fallbacks only render when a client is connected.
|
|
131
|
+
const WITH_CLIENT_MARKER = "`host_bash` with CLIs";
|
|
132
|
+
|
|
133
|
+
beforeEach(() => {
|
|
134
|
+
mkdirSync(TEST_DIR, { recursive: true });
|
|
135
|
+
});
|
|
136
|
+
|
|
137
|
+
test("hasNoClient renders divergent access-preference text", () => {
|
|
138
|
+
expect(buildSystemPrompt({ hasNoClient: false })).toContain(
|
|
139
|
+
WITH_CLIENT_MARKER,
|
|
140
|
+
);
|
|
141
|
+
expect(buildSystemPrompt({ hasNoClient: true })).not.toContain(
|
|
142
|
+
WITH_CLIENT_MARKER,
|
|
143
|
+
);
|
|
144
|
+
});
|
|
145
|
+
|
|
146
|
+
test("personaOverride.hasNoClient pins the flag over the conversation-derived option", () => {
|
|
147
|
+
// Fork-retrospective case: the fork is hydrated clientless
|
|
148
|
+
// (hasNoClient: true) but pins the source's live-turn value (false) so
|
|
149
|
+
// the prompt byte-matches the source's cached prefix.
|
|
150
|
+
expect(
|
|
151
|
+
buildSystemPrompt({
|
|
152
|
+
hasNoClient: true,
|
|
153
|
+
personaOverride: { hasNoClient: false },
|
|
154
|
+
}),
|
|
155
|
+
).toContain(WITH_CLIENT_MARKER);
|
|
156
|
+
// And the pin wins in the other direction too.
|
|
157
|
+
expect(
|
|
158
|
+
buildSystemPrompt({
|
|
159
|
+
hasNoClient: false,
|
|
160
|
+
personaOverride: { hasNoClient: true },
|
|
161
|
+
}),
|
|
162
|
+
).not.toContain(WITH_CLIENT_MARKER);
|
|
163
|
+
});
|
|
164
|
+
|
|
165
|
+
test("a personaOverride without the pin leaves the conversation-derived flag untouched", () => {
|
|
166
|
+
expect(
|
|
167
|
+
buildSystemPrompt({ hasNoClient: true, personaOverride: {} }),
|
|
168
|
+
).not.toContain(WITH_CLIENT_MARKER);
|
|
169
|
+
expect(
|
|
170
|
+
buildSystemPrompt({ hasNoClient: false, personaOverride: {} }),
|
|
171
|
+
).toContain(WITH_CLIENT_MARKER);
|
|
172
|
+
});
|
|
173
|
+
});
|
|
174
|
+
|
|
76
175
|
describe("maybeReseedBootstrap — content-automation template", () => {
|
|
77
176
|
const templatesDir = join(import.meta.dirname!, "..", "templates");
|
|
78
177
|
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* Tests for the task_progress hint in the 01-
|
|
2
|
+
* Tests for the task_progress hint in the 01-progress-surface workspace
|
|
3
3
|
* system prompt section.
|
|
4
4
|
*
|
|
5
5
|
* Verifies that the task_progress guidance renders unconditionally in the
|
|
@@ -54,11 +54,10 @@ mock.module("../../config/loader.js", () => ({
|
|
|
54
54
|
setNestedValue: () => {},
|
|
55
55
|
}));
|
|
56
56
|
|
|
57
|
-
const { buildSystemPrompt, ensurePromptFiles } =
|
|
58
|
-
"../system-prompt.js"
|
|
59
|
-
);
|
|
57
|
+
const { buildSystemPrompt, ensurePromptFiles } =
|
|
58
|
+
await import("../system-prompt.js");
|
|
60
59
|
|
|
61
|
-
describe("task_progress hint in
|
|
60
|
+
describe("task_progress hint in progress-surface section", () => {
|
|
62
61
|
beforeEach(() => {
|
|
63
62
|
ensurePromptFiles();
|
|
64
63
|
});
|
|
@@ -66,7 +65,7 @@ describe("task_progress hint in parallel-tool-calls section", () => {
|
|
|
66
65
|
test("buildSystemPrompt() includes task_progress guidance", () => {
|
|
67
66
|
const result = buildSystemPrompt();
|
|
68
67
|
expect(result).toContain("task_progress");
|
|
69
|
-
expect(result).toContain("
|
|
68
|
+
expect(result).toContain("Show Progress on Long Turns");
|
|
70
69
|
});
|
|
71
70
|
|
|
72
71
|
test("renders unconditionally — no options required", () => {
|
|
@@ -85,5 +84,4 @@ describe("task_progress hint in parallel-tool-calls section", () => {
|
|
|
85
84
|
expect(withoutClientFlag).toContain("task_progress");
|
|
86
85
|
expect(withExcludePrefix).toContain("task_progress");
|
|
87
86
|
});
|
|
88
|
-
|
|
89
87
|
});
|
|
@@ -82,6 +82,7 @@ export function normalizePriorAssistants(assistants: string[]): string[] {
|
|
|
82
82
|
|
|
83
83
|
export interface NormalizedOnboarding {
|
|
84
84
|
preferredName?: string;
|
|
85
|
+
occupation?: string;
|
|
85
86
|
commonWork: string[];
|
|
86
87
|
dailyTools: string[];
|
|
87
88
|
tone?: string;
|
|
@@ -123,6 +124,7 @@ export function normalizeOnboardingContext(
|
|
|
123
124
|
): NormalizedOnboarding {
|
|
124
125
|
return {
|
|
125
126
|
preferredName: ctx.userName?.trim() || undefined,
|
|
127
|
+
occupation: ctx.occupation?.trim() || undefined,
|
|
126
128
|
commonWork: normalizeTasks(ctx.tasks),
|
|
127
129
|
dailyTools: normalizeTools(ctx.tools),
|
|
128
130
|
tone: ctx.tone,
|
|
@@ -348,6 +348,9 @@ function buildOnboardingSection(normalized: NormalizedOnboarding): string {
|
|
|
348
348
|
if (normalized.preferredName) {
|
|
349
349
|
lines.push(`- **Preferred name:** ${normalized.preferredName}`);
|
|
350
350
|
}
|
|
351
|
+
if (normalized.occupation) {
|
|
352
|
+
lines.push(`- **Role:** ${normalized.occupation}`);
|
|
353
|
+
}
|
|
351
354
|
if (normalized.commonWork.length > 0) {
|
|
352
355
|
lines.push(`- **Common work:** ${normalized.commonWork.join("; ")}`);
|
|
353
356
|
}
|
|
@@ -304,6 +304,33 @@ export function applyBootstrapTemplate(
|
|
|
304
304
|
}
|
|
305
305
|
}
|
|
306
306
|
|
|
307
|
+
/**
|
|
308
|
+
* Explicit prompt-build override for builds that run outside the
|
|
309
|
+
* inbound-turn pipeline (agent wakes). Each field, when present, takes
|
|
310
|
+
* precedence over the corresponding derivation in {@link buildSystemPrompt}.
|
|
311
|
+
* Prompt-build selection only — trust class and approval semantics are
|
|
312
|
+
* unaffected.
|
|
313
|
+
*/
|
|
314
|
+
export interface SystemPromptPersonaOverride {
|
|
315
|
+
/** Renders `users/<slug>.md` as the user persona section. */
|
|
316
|
+
userSlug?: string;
|
|
317
|
+
/** Renders `channels/<slug>.md` as the channel persona section. */
|
|
318
|
+
channelSlug?: string;
|
|
319
|
+
/**
|
|
320
|
+
* Pins the `hasNoClient` flag for the prompt build, taking precedence over
|
|
321
|
+
* the top-level `BuildSystemPromptOptions.hasNoClient` (which mirrors the
|
|
322
|
+
* conversation's live client state). The `05-access-preference` section
|
|
323
|
+
* renders different text under the flag — early in the prompt, so a
|
|
324
|
+
* mismatch breaks byte-parity with a cached prefix even when persona and
|
|
325
|
+
* profile match. Used by fork-based memory retrospectives: the fork is
|
|
326
|
+
* hydrated clientless (`hasNoClient = true`) while the source's live turns
|
|
327
|
+
* ran under the source's own client state (`false` for interactive
|
|
328
|
+
* interfaces, `true` for channel-routed sources) — the pin carries that
|
|
329
|
+
* live-turn value.
|
|
330
|
+
*/
|
|
331
|
+
hasNoClient?: boolean;
|
|
332
|
+
}
|
|
333
|
+
|
|
307
334
|
export interface BuildSystemPromptOptions {
|
|
308
335
|
hasNoClient?: boolean;
|
|
309
336
|
excludeBootstrap?: boolean;
|
|
@@ -311,6 +338,15 @@ export interface BuildSystemPromptOptions {
|
|
|
311
338
|
trustContext?: TrustContext;
|
|
312
339
|
channelCapabilities?: ChannelCapabilities;
|
|
313
340
|
onboardingContext?: OnboardingContext;
|
|
341
|
+
/**
|
|
342
|
+
* Explicit persona/channel slugs, taking precedence over the
|
|
343
|
+
* trust-context-derived `userSlug` and capabilities-derived `channelSlug`.
|
|
344
|
+
* Used by fork-based memory retrospectives so the fork's prompt renders the
|
|
345
|
+
* SOURCE conversation's persona sections (review quality + byte-parity with
|
|
346
|
+
* the source's cached system-prompt prefix) even though the wake itself
|
|
347
|
+
* carries an internal guardian trust context with no requester identity.
|
|
348
|
+
*/
|
|
349
|
+
personaOverride?: SystemPromptPersonaOverride;
|
|
314
350
|
/**
|
|
315
351
|
* Conversation this prompt is being built for. Optional because several
|
|
316
352
|
* callers build a prompt outside a conversation (e.g. home greeting,
|
|
@@ -344,8 +380,20 @@ export function buildSystemPrompt(options?: BuildSystemPromptOptions): string {
|
|
|
344
380
|
// `users/<slug>.md → users/default.md` fallback lives in the
|
|
345
381
|
// section's `workspacePath` array. `channelSlug` is the channel
|
|
346
382
|
// identifier from `channelCapabilities`, defaulting to "vellum".
|
|
347
|
-
|
|
348
|
-
const
|
|
383
|
+
// An explicit `personaOverride` slug wins over either derivation.
|
|
384
|
+
const userSlug =
|
|
385
|
+
options?.personaOverride?.userSlug ??
|
|
386
|
+
resolveUserSlug(options?.trustContext) ??
|
|
387
|
+
"default";
|
|
388
|
+
const channelSlug =
|
|
389
|
+
options?.personaOverride?.channelSlug ??
|
|
390
|
+
options?.channelCapabilities?.channel ??
|
|
391
|
+
"vellum";
|
|
392
|
+
// The override's `hasNoClient` pin wins over the conversation-derived
|
|
393
|
+
// top-level option (see the interface doc); placed after the `...options`
|
|
394
|
+
// spread below so it overrides the spread-in value.
|
|
395
|
+
const hasNoClient =
|
|
396
|
+
options?.personaOverride?.hasNoClient ?? options?.hasNoClient;
|
|
349
397
|
|
|
350
398
|
// Section render context. Workspace section frontmatter `enabled:`
|
|
351
399
|
// predicates, `{{key}}` / `{{#flag}}...{{/flag}}` body interpolation,
|
|
@@ -358,6 +406,7 @@ export function buildSystemPrompt(options?: BuildSystemPromptOptions): string {
|
|
|
358
406
|
// no explicit normalization needed; `...options` is enough.
|
|
359
407
|
const ctx = {
|
|
360
408
|
...options,
|
|
409
|
+
hasNoClient,
|
|
361
410
|
isContainerized: getIsContainerized(),
|
|
362
411
|
workspaceDir: getWorkspaceDir(),
|
|
363
412
|
userSlug,
|
|
@@ -35,6 +35,8 @@ Surface the outcome as a clickable component, strongest first. The component is
|
|
|
35
35
|
|
|
36
36
|
DONE: you cannot exit Propose without emitting at least one `ui_show` offer surface (card / choice). A proposal in prose is not a proposal.
|
|
37
37
|
|
|
38
|
+
Thin signal still ships a proposal. Weak findings mean you make the smallest defensible offer and name the confidence in a clause ("low confidence, but —") — they do not mean you skip the move. Never resolve Propose into a critique of why the signal's too thin: "not enough here to suggest anything" is a stall, not a move. The least-wrong job for someone who looks like them beats an honest shrug, every time.
|
|
39
|
+
|
|
38
40
|
**Run.** Do it. Real tools, real data. The user watches something happen.
|
|
39
41
|
|
|
40
42
|
DONE: a real tool ran against real data and the user can see the result.
|
|
@@ -69,7 +71,7 @@ These are enforcement rules, not advice.
|
|
|
69
71
|
|
|
70
72
|
**Self-check before final emit.** If this turn contains an offer or a follow-up, render it via `ui_show`, not prose.
|
|
71
73
|
|
|
72
|
-
**Long turns show progress.** Any post-submit / post-skill-load turn must render a `task_progress` card within ~5s, or fall back to streaming text. Bind "long turn" → "task_progress emitted": a long-running turn that produces neither a progress card nor streaming text didn't satisfy this move.
|
|
74
|
+
**Long turns show progress.** Any post-submit / post-skill-load turn must render a `task_progress` card within ~5s, or fall back to streaming text. Bind "long turn" → "task_progress emitted": a long-running turn that produces neither a progress card nor streaming text didn't satisfy this move. Progress is the card, not a play-by-play. Tool selection, routing, retries, and failures ("search blocked, trying the browser…", "that captcha'd, falling back to —") belong in thinking, never in visible prose: the card shows progress, the result surface shows outcome, and the mechanics in between stay out of the user's view.
|
|
73
75
|
|
|
74
76
|
**Action Trust-Guarantee.** Sibling to the OAuth Trust-Guarantee. Before a bulk write / delete / destructive op, render a `ui_show` preview — a table surface showing total count, breakdown, sample rows, and the categories to confirm. The user commits or refines on that surface; only then do you execute. Single-item actions use the natural draft instead. Threshold for the preview gate: bulk _and_ low recoverability. One of the two alone doesn't trip it.
|
|
75
77
|
|
|
@@ -269,9 +269,14 @@ Batch independent tool calls into the same response. An extra LLM round trip cos
|
|
|
269
269
|
Before emitting a single tool call, ask whether your next turn would be another tool call that doesn't consume this one's output — if so, they belong together. Serialized tool calls without a real data dependency are a bug.
|
|
270
270
|
|
|
271
271
|
For non-trivial independent workstreams — research, coding, multi-step investigations — delegate to subagents (load the \`subagent\` skill) and spawn them early and in parallel; an unnecessary subagent is cheaper than serialized work.
|
|
272
|
-
|
|
273
|
-
**Before your first tool call**, check: does this turn involve a web search, file operations, multi-step work, or anything that will take more than a few seconds? If yes, call ui_show with surface_type "card" and template "task_progress" first, then update steps via ui_update as work progresses. No exceptions.
|
|
274
272
|
</use_parallel_tool_calls>
|
|
273
|
+
`,
|
|
274
|
+
},
|
|
275
|
+
{
|
|
276
|
+
id: "01-progress-surface",
|
|
277
|
+
body: `## Show Progress on Long Turns
|
|
278
|
+
|
|
279
|
+
When a turn will take more than a few seconds — web searches, multi-step file work, research — show the user a progress card early: call ui_show with surface_type "card" and template "task_progress", then flip each step pending → in_progress → completed via ui_update as you go. Coarse steps are fine; a rough "Working on X" beats no signal at all. You can add or revise steps as the work takes shape — you are not committed to your first list. Skip the card when the turn is quick or you are already wrapping up; never let it get in the way of doing the actual work.
|
|
275
280
|
`,
|
|
276
281
|
},
|
|
277
282
|
{
|
|
@@ -332,7 +337,7 @@ Priority: (1) sandbox \`bash\` - install tools yourself, only fall back to host
|
|
|
332
337
|
id: "06-credential-security",
|
|
333
338
|
body: `## Credential Security
|
|
334
339
|
|
|
335
|
-
Never ask users to share secrets (API keys, tokens, passwords, webhook secrets) in chat — secret messages may be blocked at ingress.
|
|
340
|
+
Never ask users to share secrets (API keys, tokens, passwords, webhook secrets) in chat — secret messages may be blocked at ingress. Run \`assistant credentials prompt\` (via the bash tool) instead; it collects secrets through a secure UI that never exposes the value in the conversation. This command blocks until the user submits the secret, so set the bash tool's \`timeout_seconds\` to at least 330 — the default (120s) cuts the prompt off before the user can respond. Non-secret values (Client IDs, Account SIDs, usernames) may be collected conversationally.
|
|
336
341
|
`,
|
|
337
342
|
},
|
|
338
343
|
{
|
|
@@ -141,13 +141,16 @@ export class CallSiteRoutingProvider implements Provider {
|
|
|
141
141
|
if (!callSite) return this.defaultProvider;
|
|
142
142
|
|
|
143
143
|
const overrideProfile = options?.config?.overrideProfile;
|
|
144
|
-
// Forward the per-conversation mix seed so
|
|
145
|
-
// same
|
|
146
|
-
//
|
|
144
|
+
// Forward `forceOverrideProfile` and the per-conversation mix seed so
|
|
145
|
+
// transport selection resolves the same profile/arm as wire-param
|
|
146
|
+
// normalization in `retry.ts` — otherwise a forced profile (or a mix)
|
|
147
|
+
// spanning providers could route the transport differently than the
|
|
147
148
|
// request params.
|
|
149
|
+
const forceOverrideProfile = options?.config?.forceOverrideProfile;
|
|
148
150
|
const selectionSeed = options?.config?.selectionSeed;
|
|
149
151
|
const resolved = resolveCallSiteConfig(callSite, getConfig().llm, {
|
|
150
152
|
overrideProfile,
|
|
153
|
+
forceOverrideProfile,
|
|
151
154
|
selectionSeed,
|
|
152
155
|
});
|
|
153
156
|
|
|
@@ -35,6 +35,9 @@ export class FireworksProvider extends OpenAIChatCompletionsProvider {
|
|
|
35
35
|
// {@link resolveMaxReasoningEffort}.
|
|
36
36
|
maxReasoningEffort: "high",
|
|
37
37
|
assistantReasoningField: "reasoning_content",
|
|
38
|
+
// minimax-m3's function-call serialization collapses object-typed tool
|
|
39
|
+
// args to `{}` on the wire; present them as JSON strings and decode back.
|
|
40
|
+
coerceObjectArgsToJsonString: /minimax/i.test(model),
|
|
38
41
|
});
|
|
39
42
|
}
|
|
40
43
|
|