@vellumai/assistant 0.11.5 → 0.11.6-staging.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/AGENTS.md +5 -1
- package/node_modules/@vellumai/ces-client/node_modules/@vellumai/service-contracts/src/__tests__/ingress.test.ts +118 -0
- package/node_modules/@vellumai/ces-client/node_modules/@vellumai/service-contracts/src/ingress.ts +103 -0
- package/node_modules/@vellumai/gateway-client/node_modules/@vellumai/service-contracts/src/__tests__/ingress.test.ts +118 -0
- package/node_modules/@vellumai/gateway-client/node_modules/@vellumai/service-contracts/src/ingress.ts +103 -0
- package/node_modules/@vellumai/gateway-client/src/gateway-ipc-contracts.ts +70 -0
- package/node_modules/@vellumai/gateway-client/src/inbound-contract.ts +16 -1
- package/node_modules/@vellumai/gateway-client/src/index.ts +6 -2
- package/node_modules/@vellumai/gateway-client/src/outbound-contract.ts +121 -61
- package/node_modules/@vellumai/service-contracts/src/__tests__/ingress.test.ts +118 -0
- package/node_modules/@vellumai/service-contracts/src/ingress.ts +103 -0
- package/openapi.yaml +421 -15
- package/package.json +1 -1
- package/scripts/sync-web-search-catalog.ts +6 -0
- package/src/__tests__/app-pin-store.test.ts +149 -0
- package/src/__tests__/channel-availability-routes.test.ts +23 -1
- package/src/__tests__/channel-readiness-discord.test.ts +231 -0
- package/src/__tests__/channel-readiness-service.test.ts +126 -0
- package/src/__tests__/channel-readiness-slack-remote.test.ts +141 -0
- package/src/__tests__/channel-reply-delivery.test.ts +4 -4
- package/src/__tests__/client-os-metadata-persistence.test.ts +23 -10
- package/src/__tests__/conversation-delete-watch-timeline.test.ts +231 -0
- package/src/__tests__/conversation-error.test.ts +17 -0
- package/src/__tests__/conversation-seed-composer.test.ts +8 -0
- package/src/__tests__/conversation-slash-commands.test.ts +8 -0
- package/src/__tests__/disk-pressure-policy.test.ts +6 -0
- package/src/__tests__/gemini-provider.test.ts +138 -0
- package/src/__tests__/history-repair.test.ts +105 -3
- package/src/__tests__/identity-routes.test.ts +1 -0
- package/src/__tests__/llm-catalog-parity.test.ts +45 -0
- package/src/__tests__/migration-import-from-path.test.ts +349 -0
- package/src/__tests__/notification-telegram-adapter.test.ts +102 -0
- package/src/__tests__/oauth-commands-routes.test.ts +89 -0
- package/src/__tests__/oauth-provider-profiles.test.ts +7 -6
- package/src/__tests__/openai-provider.test.ts +18 -0
- package/src/__tests__/openai-responses-provider.test.ts +18 -0
- package/src/__tests__/platform-callback-registration.test.ts +184 -0
- package/src/__tests__/plugin-api-store-credential.test.ts +71 -3
- package/src/__tests__/pricing.test.ts +2 -2
- package/src/__tests__/public-ingress-urls.test.ts +36 -0
- package/src/__tests__/resolve-trust-class.test.ts +0 -48
- package/src/__tests__/sanitize-config-for-transfer.test.ts +28 -0
- package/src/__tests__/secret-routes-platform-proxy.test.ts +49 -0
- package/src/__tests__/settings-routes.test.ts +85 -3
- package/src/__tests__/web-search-catalog-parity.test.ts +8 -0
- package/src/agent/history-repair/history-repair.ts +45 -14
- package/src/agent/loop.ts +4 -1
- package/src/api/constants/profile-config-validation.ts +60 -0
- package/src/api/events/tool-result.ts +6 -1
- package/src/api/events/watch-retro-completed.ts +52 -0
- package/src/api/index.ts +11 -0
- package/src/apps/app-pin-reconciler.ts +92 -0
- package/src/apps/app-pin-store.ts +125 -0
- package/src/channels/gateway-channel-socket-health.ts +32 -0
- package/src/channels/gateway-discord-admission.ts +32 -0
- package/src/channels/types.ts +20 -0
- package/src/cli/commands/__tests__/conversations-slack.test.ts +1 -1
- package/src/cli/commands/__tests__/inference-profiles.test.ts +16 -4
- package/src/cli/commands/__tests__/inference-providers.test.ts +67 -2
- package/src/cli/commands/channels/__tests__/channels.test.ts +85 -0
- package/src/cli/commands/channels/index.ts +45 -31
- package/src/cli/commands/inference-profiles.ts +56 -3
- package/src/cli/commands/inference-providers.ts +28 -2
- package/src/cli/commands/oauth/index.help.ts +7 -1
- package/src/cli/commands/oauth/request.test.ts +290 -0
- package/src/cli/commands/oauth/request.ts +57 -41
- package/src/cli/lib/bundled-marketplace.json +14 -1
- package/src/cli/lib/open-browser.test.ts +67 -0
- package/src/cli/lib/open-browser.ts +24 -5
- package/src/config/__tests__/profile-materialization.test.ts +26 -0
- package/src/config/bundled-skills/phone-calls/references/TROUBLESHOOTING.md +6 -0
- package/src/config/bundled-skills/schedule/SKILL.md +1 -1
- package/src/config/feature-flag-registry.json +17 -1
- package/src/config/profile-materialization.ts +29 -0
- package/src/config/sanitize-for-transfer.ts +16 -0
- package/src/config/schemas/llm.ts +7 -0
- package/src/config/schemas/services.ts +6 -0
- package/src/context/outbound-sanitize.ts +6 -0
- package/src/daemon/__tests__/lifecycle-watch-timeline-sweep.test.ts +98 -0
- package/src/daemon/conversation-error.ts +24 -2
- package/src/daemon/conversation-slash.ts +6 -15
- package/src/daemon/daemon-control.ts +1 -0
- package/src/daemon/disk-pressure-policy.ts +7 -1
- package/src/daemon/handlers/__tests__/config-ingress-tunnel-records.test.ts +208 -0
- package/src/daemon/handlers/config-ingress.ts +115 -5
- package/src/daemon/lifecycle.ts +26 -0
- package/src/daemon/message-types/web-activity.ts +3 -2
- package/src/daemon/trust-context.ts +0 -37
- package/src/inbound/__tests__/tunnel-probe.test.ts +448 -0
- package/src/inbound/platform-callback-registration.ts +28 -2
- package/src/inbound/public-ingress-urls.ts +12 -0
- package/src/inbound/tunnel-probe.ts +261 -0
- package/src/live-voice/__tests__/live-voice-connection.test.ts +25 -0
- package/src/live-voice/__tests__/live-voice-flux-turn-end.test.ts +8 -2
- package/src/live-voice/__tests__/live-voice-session-manager.test.ts +212 -10
- package/src/live-voice/__tests__/live-voice-session-telemetry.test.ts +5 -2
- package/src/live-voice/live-voice-connection.ts +46 -8
- package/src/live-voice/live-voice-manager.ts +25 -0
- package/src/live-voice/live-voice-session-manager.ts +318 -2
- package/src/live-voice/live-voice-session.ts +52 -2
- package/src/messaging/providers/__tests__/transport-dispatch.test.ts +126 -68
- package/src/messaging/providers/channel-transport.ts +64 -47
- package/src/messaging/providers/discord/send.test.ts +46 -1
- package/src/messaging/providers/discord/send.ts +51 -0
- package/src/messaging/providers/discord/transport.ts +26 -3
- package/src/messaging/providers/index.ts +22 -47
- package/src/messaging/providers/slack/send.test.ts +83 -26
- package/src/messaging/providers/slack/send.ts +120 -51
- package/src/messaging/providers/slack/stream-tasks.test.ts +26 -0
- package/src/messaging/providers/slack/stream-tasks.ts +39 -0
- package/src/messaging/providers/slack/transport.ts +24 -22
- package/src/messaging/providers/telegram-bot/send.test.ts +109 -12
- package/src/messaging/providers/telegram-bot/send.ts +43 -0
- package/src/messaging/providers/telegram-bot/transport.ts +25 -8
- package/src/notifications/__tests__/assistant-reply-producer.test.ts +30 -7
- package/src/notifications/adapters/telegram.ts +48 -1
- package/src/notifications/assistant-reply-producer.ts +7 -7
- package/src/notifications/conversation-seed-composer.ts +7 -2
- package/src/oauth/byo-connection.test.ts +63 -0
- package/src/oauth/byo-connection.ts +16 -15
- package/src/oauth/connection.test.ts +111 -0
- package/src/oauth/connection.ts +142 -1
- package/src/oauth/platform-connection.test.ts +34 -0
- package/src/oauth/platform-connection.ts +28 -5
- package/src/oauth/seed-providers.ts +15 -1
- package/src/permissions/types.ts +3 -1
- package/src/persistence/conversation-crud.ts +46 -0
- package/src/persistence/conversation-types.ts +11 -9
- package/src/persistence/db-async-query.ts +2 -1
- package/src/persistence/db-maintenance.ts +15 -0
- package/src/persistence/embeddings/qdrant-manager.ts +1 -0
- package/src/persistence/migrations/367-create-watch-timeline-entries.ts +46 -0
- package/src/persistence/migrations/368-watch-timeline-screenshot-blob.ts +33 -0
- package/src/persistence/migrations/369-create-app-pins.ts +37 -0
- package/src/persistence/migrations/__tests__/367-create-watch-timeline-entries.test.ts +98 -0
- package/src/persistence/migrations/__tests__/368-watch-timeline-screenshot-blob.test.ts +98 -0
- package/src/persistence/schema/index.ts +1 -0
- package/src/persistence/schema/infrastructure.ts +17 -0
- package/src/persistence/schema/watch.ts +29 -0
- package/src/persistence/steps.ts +6 -0
- package/src/plugins/mtime-cache.ts +11 -0
- package/src/providers/__tests__/retry-network-error.test.ts +84 -0
- package/src/providers/connection-resolution.ts +23 -1
- package/src/providers/content-blocks.ts +9 -0
- package/src/providers/fetch-provider-catalog.ts +19 -0
- package/src/providers/gemini/client.ts +13 -5
- package/src/providers/inference/__tests__/endpoint-probe.test.ts +92 -0
- package/src/providers/inference/__tests__/profile-config-validation.test.ts +39 -0
- package/src/providers/inference/__tests__/profile-probe-classify.test.ts +66 -0
- package/src/providers/inference/adapter-factory.ts +0 -9
- package/src/providers/inference/credential-rotation.ts +61 -0
- package/src/providers/inference/endpoint-probe.ts +115 -0
- package/src/providers/inference/profile-probe.ts +256 -0
- package/src/providers/model-catalog.ts +170 -125
- package/src/providers/openai/__tests__/api-error-normalization.test.ts +17 -1
- package/src/providers/openai/__tests__/chat-completions-provider-reasoning.test.ts +42 -60
- package/src/providers/openai/__tests__/connection-error-wrap.test.ts +44 -0
- package/src/providers/openai/__tests__/orphan-tool-result-guard.test.ts +34 -2
- package/src/providers/openai/api-error-normalization.ts +16 -2
- package/src/providers/openai/chat-completions-provider.ts +75 -29
- package/src/providers/openai/responses-provider.ts +5 -2
- package/src/providers/openrouter/client.ts +0 -1
- package/src/providers/provider-send-message.ts +11 -0
- package/src/providers/retry.ts +6 -0
- package/src/providers/search-provider-catalog.ts +20 -0
- package/src/providers/vercel-ai-gateway/client.ts +0 -1
- package/src/runtime/AGENTS.md +1 -0
- package/src/runtime/__tests__/desktop-presence.test.ts +27 -4
- package/src/runtime/__tests__/host-observe.test.ts +302 -0
- package/src/runtime/channel-readiness-service.ts +214 -14
- package/src/runtime/channel-readiness-types.ts +49 -2
- package/src/runtime/channel-reply-delivery.ts +2 -2
- package/src/runtime/desktop-presence.ts +24 -21
- package/src/runtime/host-observe.ts +246 -0
- package/src/runtime/http-server.ts +181 -1
- package/src/runtime/migrations/__tests__/staged-import-path.test.ts +104 -0
- package/src/runtime/migrations/staged-import-path.ts +116 -0
- package/src/runtime/routes/__tests__/app-pin-routes.test.ts +383 -0
- package/src/runtime/routes/__tests__/conversation-query-routes.test.ts +80 -0
- package/src/runtime/routes/__tests__/inference-profiles-routes.test.ts +118 -0
- package/src/runtime/routes/__tests__/inference-provider-connection-routes.test.ts +20 -0
- package/src/runtime/routes/__tests__/ingress-status-routes.test.ts +508 -0
- package/src/runtime/routes/__tests__/plugins-routes.test.ts +35 -56
- package/src/runtime/routes/__tests__/watch-routes-guardian-cache.test.ts +139 -0
- package/src/runtime/routes/__tests__/watch-routes.test.ts +598 -0
- package/src/runtime/routes/app-management-routes.ts +140 -29
- package/src/runtime/routes/channel-availability-routes.ts +1 -0
- package/src/runtime/routes/channel-readiness-routes.ts +14 -2
- package/src/runtime/routes/conversation-query-routes.ts +10 -0
- package/src/runtime/routes/guardian-approval-interception.ts +24 -33
- package/src/runtime/routes/host-cu-routes.ts +18 -0
- package/src/runtime/routes/identity-routes.ts +2 -0
- package/src/runtime/routes/inbound-message-handler.ts +10 -7
- package/src/runtime/routes/inbound-stages/background-dispatch.test.ts +166 -308
- package/src/runtime/routes/inbound-stages/background-dispatch.ts +158 -335
- package/src/runtime/routes/index.ts +2 -0
- package/src/runtime/routes/inference-profiles-routes.ts +232 -31
- package/src/runtime/routes/inference-provider-connection-routes.ts +24 -4
- package/src/runtime/routes/ingress-status-routes.ts +180 -0
- package/src/runtime/routes/live-voice-routes.test.ts +40 -1
- package/src/runtime/routes/live-voice-routes.ts +34 -0
- package/src/runtime/routes/migration-routes.ts +218 -10
- package/src/runtime/routes/oauth-commands-routes.ts +23 -16
- package/src/runtime/routes/plugins-routes.ts +12 -28
- package/src/runtime/routes/question-routes.ts +6 -0
- package/src/runtime/routes/secret-routes.ts +7 -27
- package/src/runtime/routes/settings-routes.ts +9 -6
- package/src/runtime/routes/watch-routes.ts +807 -0
- package/src/runtime/slack-reply-session.test.ts +230 -121
- package/src/runtime/slack-reply-session.ts +113 -81
- package/src/runtime/{slack-task-progress.test.ts → task-progress.test.ts} +1 -28
- package/src/runtime/{slack-task-progress.ts → task-progress.ts} +30 -51
- package/src/security/__tests__/untrusted-content.test.ts +42 -0
- package/src/security/untrusted-content.ts +28 -9
- package/src/telemetry/__tests__/live-voice-funnel.test.ts +108 -0
- package/src/telemetry/live-voice-funnel.ts +75 -8
- package/src/tools/credentials/store.ts +18 -6
- package/src/tools/network/__tests__/firecrawl-compat.test.ts +77 -0
- package/src/tools/network/__tests__/web-fetch-fastcrw.test.ts +169 -0
- package/src/tools/network/__tests__/web-search.test.ts +97 -2
- package/src/tools/network/firecrawl-compat.ts +90 -0
- package/src/tools/network/web-fetch.ts +142 -62
- package/src/tools/network/web-search.ts +141 -55
- package/src/tools/types.ts +2 -1
- package/src/util/oauth-request-body.test.ts +74 -0
- package/src/util/oauth-request-body.ts +60 -0
- package/src/util/worker-process.ts +1 -0
- package/src/watch/__tests__/watch-retro.test.ts +665 -0
- package/src/watch/__tests__/watch-session-manager.test.ts +566 -0
- package/src/watch/__tests__/watch-timeline.test.ts +670 -0
- package/src/watch/watch-retro.ts +480 -0
- package/src/watch/watch-session-manager.ts +575 -0
- package/src/watch/watch-timeline.ts +848 -0
|
@@ -0,0 +1,66 @@
|
|
|
1
|
+
import { describe, expect, test } from "bun:test";
|
|
2
|
+
|
|
3
|
+
import { ProviderError } from "../../../util/errors.js";
|
|
4
|
+
import { classifyProbeFailure } from "../profile-probe.js";
|
|
5
|
+
|
|
6
|
+
describe("classifyProbeFailure", () => {
|
|
7
|
+
test("blames the provider connection for a rejected credential", () => {
|
|
8
|
+
const result = classifyProbeFailure(
|
|
9
|
+
new ProviderError("401 invalid api key", "openai-compatible", 401, {
|
|
10
|
+
reason: "invalid_credentials",
|
|
11
|
+
}),
|
|
12
|
+
);
|
|
13
|
+
expect(result.blame).toBe("provider");
|
|
14
|
+
expect(result.reason).toBe("invalid_credentials");
|
|
15
|
+
});
|
|
16
|
+
|
|
17
|
+
test("blames the profile for an unknown model", () => {
|
|
18
|
+
const result = classifyProbeFailure(
|
|
19
|
+
new ProviderError("model does not exist", "openai-compatible", 404, {
|
|
20
|
+
reason: "model_not_found",
|
|
21
|
+
}),
|
|
22
|
+
);
|
|
23
|
+
expect(result.blame).toBe("profile");
|
|
24
|
+
});
|
|
25
|
+
|
|
26
|
+
test("blames the profile for an impossible token budget", () => {
|
|
27
|
+
const result = classifyProbeFailure(
|
|
28
|
+
new ProviderError(
|
|
29
|
+
"ContextWindowExceededError: maximum context length is 300000 tokens",
|
|
30
|
+
"openai-compatible",
|
|
31
|
+
400,
|
|
32
|
+
{ reason: "context_overflow" },
|
|
33
|
+
),
|
|
34
|
+
);
|
|
35
|
+
expect(result.blame).toBe("profile");
|
|
36
|
+
expect(result.detail).toContain("300000");
|
|
37
|
+
});
|
|
38
|
+
|
|
39
|
+
test("marks rate limits as transient", () => {
|
|
40
|
+
expect(
|
|
41
|
+
classifyProbeFailure(
|
|
42
|
+
new ProviderError("429", "openai", 429, { reason: "rate_limited" }),
|
|
43
|
+
).blame,
|
|
44
|
+
).toBe("transient");
|
|
45
|
+
});
|
|
46
|
+
|
|
47
|
+
test("blames the provider for transport failures", () => {
|
|
48
|
+
// The adapters stamp reason network_error on SDK connection errors
|
|
49
|
+
// (normalizeOpenAIAPIError), so the probe classifies by reason alone.
|
|
50
|
+
const result = classifyProbeFailure(
|
|
51
|
+
new ProviderError(
|
|
52
|
+
"OpenAI-compatible API error (unknown status): Connection error.",
|
|
53
|
+
"openai-compatible",
|
|
54
|
+
undefined,
|
|
55
|
+
{ reason: "network_error" },
|
|
56
|
+
),
|
|
57
|
+
);
|
|
58
|
+
expect(result.blame).toBe("provider");
|
|
59
|
+
expect(result.reason).toBe("network_error");
|
|
60
|
+
});
|
|
61
|
+
|
|
62
|
+
test("falls back to unknown for unclassified errors", () => {
|
|
63
|
+
expect(classifyProbeFailure(new Error("boom")).blame).toBe("unknown");
|
|
64
|
+
expect(classifyProbeFailure(new Error("boom")).detail).toBe("boom");
|
|
65
|
+
});
|
|
66
|
+
});
|
|
@@ -136,10 +136,6 @@ const ADAPTER_FACTORIES: Record<string, AdapterFactory> = {
|
|
|
136
136
|
// Replay thinking as `reasoning_content` so DeepSeek-compatible
|
|
137
137
|
// thinking-mode upstreams accept follow-up requests that include tools.
|
|
138
138
|
assistantReasoningField: "reasoning_content",
|
|
139
|
-
// Generic OpenAI-compat proxies may front a strict backend (vLLM,
|
|
140
|
-
// DeepSeek, Portkey). Backfill empty assistant turns so the request
|
|
141
|
-
// satisfies `content or tool_calls must be set`.
|
|
142
|
-
backfillEmptyAssistantContent: true,
|
|
143
139
|
...(baseURL ? { baseURL } : {}),
|
|
144
140
|
}),
|
|
145
141
|
// Keyless openai-compatible endpoints (e.g. LM Studio) ignore the key; the
|
|
@@ -152,11 +148,6 @@ const ADAPTER_FACTORIES: Record<string, AdapterFactory> = {
|
|
|
152
148
|
// Replay thinking as `reasoning_content` so DeepSeek-compatible
|
|
153
149
|
// thinking-mode endpoints accept follow-up requests that include tools.
|
|
154
150
|
assistantReasoningField: "reasoning_content",
|
|
155
|
-
// Custom endpoints (Portkey, vLLM, LM Studio, DeepSeek-compat) often
|
|
156
|
-
// reject `{ role: "assistant", content: null }` after a Stop mid-stream
|
|
157
|
-
// or a reasoning-only turn. Same guard as OpenRouter and Vercel AI
|
|
158
|
-
// Gateway.
|
|
159
|
-
backfillEmptyAssistantContent: true,
|
|
160
151
|
// Custom OpenAI-compatible endpoints may front strict reasoning
|
|
161
152
|
// models (DeepSeek thinking) that 400 on any explicit tool_choice.
|
|
162
153
|
omitToolChoiceWhenReasoning: true,
|
|
@@ -0,0 +1,61 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Refreshes provider state after a credential rotation.
|
|
3
|
+
*/
|
|
4
|
+
|
|
5
|
+
import { getConfig, invalidateConfigCache } from "../../config/loader.js";
|
|
6
|
+
import { evictConversationsForReload } from "../../daemon/conversation-store.js";
|
|
7
|
+
import { clearEmbeddingBackendCache } from "../../persistence/embeddings/embedding-backend.js";
|
|
8
|
+
import { credentialKey } from "../../security/credential-key.js";
|
|
9
|
+
import { getLogger } from "../../util/logger.js";
|
|
10
|
+
import { initializeProviders } from "../registry.js";
|
|
11
|
+
import { findConnectionsUsingCredential } from "./credential-usage.js";
|
|
12
|
+
|
|
13
|
+
const log = getLogger("credential-rotation");
|
|
14
|
+
|
|
15
|
+
export async function refreshProvidersAfterSecretChange(): Promise<void> {
|
|
16
|
+
clearEmbeddingBackendCache();
|
|
17
|
+
invalidateConfigCache();
|
|
18
|
+
await initializeProviders(getConfig());
|
|
19
|
+
|
|
20
|
+
// Provider instances are captured when conversations are created, so a key
|
|
21
|
+
// change must evict or mark them stale before the next turn. Best-effort:
|
|
22
|
+
// the credential write has already succeeded, so a disposal failure must not
|
|
23
|
+
// surface as a 500 that makes clients think the secret change failed.
|
|
24
|
+
try {
|
|
25
|
+
evictConversationsForReload();
|
|
26
|
+
} catch (err) {
|
|
27
|
+
log.warn(
|
|
28
|
+
{ error: err instanceof Error ? err.message : String(err) },
|
|
29
|
+
"Error evicting conversations after credential change (non-fatal)",
|
|
30
|
+
);
|
|
31
|
+
}
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
export async function refreshProvidersForRotatedCredential(
|
|
35
|
+
service: string,
|
|
36
|
+
field: string,
|
|
37
|
+
): Promise<void> {
|
|
38
|
+
const key = credentialKey(service, field);
|
|
39
|
+
|
|
40
|
+
// A connection resolves its auth through this credential account, and
|
|
41
|
+
// the resolved adapter bakes the key it read into its headers, so the
|
|
42
|
+
// rotation only reaches dispatch once the cached adapter is dropped.
|
|
43
|
+
// The credential write has already succeeded: an enumeration failure
|
|
44
|
+
// refreshes rather than surfacing as a 500.
|
|
45
|
+
try {
|
|
46
|
+
const connections = findConnectionsUsingCredential(key);
|
|
47
|
+
if (connections.length > 0) {
|
|
48
|
+
await refreshProvidersAfterSecretChange();
|
|
49
|
+
}
|
|
50
|
+
} catch (err) {
|
|
51
|
+
log.warn(
|
|
52
|
+
{
|
|
53
|
+
service,
|
|
54
|
+
field,
|
|
55
|
+
error: err instanceof Error ? err.message : String(err),
|
|
56
|
+
},
|
|
57
|
+
"Failed to inspect provider connections after credential update; refreshing providers anyway",
|
|
58
|
+
);
|
|
59
|
+
await refreshProvidersAfterSecretChange();
|
|
60
|
+
}
|
|
61
|
+
}
|
|
@@ -0,0 +1,115 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Save-time probe for connections with a custom base URL.
|
|
3
|
+
*
|
|
4
|
+
* A wrong base path (e.g. NVIDIA without `/v1`) otherwise stays invisible
|
|
5
|
+
* until the first chat turn fails with an opaque provider 404. The probe
|
|
6
|
+
* fires one minimal chat-completions request against the stored endpoint and
|
|
7
|
+
* reports the outcome as a non-blocking hint. The save always succeeds.
|
|
8
|
+
*/
|
|
9
|
+
|
|
10
|
+
import { z } from "zod";
|
|
11
|
+
|
|
12
|
+
import { getLogger } from "../../util/logger.js";
|
|
13
|
+
import type { Auth, ConnectionModel } from "./auth.js";
|
|
14
|
+
import { resolveAuth } from "./resolve-auth.js";
|
|
15
|
+
|
|
16
|
+
const log = getLogger("inference-endpoint-probe");
|
|
17
|
+
|
|
18
|
+
/**
|
|
19
|
+
* Bound on every save-path probe (this endpoint probe and the profile probe)
|
|
20
|
+
* so a dead host can't hang a save. Matches the API-key validation timeout.
|
|
21
|
+
*/
|
|
22
|
+
export const PROBE_TIMEOUT_MS = 10_000;
|
|
23
|
+
|
|
24
|
+
export const EndpointCheckSchema = z
|
|
25
|
+
.object({
|
|
26
|
+
ok: z.boolean(),
|
|
27
|
+
status: z.number().int().optional(),
|
|
28
|
+
resolved_url: z.string(),
|
|
29
|
+
error_class: z.enum(["http_error", "timeout", "network"]).optional(),
|
|
30
|
+
/** Human-readable warning, present when `ok` is false. */
|
|
31
|
+
hint: z.string().optional(),
|
|
32
|
+
})
|
|
33
|
+
.meta({ id: "EndpointCheck" });
|
|
34
|
+
|
|
35
|
+
export type EndpointCheck = z.infer<typeof EndpointCheckSchema>;
|
|
36
|
+
|
|
37
|
+
function hintForStatus(status: number): string {
|
|
38
|
+
if (status === 404) {
|
|
39
|
+
return "The endpoint returned 404 for a test request: check the base path for this provider (e.g. NVIDIA needs /v1, OpenRouter needs /api/v1). Some providers gate requests behind auth, so this may be a false alarm.";
|
|
40
|
+
}
|
|
41
|
+
if (status === 401 || status === 403) {
|
|
42
|
+
return `The endpoint rejected the credential for a test request (HTTP ${status}). Check the API key.`;
|
|
43
|
+
}
|
|
44
|
+
return `The endpoint returned HTTP ${status} for a test request.`;
|
|
45
|
+
}
|
|
46
|
+
|
|
47
|
+
/**
|
|
48
|
+
* Probe a connection's custom endpoint with a minimal chat-completions
|
|
49
|
+
* request (`max_tokens: 1`). Returns `null` when there is nothing to probe:
|
|
50
|
+
* no custom base URL, no model id to send, or auth that cannot be resolved
|
|
51
|
+
* (a missing credential surfaces through its own error path).
|
|
52
|
+
*/
|
|
53
|
+
export async function testInferenceConnection(
|
|
54
|
+
connection: {
|
|
55
|
+
provider: string;
|
|
56
|
+
auth: Auth;
|
|
57
|
+
baseUrl?: string | null;
|
|
58
|
+
models?: ConnectionModel[] | null;
|
|
59
|
+
},
|
|
60
|
+
fetchImpl: typeof fetch = fetch,
|
|
61
|
+
): Promise<EndpointCheck | null> {
|
|
62
|
+
const model = connection.models?.[0]?.id;
|
|
63
|
+
if (!connection.baseUrl || !model) {
|
|
64
|
+
return null;
|
|
65
|
+
}
|
|
66
|
+
|
|
67
|
+
const resolved = await resolveAuth(connection.auth, connection.provider, {
|
|
68
|
+
baseUrl: connection.baseUrl,
|
|
69
|
+
});
|
|
70
|
+
if (!resolved.ok || resolved.resolved.kind === "runtime_proxy") {
|
|
71
|
+
return null;
|
|
72
|
+
}
|
|
73
|
+
const authHeaders =
|
|
74
|
+
resolved.resolved.kind === "header" ? resolved.resolved.headers : {};
|
|
75
|
+
|
|
76
|
+
const url = `${connection.baseUrl.replace(/\/+$/, "")}/chat/completions`;
|
|
77
|
+
try {
|
|
78
|
+
const res = await fetchImpl(url, {
|
|
79
|
+
method: "POST",
|
|
80
|
+
headers: { ...authHeaders, "Content-Type": "application/json" },
|
|
81
|
+
body: JSON.stringify({
|
|
82
|
+
model,
|
|
83
|
+
messages: [{ role: "user", content: "ping" }],
|
|
84
|
+
max_tokens: 1,
|
|
85
|
+
stream: false,
|
|
86
|
+
}),
|
|
87
|
+
signal: AbortSignal.timeout(PROBE_TIMEOUT_MS),
|
|
88
|
+
});
|
|
89
|
+
void res.body?.cancel();
|
|
90
|
+
if (res.ok) {
|
|
91
|
+
return { ok: true, status: res.status, resolved_url: url };
|
|
92
|
+
}
|
|
93
|
+
return {
|
|
94
|
+
ok: false,
|
|
95
|
+
status: res.status,
|
|
96
|
+
resolved_url: url,
|
|
97
|
+
error_class: "http_error",
|
|
98
|
+
hint: hintForStatus(res.status),
|
|
99
|
+
};
|
|
100
|
+
} catch (err) {
|
|
101
|
+
const isTimeout = err instanceof Error && err.name === "TimeoutError";
|
|
102
|
+
log.info(
|
|
103
|
+
{ url, error: err instanceof Error ? err.message : String(err) },
|
|
104
|
+
"Endpoint probe failed to reach the endpoint",
|
|
105
|
+
);
|
|
106
|
+
return {
|
|
107
|
+
ok: false,
|
|
108
|
+
resolved_url: url,
|
|
109
|
+
error_class: isTimeout ? "timeout" : "network",
|
|
110
|
+
hint: isTimeout
|
|
111
|
+
? `The endpoint did not respond within ${PROBE_TIMEOUT_MS / 1000}s for a test request.`
|
|
112
|
+
: `Could not reach the endpoint for a test request: ${err instanceof Error ? err.message : String(err)}`,
|
|
113
|
+
};
|
|
114
|
+
}
|
|
115
|
+
}
|
|
@@ -0,0 +1,256 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Live save-time probe for a user-authored inference profile: one minimal
|
|
3
|
+
* request dispatched through the profile's own resolved model and connection,
|
|
4
|
+
* so a wrong model name, a rejected key, or an impossible token budget
|
|
5
|
+
* surfaces when the profile is saved instead of on the first chat message.
|
|
6
|
+
*
|
|
7
|
+
* The verdict is advisory and classified by which object the user should fix
|
|
8
|
+
* (this profile vs its provider connection), using the semantic
|
|
9
|
+
* `ProviderErrorReason` the provider adapters stamp at their throw sites.
|
|
10
|
+
* Managed and routing-identity profiles are never probed: their pins are
|
|
11
|
+
* hand-validated, and a probe there would spend managed credits on every
|
|
12
|
+
* save.
|
|
13
|
+
*/
|
|
14
|
+
|
|
15
|
+
import type { ResolutionFallbackReason } from "../../config/llm-resolver.js";
|
|
16
|
+
import {
|
|
17
|
+
resolveCallSiteConfig,
|
|
18
|
+
selectWinningProfile,
|
|
19
|
+
} from "../../config/llm-resolver.js";
|
|
20
|
+
import { getConfigReadOnly } from "../../config/loader.js";
|
|
21
|
+
import { getDb } from "../../persistence/db-connection.js";
|
|
22
|
+
import { ProviderError, type ProviderErrorReason } from "../../util/errors.js";
|
|
23
|
+
import { getLogger } from "../../util/logger.js";
|
|
24
|
+
import { dispatchProviderResolvable } from "../connection-resolution.js";
|
|
25
|
+
import {
|
|
26
|
+
createTimeout,
|
|
27
|
+
getConfiguredProvider,
|
|
28
|
+
userMessage,
|
|
29
|
+
} from "../provider-send-message.js";
|
|
30
|
+
import { ROUTING_IDENTITY_PROVIDERS } from "./auth.js";
|
|
31
|
+
import { getConnection, MANAGED_CONNECTION_NAMES } from "./connections.js";
|
|
32
|
+
import { PROBE_TIMEOUT_MS } from "./endpoint-probe.js";
|
|
33
|
+
|
|
34
|
+
const log = getLogger("inference-profile-probe");
|
|
35
|
+
|
|
36
|
+
/** Which object the user should fix when the probe fails. */
|
|
37
|
+
export type ProfileCheckBlame =
|
|
38
|
+
| "profile"
|
|
39
|
+
| "provider"
|
|
40
|
+
| "transient"
|
|
41
|
+
| "unknown";
|
|
42
|
+
|
|
43
|
+
export interface ProfileCheck {
|
|
44
|
+
ok: boolean;
|
|
45
|
+
blame?: ProfileCheckBlame;
|
|
46
|
+
/** Semantic reason (ProviderErrorReason, or "resolution"/"not_configured"). */
|
|
47
|
+
reason?: string;
|
|
48
|
+
/** Upstream or resolver error text, verbatim. */
|
|
49
|
+
detail?: string;
|
|
50
|
+
/** The connection to inspect when `blame` is "provider". */
|
|
51
|
+
connection?: string;
|
|
52
|
+
/** English summary for non-localized surfaces (the CLI). */
|
|
53
|
+
message?: string;
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
const PROVIDER_BLAME_REASONS: ReadonlySet<ProviderErrorReason> = new Set([
|
|
57
|
+
"invalid_credentials",
|
|
58
|
+
"insufficient_credits",
|
|
59
|
+
"daily_limit_reached",
|
|
60
|
+
"network_error",
|
|
61
|
+
"server_error",
|
|
62
|
+
]);
|
|
63
|
+
|
|
64
|
+
const TRANSIENT_BLAME_REASONS: ReadonlySet<ProviderErrorReason> = new Set([
|
|
65
|
+
"rate_limited",
|
|
66
|
+
"overloaded",
|
|
67
|
+
]);
|
|
68
|
+
|
|
69
|
+
/**
|
|
70
|
+
* Map a failed probe dispatch onto the object the user should fix. Reasons
|
|
71
|
+
* the adapters stamp about the request itself (model unknown or restricted,
|
|
72
|
+
* parameter and token-budget rejections) blame the profile; credential,
|
|
73
|
+
* billing, and reachability reasons blame the provider connection.
|
|
74
|
+
*/
|
|
75
|
+
export function classifyProbeFailure(err: unknown): {
|
|
76
|
+
blame: ProfileCheckBlame;
|
|
77
|
+
reason?: string;
|
|
78
|
+
detail: string;
|
|
79
|
+
} {
|
|
80
|
+
const detail = err instanceof Error ? err.message : String(err);
|
|
81
|
+
if (!(err instanceof ProviderError) || err.reason === undefined) {
|
|
82
|
+
return { blame: "unknown", detail };
|
|
83
|
+
}
|
|
84
|
+
if (PROVIDER_BLAME_REASONS.has(err.reason)) {
|
|
85
|
+
return { blame: "provider", reason: err.reason, detail };
|
|
86
|
+
}
|
|
87
|
+
if (TRANSIENT_BLAME_REASONS.has(err.reason)) {
|
|
88
|
+
return { blame: "transient", reason: err.reason, detail };
|
|
89
|
+
}
|
|
90
|
+
if (err.reason === "unknown") {
|
|
91
|
+
return { blame: "unknown", reason: err.reason, detail };
|
|
92
|
+
}
|
|
93
|
+
return { blame: "profile", reason: err.reason, detail };
|
|
94
|
+
}
|
|
95
|
+
|
|
96
|
+
function checkMessage(check: {
|
|
97
|
+
blame: ProfileCheckBlame;
|
|
98
|
+
detail: string;
|
|
99
|
+
connection?: string;
|
|
100
|
+
}): string {
|
|
101
|
+
switch (check.blame) {
|
|
102
|
+
case "profile":
|
|
103
|
+
return `A test request through this profile failed: ${check.detail} Check the profile's model and parameters.`;
|
|
104
|
+
case "provider":
|
|
105
|
+
return (
|
|
106
|
+
`A test request through this profile failed: ${check.detail} ` +
|
|
107
|
+
(check.connection
|
|
108
|
+
? `Check the provider connection "${check.connection}".`
|
|
109
|
+
: `Check the profile's provider connection.`)
|
|
110
|
+
);
|
|
111
|
+
case "transient":
|
|
112
|
+
return `A test request through this profile could not complete: ${check.detail} This may be temporary.`;
|
|
113
|
+
default:
|
|
114
|
+
return `A test request through this profile failed: ${check.detail}`;
|
|
115
|
+
}
|
|
116
|
+
}
|
|
117
|
+
|
|
118
|
+
/**
|
|
119
|
+
* Probe a saved profile with one minimal request. Returns `null` when there
|
|
120
|
+
* is no verdict to give: the profile does not exist or is disabled, it rides
|
|
121
|
+
* a managed or routing-identity route (probing would bill managed credits),
|
|
122
|
+
* or the probe timed out without an answer.
|
|
123
|
+
*/
|
|
124
|
+
export async function probeInferenceProfile(
|
|
125
|
+
name: string,
|
|
126
|
+
): Promise<ProfileCheck | null> {
|
|
127
|
+
const { llm } = getConfigReadOnly();
|
|
128
|
+
const entry = llm.profiles?.[name] as Record<string, unknown> | undefined;
|
|
129
|
+
if (!entry || entry.status === "disabled" || entry.source === "managed") {
|
|
130
|
+
return null;
|
|
131
|
+
}
|
|
132
|
+
|
|
133
|
+
const selectionSeed = crypto.randomUUID();
|
|
134
|
+
|
|
135
|
+
// If the resolver would fall back past this profile, the probe must not run:
|
|
136
|
+
// it would silently exercise (and bill) a different profile. The fallback
|
|
137
|
+
// reason is itself the verdict.
|
|
138
|
+
let fallbackReason: ResolutionFallbackReason | undefined;
|
|
139
|
+
const winner = selectWinningProfile("inference", llm, {
|
|
140
|
+
overrideProfile: name,
|
|
141
|
+
selectionSeed,
|
|
142
|
+
// The same resolvability predicate dispatch applies: without it this
|
|
143
|
+
// pre-check can report the requested profile as winner while the actual
|
|
144
|
+
// dispatch below falls back to (and bills) a different profile.
|
|
145
|
+
isResolvableProvider: dispatchProviderResolvable,
|
|
146
|
+
onResolutionFallback: (info) => {
|
|
147
|
+
if (info.requested === name) {
|
|
148
|
+
fallbackReason = info.reason;
|
|
149
|
+
}
|
|
150
|
+
},
|
|
151
|
+
});
|
|
152
|
+
if (winner.profileName !== name) {
|
|
153
|
+
const detail = `The profile is ${fallbackReason ?? "unusable"} and requests would silently run on ${winner.profileName ?? "the default"} instead.`;
|
|
154
|
+
return {
|
|
155
|
+
ok: false,
|
|
156
|
+
blame: "profile",
|
|
157
|
+
reason: "resolution",
|
|
158
|
+
detail,
|
|
159
|
+
message: checkMessage({ blame: "profile", detail }),
|
|
160
|
+
};
|
|
161
|
+
}
|
|
162
|
+
|
|
163
|
+
// Billing guards run on the fully resolved call-site config — the same
|
|
164
|
+
// composition dispatch consumes (mix arm expanded, call-site overrides
|
|
165
|
+
// applied) — so neither a mix nor an `llm.callSites.inference` provider
|
|
166
|
+
// tweak can route the probe onto a managed or platform-billed path the
|
|
167
|
+
// stored entry hid. Same seed as the dispatch below, so the judged arm is
|
|
168
|
+
// the dispatched arm.
|
|
169
|
+
const resolved = resolveCallSiteConfig("inference", llm, {
|
|
170
|
+
overrideProfile: name,
|
|
171
|
+
selectionSeed,
|
|
172
|
+
});
|
|
173
|
+
const provider = resolved.provider ?? "";
|
|
174
|
+
if (ROUTING_IDENTITY_PROVIDERS.has(provider)) {
|
|
175
|
+
return null;
|
|
176
|
+
}
|
|
177
|
+
const connection = resolved.provider_connection ?? (provider || undefined);
|
|
178
|
+
// Canonical managed names always dispatch platform-billed: routing ignores
|
|
179
|
+
// a user-owned row that merely claims such a name
|
|
180
|
+
// (tryResolveProviderForConnectionName), so the name gates the probe
|
|
181
|
+
// regardless of the row's stored auth.
|
|
182
|
+
if (connection && MANAGED_CONNECTION_NAMES.has(connection)) {
|
|
183
|
+
return null;
|
|
184
|
+
}
|
|
185
|
+
// A legacy profile can declare a concrete provider while staying bound to
|
|
186
|
+
// a platform-billed connection row; the row's auth is the billing fact, so
|
|
187
|
+
// it gates the probe too.
|
|
188
|
+
const boundRow = connection ? getConnection(getDb(), connection) : null;
|
|
189
|
+
if (boundRow?.auth.type === "platform") {
|
|
190
|
+
return null;
|
|
191
|
+
}
|
|
192
|
+
|
|
193
|
+
const timeout = createTimeout(PROBE_TIMEOUT_MS);
|
|
194
|
+
try {
|
|
195
|
+
const providerInstance = await getConfiguredProvider("inference", {
|
|
196
|
+
overrideProfile: name,
|
|
197
|
+
selectionSeed,
|
|
198
|
+
});
|
|
199
|
+
if (!providerInstance) {
|
|
200
|
+
const detail = "No dispatchable provider is configured for this profile.";
|
|
201
|
+
return {
|
|
202
|
+
ok: false,
|
|
203
|
+
blame: "provider",
|
|
204
|
+
reason: "not_configured",
|
|
205
|
+
detail,
|
|
206
|
+
connection,
|
|
207
|
+
message: checkMessage({ blame: "provider", detail, connection }),
|
|
208
|
+
};
|
|
209
|
+
}
|
|
210
|
+
// Final billing gate on the route dispatch ACTUALLY resolved: a profile
|
|
211
|
+
// without a binding gets its row auto-selected inside
|
|
212
|
+
// getConfiguredProvider, so only the stamped attribution names the row
|
|
213
|
+
// this request would sign with.
|
|
214
|
+
const route = providerInstance.routeAttribution;
|
|
215
|
+
if (route?.isManagedRoute) {
|
|
216
|
+
return null;
|
|
217
|
+
}
|
|
218
|
+
const routedRow = route?.connectionName
|
|
219
|
+
? getConnection(getDb(), route.connectionName)
|
|
220
|
+
: null;
|
|
221
|
+
if (routedRow?.auth.type === "platform") {
|
|
222
|
+
return null;
|
|
223
|
+
}
|
|
224
|
+
await providerInstance.sendMessage(
|
|
225
|
+
[userMessage("Reply with only the word OK.")],
|
|
226
|
+
{
|
|
227
|
+
signal: timeout.signal,
|
|
228
|
+
config: {
|
|
229
|
+
callSite: "inference",
|
|
230
|
+
overrideProfile: name,
|
|
231
|
+
selectionSeed,
|
|
232
|
+
},
|
|
233
|
+
},
|
|
234
|
+
);
|
|
235
|
+
return { ok: true };
|
|
236
|
+
} catch (err) {
|
|
237
|
+
if (timeout.signal.aborted) {
|
|
238
|
+
// No verdict: a slow model and a dead endpoint look identical here,
|
|
239
|
+
// and the connection-level probe already covers unreachable hosts.
|
|
240
|
+
log.info({ profile: name }, "Profile probe timed out without a verdict");
|
|
241
|
+
return null;
|
|
242
|
+
}
|
|
243
|
+
const classified = classifyProbeFailure(err);
|
|
244
|
+
return {
|
|
245
|
+
ok: false,
|
|
246
|
+
...classified,
|
|
247
|
+
...(classified.blame === "provider" ? { connection } : {}),
|
|
248
|
+
message: checkMessage({
|
|
249
|
+
...classified,
|
|
250
|
+
...(classified.blame === "provider" ? { connection } : {}),
|
|
251
|
+
}),
|
|
252
|
+
};
|
|
253
|
+
} finally {
|
|
254
|
+
timeout.cleanup();
|
|
255
|
+
}
|
|
256
|
+
}
|