@vellumai/assistant 0.11.5 → 0.11.6-staging.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (233) hide show
  1. package/AGENTS.md +5 -1
  2. package/node_modules/@vellumai/ces-client/node_modules/@vellumai/service-contracts/src/__tests__/ingress.test.ts +118 -0
  3. package/node_modules/@vellumai/ces-client/node_modules/@vellumai/service-contracts/src/ingress.ts +103 -0
  4. package/node_modules/@vellumai/gateway-client/node_modules/@vellumai/service-contracts/src/__tests__/ingress.test.ts +118 -0
  5. package/node_modules/@vellumai/gateway-client/node_modules/@vellumai/service-contracts/src/ingress.ts +103 -0
  6. package/node_modules/@vellumai/gateway-client/src/gateway-ipc-contracts.ts +70 -0
  7. package/node_modules/@vellumai/gateway-client/src/inbound-contract.ts +16 -1
  8. package/node_modules/@vellumai/gateway-client/src/index.ts +6 -2
  9. package/node_modules/@vellumai/gateway-client/src/outbound-contract.ts +121 -61
  10. package/node_modules/@vellumai/service-contracts/src/__tests__/ingress.test.ts +118 -0
  11. package/node_modules/@vellumai/service-contracts/src/ingress.ts +103 -0
  12. package/openapi.yaml +421 -15
  13. package/package.json +1 -1
  14. package/scripts/sync-web-search-catalog.ts +6 -0
  15. package/src/__tests__/app-pin-store.test.ts +149 -0
  16. package/src/__tests__/channel-availability-routes.test.ts +23 -1
  17. package/src/__tests__/channel-readiness-discord.test.ts +231 -0
  18. package/src/__tests__/channel-readiness-service.test.ts +126 -0
  19. package/src/__tests__/channel-readiness-slack-remote.test.ts +141 -0
  20. package/src/__tests__/channel-reply-delivery.test.ts +4 -4
  21. package/src/__tests__/client-os-metadata-persistence.test.ts +23 -10
  22. package/src/__tests__/conversation-delete-watch-timeline.test.ts +231 -0
  23. package/src/__tests__/conversation-error.test.ts +17 -0
  24. package/src/__tests__/conversation-seed-composer.test.ts +8 -0
  25. package/src/__tests__/conversation-slash-commands.test.ts +8 -0
  26. package/src/__tests__/disk-pressure-policy.test.ts +6 -0
  27. package/src/__tests__/gemini-provider.test.ts +138 -0
  28. package/src/__tests__/history-repair.test.ts +105 -3
  29. package/src/__tests__/identity-routes.test.ts +1 -0
  30. package/src/__tests__/llm-catalog-parity.test.ts +45 -0
  31. package/src/__tests__/migration-import-from-path.test.ts +349 -0
  32. package/src/__tests__/notification-telegram-adapter.test.ts +102 -0
  33. package/src/__tests__/oauth-commands-routes.test.ts +89 -0
  34. package/src/__tests__/oauth-provider-profiles.test.ts +7 -6
  35. package/src/__tests__/openai-provider.test.ts +18 -0
  36. package/src/__tests__/openai-responses-provider.test.ts +18 -0
  37. package/src/__tests__/platform-callback-registration.test.ts +184 -0
  38. package/src/__tests__/plugin-api-store-credential.test.ts +71 -3
  39. package/src/__tests__/pricing.test.ts +2 -2
  40. package/src/__tests__/public-ingress-urls.test.ts +36 -0
  41. package/src/__tests__/resolve-trust-class.test.ts +0 -48
  42. package/src/__tests__/sanitize-config-for-transfer.test.ts +28 -0
  43. package/src/__tests__/secret-routes-platform-proxy.test.ts +49 -0
  44. package/src/__tests__/settings-routes.test.ts +85 -3
  45. package/src/__tests__/web-search-catalog-parity.test.ts +8 -0
  46. package/src/agent/history-repair/history-repair.ts +45 -14
  47. package/src/agent/loop.ts +4 -1
  48. package/src/api/constants/profile-config-validation.ts +60 -0
  49. package/src/api/events/tool-result.ts +6 -1
  50. package/src/api/events/watch-retro-completed.ts +52 -0
  51. package/src/api/index.ts +11 -0
  52. package/src/apps/app-pin-reconciler.ts +92 -0
  53. package/src/apps/app-pin-store.ts +125 -0
  54. package/src/channels/gateway-channel-socket-health.ts +32 -0
  55. package/src/channels/gateway-discord-admission.ts +32 -0
  56. package/src/channels/types.ts +20 -0
  57. package/src/cli/commands/__tests__/conversations-slack.test.ts +1 -1
  58. package/src/cli/commands/__tests__/inference-profiles.test.ts +16 -4
  59. package/src/cli/commands/__tests__/inference-providers.test.ts +67 -2
  60. package/src/cli/commands/channels/__tests__/channels.test.ts +85 -0
  61. package/src/cli/commands/channels/index.ts +45 -31
  62. package/src/cli/commands/inference-profiles.ts +56 -3
  63. package/src/cli/commands/inference-providers.ts +28 -2
  64. package/src/cli/commands/oauth/index.help.ts +7 -1
  65. package/src/cli/commands/oauth/request.test.ts +290 -0
  66. package/src/cli/commands/oauth/request.ts +57 -41
  67. package/src/cli/lib/bundled-marketplace.json +14 -1
  68. package/src/cli/lib/open-browser.test.ts +67 -0
  69. package/src/cli/lib/open-browser.ts +24 -5
  70. package/src/config/__tests__/profile-materialization.test.ts +26 -0
  71. package/src/config/bundled-skills/phone-calls/references/TROUBLESHOOTING.md +6 -0
  72. package/src/config/bundled-skills/schedule/SKILL.md +1 -1
  73. package/src/config/feature-flag-registry.json +17 -1
  74. package/src/config/profile-materialization.ts +29 -0
  75. package/src/config/sanitize-for-transfer.ts +16 -0
  76. package/src/config/schemas/llm.ts +7 -0
  77. package/src/config/schemas/services.ts +6 -0
  78. package/src/context/outbound-sanitize.ts +6 -0
  79. package/src/daemon/__tests__/lifecycle-watch-timeline-sweep.test.ts +98 -0
  80. package/src/daemon/conversation-error.ts +24 -2
  81. package/src/daemon/conversation-slash.ts +6 -15
  82. package/src/daemon/daemon-control.ts +1 -0
  83. package/src/daemon/disk-pressure-policy.ts +7 -1
  84. package/src/daemon/handlers/__tests__/config-ingress-tunnel-records.test.ts +208 -0
  85. package/src/daemon/handlers/config-ingress.ts +115 -5
  86. package/src/daemon/lifecycle.ts +26 -0
  87. package/src/daemon/message-types/web-activity.ts +3 -2
  88. package/src/daemon/trust-context.ts +0 -37
  89. package/src/inbound/__tests__/tunnel-probe.test.ts +448 -0
  90. package/src/inbound/platform-callback-registration.ts +28 -2
  91. package/src/inbound/public-ingress-urls.ts +12 -0
  92. package/src/inbound/tunnel-probe.ts +261 -0
  93. package/src/live-voice/__tests__/live-voice-connection.test.ts +25 -0
  94. package/src/live-voice/__tests__/live-voice-flux-turn-end.test.ts +8 -2
  95. package/src/live-voice/__tests__/live-voice-session-manager.test.ts +212 -10
  96. package/src/live-voice/__tests__/live-voice-session-telemetry.test.ts +5 -2
  97. package/src/live-voice/live-voice-connection.ts +46 -8
  98. package/src/live-voice/live-voice-manager.ts +25 -0
  99. package/src/live-voice/live-voice-session-manager.ts +318 -2
  100. package/src/live-voice/live-voice-session.ts +52 -2
  101. package/src/messaging/providers/__tests__/transport-dispatch.test.ts +126 -68
  102. package/src/messaging/providers/channel-transport.ts +64 -47
  103. package/src/messaging/providers/discord/send.test.ts +46 -1
  104. package/src/messaging/providers/discord/send.ts +51 -0
  105. package/src/messaging/providers/discord/transport.ts +26 -3
  106. package/src/messaging/providers/index.ts +22 -47
  107. package/src/messaging/providers/slack/send.test.ts +83 -26
  108. package/src/messaging/providers/slack/send.ts +120 -51
  109. package/src/messaging/providers/slack/stream-tasks.test.ts +26 -0
  110. package/src/messaging/providers/slack/stream-tasks.ts +39 -0
  111. package/src/messaging/providers/slack/transport.ts +24 -22
  112. package/src/messaging/providers/telegram-bot/send.test.ts +109 -12
  113. package/src/messaging/providers/telegram-bot/send.ts +43 -0
  114. package/src/messaging/providers/telegram-bot/transport.ts +25 -8
  115. package/src/notifications/__tests__/assistant-reply-producer.test.ts +30 -7
  116. package/src/notifications/adapters/telegram.ts +48 -1
  117. package/src/notifications/assistant-reply-producer.ts +7 -7
  118. package/src/notifications/conversation-seed-composer.ts +7 -2
  119. package/src/oauth/byo-connection.test.ts +63 -0
  120. package/src/oauth/byo-connection.ts +16 -15
  121. package/src/oauth/connection.test.ts +111 -0
  122. package/src/oauth/connection.ts +142 -1
  123. package/src/oauth/platform-connection.test.ts +34 -0
  124. package/src/oauth/platform-connection.ts +28 -5
  125. package/src/oauth/seed-providers.ts +15 -1
  126. package/src/permissions/types.ts +3 -1
  127. package/src/persistence/conversation-crud.ts +46 -0
  128. package/src/persistence/conversation-types.ts +11 -9
  129. package/src/persistence/db-async-query.ts +2 -1
  130. package/src/persistence/db-maintenance.ts +15 -0
  131. package/src/persistence/embeddings/qdrant-manager.ts +1 -0
  132. package/src/persistence/migrations/367-create-watch-timeline-entries.ts +46 -0
  133. package/src/persistence/migrations/368-watch-timeline-screenshot-blob.ts +33 -0
  134. package/src/persistence/migrations/369-create-app-pins.ts +37 -0
  135. package/src/persistence/migrations/__tests__/367-create-watch-timeline-entries.test.ts +98 -0
  136. package/src/persistence/migrations/__tests__/368-watch-timeline-screenshot-blob.test.ts +98 -0
  137. package/src/persistence/schema/index.ts +1 -0
  138. package/src/persistence/schema/infrastructure.ts +17 -0
  139. package/src/persistence/schema/watch.ts +29 -0
  140. package/src/persistence/steps.ts +6 -0
  141. package/src/plugins/mtime-cache.ts +11 -0
  142. package/src/providers/__tests__/retry-network-error.test.ts +84 -0
  143. package/src/providers/connection-resolution.ts +23 -1
  144. package/src/providers/content-blocks.ts +9 -0
  145. package/src/providers/fetch-provider-catalog.ts +19 -0
  146. package/src/providers/gemini/client.ts +13 -5
  147. package/src/providers/inference/__tests__/endpoint-probe.test.ts +92 -0
  148. package/src/providers/inference/__tests__/profile-config-validation.test.ts +39 -0
  149. package/src/providers/inference/__tests__/profile-probe-classify.test.ts +66 -0
  150. package/src/providers/inference/adapter-factory.ts +0 -9
  151. package/src/providers/inference/credential-rotation.ts +61 -0
  152. package/src/providers/inference/endpoint-probe.ts +115 -0
  153. package/src/providers/inference/profile-probe.ts +256 -0
  154. package/src/providers/model-catalog.ts +170 -125
  155. package/src/providers/openai/__tests__/api-error-normalization.test.ts +17 -1
  156. package/src/providers/openai/__tests__/chat-completions-provider-reasoning.test.ts +42 -60
  157. package/src/providers/openai/__tests__/connection-error-wrap.test.ts +44 -0
  158. package/src/providers/openai/__tests__/orphan-tool-result-guard.test.ts +34 -2
  159. package/src/providers/openai/api-error-normalization.ts +16 -2
  160. package/src/providers/openai/chat-completions-provider.ts +75 -29
  161. package/src/providers/openai/responses-provider.ts +5 -2
  162. package/src/providers/openrouter/client.ts +0 -1
  163. package/src/providers/provider-send-message.ts +11 -0
  164. package/src/providers/retry.ts +6 -0
  165. package/src/providers/search-provider-catalog.ts +20 -0
  166. package/src/providers/vercel-ai-gateway/client.ts +0 -1
  167. package/src/runtime/AGENTS.md +1 -0
  168. package/src/runtime/__tests__/desktop-presence.test.ts +27 -4
  169. package/src/runtime/__tests__/host-observe.test.ts +302 -0
  170. package/src/runtime/channel-readiness-service.ts +214 -14
  171. package/src/runtime/channel-readiness-types.ts +49 -2
  172. package/src/runtime/channel-reply-delivery.ts +2 -2
  173. package/src/runtime/desktop-presence.ts +24 -21
  174. package/src/runtime/host-observe.ts +246 -0
  175. package/src/runtime/http-server.ts +181 -1
  176. package/src/runtime/migrations/__tests__/staged-import-path.test.ts +104 -0
  177. package/src/runtime/migrations/staged-import-path.ts +116 -0
  178. package/src/runtime/routes/__tests__/app-pin-routes.test.ts +383 -0
  179. package/src/runtime/routes/__tests__/conversation-query-routes.test.ts +80 -0
  180. package/src/runtime/routes/__tests__/inference-profiles-routes.test.ts +118 -0
  181. package/src/runtime/routes/__tests__/inference-provider-connection-routes.test.ts +20 -0
  182. package/src/runtime/routes/__tests__/ingress-status-routes.test.ts +508 -0
  183. package/src/runtime/routes/__tests__/plugins-routes.test.ts +35 -56
  184. package/src/runtime/routes/__tests__/watch-routes-guardian-cache.test.ts +139 -0
  185. package/src/runtime/routes/__tests__/watch-routes.test.ts +598 -0
  186. package/src/runtime/routes/app-management-routes.ts +140 -29
  187. package/src/runtime/routes/channel-availability-routes.ts +1 -0
  188. package/src/runtime/routes/channel-readiness-routes.ts +14 -2
  189. package/src/runtime/routes/conversation-query-routes.ts +10 -0
  190. package/src/runtime/routes/guardian-approval-interception.ts +24 -33
  191. package/src/runtime/routes/host-cu-routes.ts +18 -0
  192. package/src/runtime/routes/identity-routes.ts +2 -0
  193. package/src/runtime/routes/inbound-message-handler.ts +10 -7
  194. package/src/runtime/routes/inbound-stages/background-dispatch.test.ts +166 -308
  195. package/src/runtime/routes/inbound-stages/background-dispatch.ts +158 -335
  196. package/src/runtime/routes/index.ts +2 -0
  197. package/src/runtime/routes/inference-profiles-routes.ts +232 -31
  198. package/src/runtime/routes/inference-provider-connection-routes.ts +24 -4
  199. package/src/runtime/routes/ingress-status-routes.ts +180 -0
  200. package/src/runtime/routes/live-voice-routes.test.ts +40 -1
  201. package/src/runtime/routes/live-voice-routes.ts +34 -0
  202. package/src/runtime/routes/migration-routes.ts +218 -10
  203. package/src/runtime/routes/oauth-commands-routes.ts +23 -16
  204. package/src/runtime/routes/plugins-routes.ts +12 -28
  205. package/src/runtime/routes/question-routes.ts +6 -0
  206. package/src/runtime/routes/secret-routes.ts +7 -27
  207. package/src/runtime/routes/settings-routes.ts +9 -6
  208. package/src/runtime/routes/watch-routes.ts +807 -0
  209. package/src/runtime/slack-reply-session.test.ts +230 -121
  210. package/src/runtime/slack-reply-session.ts +113 -81
  211. package/src/runtime/{slack-task-progress.test.ts → task-progress.test.ts} +1 -28
  212. package/src/runtime/{slack-task-progress.ts → task-progress.ts} +30 -51
  213. package/src/security/__tests__/untrusted-content.test.ts +42 -0
  214. package/src/security/untrusted-content.ts +28 -9
  215. package/src/telemetry/__tests__/live-voice-funnel.test.ts +108 -0
  216. package/src/telemetry/live-voice-funnel.ts +75 -8
  217. package/src/tools/credentials/store.ts +18 -6
  218. package/src/tools/network/__tests__/firecrawl-compat.test.ts +77 -0
  219. package/src/tools/network/__tests__/web-fetch-fastcrw.test.ts +169 -0
  220. package/src/tools/network/__tests__/web-search.test.ts +97 -2
  221. package/src/tools/network/firecrawl-compat.ts +90 -0
  222. package/src/tools/network/web-fetch.ts +142 -62
  223. package/src/tools/network/web-search.ts +141 -55
  224. package/src/tools/types.ts +2 -1
  225. package/src/util/oauth-request-body.test.ts +74 -0
  226. package/src/util/oauth-request-body.ts +60 -0
  227. package/src/util/worker-process.ts +1 -0
  228. package/src/watch/__tests__/watch-retro.test.ts +665 -0
  229. package/src/watch/__tests__/watch-session-manager.test.ts +566 -0
  230. package/src/watch/__tests__/watch-timeline.test.ts +670 -0
  231. package/src/watch/watch-retro.ts +480 -0
  232. package/src/watch/watch-session-manager.ts +575 -0
  233. package/src/watch/watch-timeline.ts +848 -0
@@ -18,6 +18,7 @@
18
18
 
19
19
  import { z } from "zod";
20
20
 
21
+ import { validateInferenceProfileConfig } from "../../api/constants/profile-config-validation.js";
21
22
  import {
22
23
  getEffectiveProfilesForProvider,
23
24
  MANAGED_PROFILE_NAMES,
@@ -34,6 +35,7 @@ import {
34
35
  } from "../../config/schemas/llm.js";
35
36
  import { getDb } from "../../persistence/db-connection.js";
36
37
  import {
38
+ catalogProviderForProfile,
37
39
  resolveEntryProviderKind,
38
40
  writableProfileProviderIssue,
39
41
  } from "../../providers/connection-resolution.js";
@@ -45,7 +47,10 @@ import {
45
47
  isUnavailable,
46
48
  } from "../../providers/inference/connection-availability.js";
47
49
  import { getConnection } from "../../providers/inference/connections.js";
50
+ import { probeInferenceProfile } from "../../providers/inference/profile-probe.js";
48
51
  import {
52
+ catalogContextWindowTokens,
53
+ catalogMaxOutputTokens,
49
54
  getModelDisplayName,
50
55
  isModelInCatalog,
51
56
  } from "../../providers/model-catalog.js";
@@ -82,6 +87,18 @@ const availabilitySchema = z
82
87
  })
83
88
  .meta({ id: "ProfileConnectionAvailability" });
84
89
 
90
+ /**
91
+ * Static config problem with the stored entry itself (unknown model,
92
+ * impossible token budget), distinct from `availability` which judges the
93
+ * connection and credential behind it. Absent when the config checks out.
94
+ */
95
+ const profileConfigIssueSchema = z
96
+ .object({
97
+ code: z.enum(["model_unknown", "over_output_cap", "no_input_room"]),
98
+ message: z.string(),
99
+ })
100
+ .meta({ id: "InferenceProfileConfigIssue" });
101
+
85
102
  const profileSummarySchema = z
86
103
  .object({
87
104
  name: z.string(),
@@ -93,6 +110,7 @@ const profileSummarySchema = z
93
110
  provider_connection: z.string().optional(),
94
111
  /** Null when the profile has no provider to judge (e.g. mix profiles). */
95
112
  availability: availabilitySchema.nullable(),
113
+ config_issue: profileConfigIssueSchema.optional(),
96
114
  })
97
115
  .meta({ id: "InferenceProfileSummary" });
98
116
 
@@ -101,6 +119,7 @@ const profileDetailSchema = z
101
119
  name: z.string(),
102
120
  entry: z.record(z.string(), z.unknown()),
103
121
  availability: availabilitySchema.nullable(),
122
+ config_issue: profileConfigIssueSchema.optional(),
104
123
  })
105
124
  .meta({ id: "InferenceProfileDetail" });
106
125
 
@@ -119,6 +138,17 @@ const profileWriteResultSchema = z
119
138
  })
120
139
  .meta({ id: "InferenceProfileWriteResult" });
121
140
 
141
+ const profileCheckSchema = z
142
+ .object({
143
+ ok: z.boolean(),
144
+ blame: z.enum(["profile", "provider", "transient", "unknown"]).optional(),
145
+ reason: z.string().optional(),
146
+ detail: z.string().optional(),
147
+ connection: z.string().optional(),
148
+ message: z.string().optional(),
149
+ })
150
+ .meta({ id: "InferenceProfileCheck" });
151
+
122
152
  const createRequestSchema = z.object({
123
153
  name: z.string().min(1),
124
154
  provider: z.string().min(1),
@@ -160,63 +190,147 @@ function assertValidProvider(provider: string): void {
160
190
  }
161
191
 
162
192
  /**
163
- * Validate a (provider, model) pair against the catalog. Returns warnings
164
- * (never throws) when `allowUnlisted`; throws otherwise for an uncataloged
165
- * model. An uncataloged model always warns, whether or not it is allowed.
193
+ * Why a (provider, model, connection) triple cannot be vouched for, or null.
194
+ * The same reach checks dispatch applies: routing identities against their
195
+ * routing table, entry names against their row's dispatchable kind, vendor
196
+ * ids against the catalog, with the named connection's advertised model list
197
+ * authoritative for models the code-owned catalog doesn't know (a custom
198
+ * endpoint declares its own models at connection-create time).
166
199
  */
167
- function validateModel(
200
+ function modelReachIssue(
168
201
  provider: string,
169
202
  model: string,
170
- allowUnlisted: boolean,
171
203
  connectionName?: string,
172
- ): string[] {
173
- // Routing identities key no catalog entries; they validate against their
174
- // route's actual reach — the same checks dispatch applies per-request.
175
- // allowUnlisted deliberately does not apply: the routing table ships in
176
- // this build, so an unroutable pair fails every request, and the schema
177
- // strips it on the next config read.
204
+ ): { identity: boolean; catalogProvider: string; message: string } | null {
178
205
  if (ROUTING_IDENTITY_PROVIDERS.has(provider)) {
179
206
  const issue = routingIdentityModelIssue(provider, model);
180
- if (issue) {
181
- throw new BadRequestError(
182
- `${issue} Pick a model this route serves, or a concrete provider.`,
183
- );
184
- }
185
- return [];
207
+ return issue
208
+ ? { identity: true, catalogProvider: provider, message: issue }
209
+ : null;
186
210
  }
187
- // An entry-name provider validates against its row's dispatchable kind,
188
- // the same translation dispatch uses; the entry itself also serves as the
189
- // model-list connection for custom endpoints below.
190
211
  const entryKind = resolveEntryProviderKind(provider, model);
191
212
  const catalogProvider = entryKind ?? provider;
192
213
  if (isModelInCatalog(catalogProvider, model)) {
193
- return [];
214
+ return null;
194
215
  }
195
- // The named connection's advertised model list is authoritative for models
196
- // the code-owned catalog doesn't know — a custom (openai-compatible)
197
- // endpoint declares its own models at connection-create time.
198
216
  const modelListConnection =
199
217
  connectionName ?? (entryKind !== null ? provider : undefined);
200
218
  if (modelListConnection) {
201
219
  const connection = getConnection(getDb(), modelListConnection);
202
220
  if (connection?.models?.some((m) => m.id === model)) {
203
- return [];
221
+ return null;
204
222
  }
205
223
  }
224
+ return {
225
+ identity: false,
226
+ catalogProvider,
227
+ message: `Model "${model}" is not in the catalog for provider "${catalogProvider}".`,
228
+ };
229
+ }
230
+
231
+ /**
232
+ * Validate a (provider, model) pair against the catalog. Returns warnings
233
+ * (never throws) when `allowUnlisted`; throws otherwise for an uncataloged
234
+ * model. An uncataloged model always warns, whether or not it is allowed.
235
+ */
236
+ function validateModel(
237
+ provider: string,
238
+ model: string,
239
+ allowUnlisted: boolean,
240
+ connectionName?: string,
241
+ ): string[] {
242
+ const issue = modelReachIssue(provider, model, connectionName);
243
+ if (!issue) {
244
+ return [];
245
+ }
246
+ // allowUnlisted deliberately does not apply to routing identities: the
247
+ // routing table ships in this build, so an unroutable pair fails every
248
+ // request, and the schema strips it on the next config read.
249
+ if (issue.identity) {
250
+ throw new BadRequestError(
251
+ `${issue.message} Pick a model this route serves, or a concrete provider.`,
252
+ );
253
+ }
206
254
  if (!allowUnlisted) {
207
255
  const remedy =
208
- catalogProvider === "openai-compatible"
256
+ issue.catalogProvider === "openai-compatible"
209
257
  ? `Pass allowUnlisted to create it anyway, or declare the model on the connection ` +
210
258
  `("assistant inference providers update <name> --model ${model}").`
211
259
  : `Pass allowUnlisted to create it anyway, or run ` +
212
260
  `"assistant inference models list --provider ${provider}" to see valid ids.`;
213
- throw new BadRequestError(
214
- `Model "${model}" is not in the catalog for provider "${catalogProvider}". ${remedy}`,
261
+ throw new BadRequestError(`${issue.message} ${remedy}`);
262
+ }
263
+ return [`${issue.message} Created anyway (allowUnlisted).`];
264
+ }
265
+
266
+ /**
267
+ * Static config verdict for a stored profile: the model-reach and
268
+ * token-budget judgments the write routes enforce, recomputed over the
269
+ * entry so rows that predate validation (or were written through the
270
+ * generic config escape hatch) surface their problem in listings. Cheap
271
+ * and offline; the live probe covers what only a request can prove. Mix
272
+ * arms are judged on their own rows, and managed bodies are code-owned.
273
+ */
274
+ function profileConfigIssue(record: Record<string, unknown>): {
275
+ code: "model_unknown" | "over_output_cap" | "no_input_room";
276
+ message: string;
277
+ } | null {
278
+ if (record.mix != null || record.source === "managed") {
279
+ return null;
280
+ }
281
+ const provider =
282
+ typeof record.provider === "string" ? record.provider : undefined;
283
+ const model = typeof record.model === "string" ? record.model : undefined;
284
+ if (!provider || !model) {
285
+ // A missing provider/model is availability's `incomplete` verdict.
286
+ return null;
287
+ }
288
+ // A stored allowUnlisted marker records that catalog absence was accepted
289
+ // deliberately at write time; the live probe is the check for those.
290
+ if (record.allowUnlisted !== true) {
291
+ const reach = modelReachIssue(
292
+ provider,
293
+ model,
294
+ typeof record.provider_connection === "string"
295
+ ? record.provider_connection
296
+ : undefined,
215
297
  );
298
+ if (reach) {
299
+ return { code: "model_unknown", message: reach.message };
300
+ }
301
+ }
302
+ if (typeof record.maxTokens === "number") {
303
+ const budget = maxTokensBudgetIssue(provider, model, record.maxTokens);
304
+ if (budget) {
305
+ return { code: budget.code, message: budget.message };
306
+ }
216
307
  }
217
- return [
218
- `Model "${model}" is not in the catalog for provider "${catalogProvider}"; created anyway (allowUnlisted).`,
219
- ];
308
+ return null;
309
+ }
310
+
311
+ /**
312
+ * The shared token-budget judgment over a (provider, model, maxTokens)
313
+ * triple with the catalog-provider translation applied. Null when the budget
314
+ * passes or the catalog knows nothing to judge against. One implementation
315
+ * for the write routes and the listing verdict so the two cannot drift.
316
+ */
317
+ function maxTokensBudgetIssue(
318
+ provider: string,
319
+ model: string,
320
+ maxTokens: number,
321
+ ): ReturnType<typeof validateInferenceProfileConfig> {
322
+ const catalogProvider = catalogProviderForProfile(provider, model);
323
+ if (catalogProvider === null) {
324
+ return null;
325
+ }
326
+ return validateInferenceProfileConfig({
327
+ maxTokens,
328
+ modelMaxOutputTokens: catalogMaxOutputTokens(catalogProvider, model),
329
+ modelContextWindowTokens: catalogContextWindowTokens(
330
+ catalogProvider,
331
+ model,
332
+ ),
333
+ });
220
334
  }
221
335
 
222
336
  function assertConnectionExists(name: string): void {
@@ -373,6 +487,27 @@ function validateProfileEntry(entry: Record<string, unknown>): void {
373
487
  }
374
488
  }
375
489
 
490
+ /**
491
+ * Reject an explicit `maxTokens` the catalog can prove impossible for the
492
+ * model (over its output cap, or reserving the whole context window so no
493
+ * input fits). Judged against the same catalog-provider translation
494
+ * `validateModel` uses; models the catalog does not know are left to the
495
+ * live probe.
496
+ */
497
+ function assertSaneMaxTokens(
498
+ provider: string,
499
+ model: string,
500
+ maxTokens: number | undefined,
501
+ ): void {
502
+ if (maxTokens === undefined) {
503
+ return;
504
+ }
505
+ const issue = maxTokensBudgetIssue(provider, model, maxTokens);
506
+ if (issue) {
507
+ throw new BadRequestError(issue.message);
508
+ }
509
+ }
510
+
376
511
  // ---------------------------------------------------------------------------
377
512
  // Handlers
378
513
  // ---------------------------------------------------------------------------
@@ -386,6 +521,7 @@ async function handleListProfiles() {
386
521
  const profiles = await Promise.all(
387
522
  Object.entries(effective).map(async ([name, entry]) => {
388
523
  const record = entry as Record<string, unknown>;
524
+ const configIssue = profileConfigIssue(record);
389
525
  return {
390
526
  name,
391
527
  label: typeof record.label === "string" ? record.label : null,
@@ -397,6 +533,7 @@ async function handleListProfiles() {
397
533
  ? { provider_connection: record.provider_connection }
398
534
  : {}),
399
535
  availability: await computeProfileAvailability(record),
536
+ ...(configIssue ? { config_issue: configIssue } : {}),
400
537
  };
401
538
  }),
402
539
  );
@@ -418,10 +555,12 @@ async function handleGetProfile({ pathParams = {} }: RouteHandlerArgs) {
418
555
  throw new NotFoundError(`Profile "${name}" not found.`);
419
556
  }
420
557
  const record = entry as Record<string, unknown>;
558
+ const configIssue = profileConfigIssue(record);
421
559
  return {
422
560
  name,
423
561
  entry: record,
424
562
  availability: await computeProfileAvailability(record),
563
+ ...(configIssue ? { config_issue: configIssue } : {}),
425
564
  };
426
565
  }
427
566
 
@@ -456,12 +595,19 @@ async function handleCreateProfile({ body = {} }: RouteHandlerArgs) {
456
595
  input.allowUnlisted ?? false,
457
596
  input.connection,
458
597
  );
598
+ assertSaneMaxTokens(input.provider, input.model, input.maxTokens);
459
599
 
460
600
  const entry: Record<string, unknown> = {
461
601
  ...fragmentFromBody(body as Record<string, unknown>),
462
602
  provider: input.provider,
463
603
  model: input.model,
464
604
  source: "user",
605
+ // `warnings` here can only be validateModel's unlisted verdict (the
606
+ // availability warning pushes later), so its presence is exactly the
607
+ // deliberate allowUnlisted acceptance the listing verdict must honor.
608
+ ...(input.allowUnlisted && warnings.length > 0
609
+ ? { allowUnlisted: true }
610
+ : {}),
465
611
  };
466
612
  // Pickers render the label, so a label-less profile shows its raw config
467
613
  // key (e.g. "gemini-latest"). Default it to the model's human-readable
@@ -585,6 +731,24 @@ async function handleUpdateProfile({
585
731
  nextConnection,
586
732
  );
587
733
  }
734
+ // Gated on the fields that feed the judgment, mirroring the availability
735
+ // guard: metadata-only edits must not start rejecting a stored budget.
736
+ if (
737
+ (input.maxTokens !== undefined ||
738
+ input.model !== undefined ||
739
+ input.provider !== undefined) &&
740
+ typeof nextProvider === "string" &&
741
+ typeof nextModel === "string"
742
+ ) {
743
+ assertSaneMaxTokens(
744
+ nextProvider,
745
+ nextModel,
746
+ input.maxTokens ??
747
+ (typeof existing.maxTokens === "number"
748
+ ? existing.maxTokens
749
+ : undefined),
750
+ );
751
+ }
588
752
 
589
753
  const merged: Record<string, unknown> = {
590
754
  ...existing,
@@ -594,6 +758,17 @@ async function handleUpdateProfile({
594
758
  source:
595
759
  existing.source === "managed" ? "user" : (existing.source ?? "user"),
596
760
  };
761
+ // Keep the allowUnlisted marker faithful to the pair being stored: stamp
762
+ // it on a deliberate unlisted acceptance (warnings can only be the
763
+ // unlisted verdict here), and drop a stale one when the pair is now
764
+ // vouched for. Untouched pairs keep their stored marker.
765
+ if (input.provider !== undefined || input.model !== undefined) {
766
+ if (input.allowUnlisted && warnings.length > 0) {
767
+ merged.allowUnlisted = true;
768
+ } else if (warnings.length === 0) {
769
+ delete merged.allowUnlisted;
770
+ }
771
+ }
597
772
  validateProfileEntry(merged);
598
773
 
599
774
  // The availability guard runs only when the write touches the fields that
@@ -632,6 +807,14 @@ async function handleUpdateProfile({
632
807
  };
633
808
  }
634
809
 
810
+ async function handleValidateProfile({ pathParams = {} }: RouteHandlerArgs) {
811
+ const name = (pathParams.name ?? "").trim();
812
+ if (!name) {
813
+ throw new BadRequestError("Profile name must be a non-empty string");
814
+ }
815
+ return { check: await probeInferenceProfile(name) };
816
+ }
817
+
635
818
  async function handleDeleteProfile({ pathParams = {} }: RouteHandlerArgs) {
636
819
  const name = (pathParams.name ?? "").trim();
637
820
  if (!name) {
@@ -858,4 +1041,22 @@ export const ROUTES: RouteDefinition[] = [
858
1041
  },
859
1042
  handler: handleSetActiveProfile,
860
1043
  },
1044
+ {
1045
+ operationId: "inference_profiles_validate",
1046
+ endpoint: "inference/profiles/:name/validate",
1047
+ method: "POST",
1048
+ policy: {
1049
+ requiredScopes: ["settings.write"],
1050
+ allowedPrincipalTypes: ACTOR_PRINCIPALS,
1051
+ },
1052
+ summary: "Probe a saved profile with one minimal request",
1053
+ description:
1054
+ "Dispatch one minimal test request through the named profile's resolved model and connection, and classify any failure by which object the user should fix (the profile vs its provider connection). Advisory: the probe never mutates the profile. Returns a null check when there is no verdict to give (missing, disabled, managed, or routing-identity profiles, or a probe timeout). The probe spends one tiny request on the profile's own key.",
1055
+ tags: ["inference"],
1056
+ pathParams: [{ name: "name", description: "Profile name" }],
1057
+ responseBody: z.object({
1058
+ check: profileCheckSchema.nullable(),
1059
+ }),
1060
+ handler: handleValidateProfile,
1061
+ },
861
1062
  ];
@@ -39,6 +39,10 @@ import {
39
39
  MANAGED_CONNECTION_NAMES,
40
40
  updateConnection,
41
41
  } from "../../providers/inference/connections.js";
42
+ import {
43
+ EndpointCheckSchema,
44
+ testInferenceConnection,
45
+ } from "../../providers/inference/endpoint-probe.js";
42
46
  import { PROVIDER_CATALOG } from "../../providers/model-catalog.js";
43
47
  import {
44
48
  isVellumManagedConnection,
@@ -64,6 +68,16 @@ const log = getLogger("routes/inference-provider-connections");
64
68
 
65
69
  const providerConnectionResponseSchema = ProviderConnectionSchema;
66
70
 
71
+ /**
72
+ * Create/update responses carry the save-time endpoint probe result for
73
+ * connections with a custom base URL. Advisory only: a failed probe never
74
+ * fails the save (some endpoints legitimately reject unauthenticated or
75
+ * minimal requests); clients render `hint` as a warning.
76
+ */
77
+ const savedConnectionResponseSchema = ProviderConnectionSchema.extend({
78
+ endpoint_check: EndpointCheckSchema.optional(),
79
+ }).meta({ id: "SavedProviderConnection" });
80
+
67
81
  // ---------------------------------------------------------------------------
68
82
  // Custom provider field parsing (openai-compatible base_url + models)
69
83
  // ---------------------------------------------------------------------------
@@ -428,7 +442,10 @@ async function handleCreateConnection({ body = {} }: RouteHandlerArgs) {
428
442
  throw new BadRequestError("Invalid auth configuration.");
429
443
  }
430
444
 
431
- return result.connection;
445
+ const endpointCheck = await testInferenceConnection(result.connection);
446
+ return endpointCheck
447
+ ? { ...result.connection, endpoint_check: endpointCheck }
448
+ : result.connection;
432
449
  }
433
450
 
434
451
  async function handleUpdateConnection({
@@ -547,7 +564,10 @@ async function handleUpdateConnection({
547
564
  throw new BadRequestError("Invalid auth configuration.");
548
565
  }
549
566
 
550
- return result.connection;
567
+ const endpointCheck = await testInferenceConnection(result.connection);
568
+ return endpointCheck
569
+ ? { ...result.connection, endpoint_check: endpointCheck }
570
+ : result.connection;
551
571
  }
552
572
 
553
573
  async function handleDeleteConnection({ pathParams = {} }: RouteHandlerArgs) {
@@ -751,7 +771,7 @@ export const ROUTES: RouteDefinition[] = [
751
771
  base_url: z.string().url().nullable().optional(),
752
772
  models: z.array(ConnectionModelSchema).nullable().optional(),
753
773
  }),
754
- responseBody: providerConnectionResponseSchema,
774
+ responseBody: savedConnectionResponseSchema,
755
775
  responseStatus: "201",
756
776
  additionalResponses: {
757
777
  "400": { description: "Invalid provider or auth schema" },
@@ -779,7 +799,7 @@ export const ROUTES: RouteDefinition[] = [
779
799
  base_url: z.string().url().nullable().optional(),
780
800
  models: z.array(ConnectionModelSchema).nullable().optional(),
781
801
  }),
782
- responseBody: providerConnectionResponseSchema,
802
+ responseBody: savedConnectionResponseSchema,
783
803
  additionalResponses: {
784
804
  "400": {
785
805
  description:
@@ -0,0 +1,180 @@
1
+ /**
2
+ * Ingress status route: is the assistant's public tunnel actually up?
3
+ *
4
+ * The recorded `ingress.publicBaseUrl` only says a tunnel was configured at
5
+ * some point. It stays behind in both directions: a killed tunnel leaves the
6
+ * URL in place, and an edge can survive while the gateway behind it dies or
7
+ * while it starts fronting a different assistant. This route resolves that by
8
+ * probing the URL live (`inbound/tunnel-probe.ts`) and reporting one of six
9
+ * states, so a client can tell the user what to do instead of guessing. The
10
+ * card the states drive exists to pair devices, so an address that answers
11
+ * without serving the pairing app is its own state (`unpairable`) rather than
12
+ * a healthy one.
13
+ *
14
+ * Platform-hosted assistants receive webhooks through platform callback
15
+ * routing and never run `vellum tunnel`, so they report `unconfigured`
16
+ * rather than a tunnel state they cannot act on. A self-hosted gateway whose
17
+ * Velay tunnel owns the URL is the same story one layer down: that URL
18
+ * carries webhooks alone, so it is `unconfigured` for pairing too, and the
19
+ * gateway publishes and restores it without a tunnel command either way.
20
+ */
21
+ import { normalizePublicBaseUrl } from "@vellumai/service-contracts/ingress";
22
+ import { z } from "zod";
23
+
24
+ import {
25
+ getIngressConfigResult,
26
+ isVelayManagedIngress,
27
+ loadLastTunnelRecord,
28
+ loadPairingTunnelRecord,
29
+ loadRecordedAssistantId,
30
+ } from "../../daemon/handlers/config-ingress.js";
31
+ import { probeTunnel } from "../../inbound/tunnel-probe.js";
32
+ import { ACTOR_PRINCIPALS } from "../auth/route-policy.js";
33
+ import type { RouteDefinition } from "./types.js";
34
+
35
+ // ── Schemas ─────────────────────────────────────────────────────────────
36
+
37
+ /**
38
+ * Flat rather than a union on `state` so the generated client SDK stays a
39
+ * single type; which fields are populated follows from `state`.
40
+ */
41
+ const IngressStatusResponseSchema = z.object({
42
+ state: z.enum([
43
+ "unconfigured",
44
+ "stopped",
45
+ "healthy",
46
+ "unpairable",
47
+ "unreachable",
48
+ "foreign",
49
+ ]),
50
+ publicBaseUrl: z
51
+ .string()
52
+ .optional()
53
+ .describe(
54
+ "The probed URL. Set for healthy, unpairable, unreachable, and foreign.",
55
+ ),
56
+ lastTunnel: z
57
+ .object({ provider: z.string(), publicBaseUrl: z.string() })
58
+ .optional()
59
+ .describe(
60
+ "The tunnel to restart. Set on stopped whenever one is recorded, and on unpairable, unreachable and foreign when the recorded tunnel fronted the configured URL.",
61
+ ),
62
+ servingAssistantName: z
63
+ .string()
64
+ .optional()
65
+ .describe(
66
+ "Name the edge serves under. Set for foreign when it reports one.",
67
+ ),
68
+ detail: z
69
+ .string()
70
+ .optional()
71
+ .describe(
72
+ "Short failure reason, free of the URL. Set for unreachable, and for unpairable when the edge gave one.",
73
+ ),
74
+ checkedAt: z
75
+ .string()
76
+ .optional()
77
+ .describe("ISO-8601 instant the probe finished. Set whenever one ran."),
78
+ });
79
+ type IngressStatusResponse = z.infer<typeof IngressStatusResponseSchema>;
80
+
81
+ // ── Handlers ────────────────────────────────────────────────────────────
82
+
83
+ async function handleIngressStatus(): Promise<IngressStatusResponse> {
84
+ const config = getIngressConfigResult();
85
+ if (config.managedCallbacks) {
86
+ return { state: "unconfigured" };
87
+ }
88
+
89
+ // A tunnel published for pairing alone is the address this route reports on:
90
+ // it exists to answer the pairing card, and the `publicBaseUrl` beside it is
91
+ // a webhook callback base that tunnel deliberately left in place. The user
92
+ // owns and restarts that tunnel even where Velay owns the callback ingress.
93
+ const pairingTunnel = loadPairingTunnelRecord();
94
+ // A Velay-managed URL belongs to the gateway, which writes it, clears it,
95
+ // and reconnects on its own, so there is no tunnel for the user to restart.
96
+ // Velay's allowlist (`gateway/src/velay/allowed-paths.ts`) also exposes
97
+ // neither `/healthz` nor the `/assistant/*` pairing surface, so that URL can
98
+ // only ever probe as dead and prefills a page pairing cannot open. Without a
99
+ // pairing tunnel beside it there is no address to report at all.
100
+ if (isVelayManagedIngress() && !pairingTunnel) {
101
+ return { state: "unconfigured" };
102
+ }
103
+ const lastTunnel = pairingTunnel ?? loadLastTunnelRecord();
104
+
105
+ const publicBaseUrl =
106
+ pairingTunnel?.publicBaseUrl ?? config.publicBaseUrl.trim();
107
+ if (!publicBaseUrl && !lastTunnel) {
108
+ return { state: "unconfigured" };
109
+ }
110
+ // `ingress.enabled` governs callback ingress alone, and the URL survives
111
+ // that toggle, so probing a callback base while ingress is opted out would
112
+ // report an address the user switched off as healthy. A pairing tunnel is
113
+ // published outside the toggle, so the opt-out says nothing about it. An
114
+ // unset flag is not an opt-out either: a bring-your-own-HTTPS front sets
115
+ // only the URL.
116
+ if (!publicBaseUrl || (!pairingTunnel && config.explicitlyDisabled)) {
117
+ return { state: "stopped", ...(lastTunnel ? { lastTunnel } : {}) };
118
+ }
119
+
120
+ // A record left over from an address that is no longer configured names a
121
+ // tunnel that never fronted this one, so it is no help in getting back.
122
+ const restartHint =
123
+ lastTunnel &&
124
+ normalizePublicBaseUrl(lastTunnel.publicBaseUrl) ===
125
+ normalizePublicBaseUrl(publicBaseUrl)
126
+ ? { lastTunnel }
127
+ : {};
128
+
129
+ // Omitted when unrecorded so the probe skips the identity check entirely
130
+ // rather than reading every served id as a mismatch.
131
+ const expectedAssistantId = loadRecordedAssistantId();
132
+ const result = await probeTunnel({
133
+ publicBaseUrl,
134
+ ...(expectedAssistantId ? { expectedAssistantId } : {}),
135
+ });
136
+ const checked = { publicBaseUrl, checkedAt: new Date().toISOString() };
137
+
138
+ switch (result.kind) {
139
+ case "healthy":
140
+ return { state: "healthy", ...checked };
141
+ // One remedy for both: start a tunnel that serves the web app.
142
+ case "unpairable":
143
+ case "unreachable":
144
+ return {
145
+ state: result.kind,
146
+ ...checked,
147
+ ...restartHint,
148
+ ...(result.detail ? { detail: result.detail } : {}),
149
+ };
150
+ case "foreign":
151
+ return {
152
+ state: "foreign",
153
+ ...checked,
154
+ ...restartHint,
155
+ ...(result.assistantName
156
+ ? { servingAssistantName: result.assistantName }
157
+ : {}),
158
+ };
159
+ }
160
+ }
161
+
162
+ // ── Route definitions ───────────────────────────────────────────────────
163
+
164
+ export const ROUTES: RouteDefinition[] = [
165
+ {
166
+ operationId: "integrations_ingress_status",
167
+ endpoint: "integrations/ingress/status",
168
+ method: "GET",
169
+ policy: {
170
+ requiredScopes: ["settings.read"],
171
+ allowedPrincipalTypes: ACTOR_PRINCIPALS,
172
+ },
173
+ summary: "Get ingress tunnel status",
174
+ description:
175
+ "Probe the configured public ingress URL and report whether a tunnel is serving this assistant. `unconfigured` means no tunnel has ever been recorded (or the ingress is managed for this assistant, as it is when platform-hosted or fronted by a Velay tunnel), `stopped` means a tunnel ran before and is not running now (callback ingress is explicitly switched off, or its URL is gone), `healthy` means the URL answers and fronts this assistant's web app, `unpairable` means it answers but does not serve that app, so a pairing link minted against it would not open, `unreachable` means it does not answer, and `foreign` means it answers for a different assistant. `lastTunnel` names the tunnel to restart, on the states where a recorded one applies to the configured URL.",
176
+ tags: ["config"],
177
+ handler: handleIngressStatus,
178
+ responseBody: IngressStatusResponseSchema,
179
+ },
180
+ ];