@vellumai/assistant 0.10.6-staging.1 → 0.10.6

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (58) hide show
  1. package/openapi.yaml +43 -6
  2. package/package.json +1 -1
  3. package/src/__tests__/attachment-stored-path-annotation.test.ts +151 -0
  4. package/src/__tests__/attachments-store.test.ts +37 -1
  5. package/src/__tests__/attachments.test.ts +53 -0
  6. package/src/__tests__/channel-reply-delivery.test.ts +214 -5
  7. package/src/__tests__/config-loader-backfill.test.ts +16 -36
  8. package/src/__tests__/image-source-path-reinject.test.ts +155 -14
  9. package/src/__tests__/managed-profile-guard.test.ts +2 -0
  10. package/src/__tests__/mtime-cache.test.ts +265 -80
  11. package/src/__tests__/plugin-config-data-migration.test.ts +2 -4
  12. package/src/__tests__/plugin-effective-enabled-set.test.ts +9 -6
  13. package/src/__tests__/workspace-migration-125-repoint-managed-connections-to-vellum.test.ts +140 -0
  14. package/src/agent/attachments.ts +55 -13
  15. package/src/cli/commands/platform/__tests__/credits.test.ts +111 -0
  16. package/src/cli/commands/platform/index.ts +86 -4
  17. package/src/cli/lib/__tests__/confirm-prompt.test.ts +9 -9
  18. package/src/cli/lib/diff-plugin.ts +7 -3
  19. package/src/cli/lib/inspect-plugin.ts +28 -10
  20. package/src/cli/lib/install-from-github.ts +69 -24
  21. package/src/cli/lib/ipc-params.ts +3 -3
  22. package/src/cli/lib/list-installed-plugins.ts +11 -5
  23. package/src/cli/lib/merge-plugin-tree.ts +15 -9
  24. package/src/cli/lib/plugin-fingerprint.ts +0 -6
  25. package/src/cli/lib/publish-plugin.ts +1 -1
  26. package/src/cli/lib/unknown-command.ts +3 -1
  27. package/src/cli/lib/upgrade-plugin.ts +2 -5
  28. package/src/config/__tests__/sync-gated-profiles.test.ts +2 -2
  29. package/src/config/seed-inference-profiles.ts +5 -4
  30. package/src/daemon/conversation-lifecycle.ts +37 -15
  31. package/src/daemon/conversation-messaging.ts +56 -28
  32. package/src/daemon/conversation.ts +2 -2
  33. package/src/heartbeat/__tests__/heartbeat-service.test.ts +6 -0
  34. package/src/hooks/__tests__/hook-live-reload.test.ts +110 -96
  35. package/src/hooks/hook-loader.ts +57 -144
  36. package/src/hooks/registry.ts +13 -9
  37. package/src/persistence/attachments-store.ts +2 -1
  38. package/src/persistence/conversation-crud.ts +24 -0
  39. package/src/plugin-api/types.ts +4 -1
  40. package/src/plugins/mtime-cache.ts +347 -33
  41. package/src/plugins/pipeline.ts +3 -2
  42. package/src/providers/__tests__/inference.test.ts +42 -31
  43. package/src/providers/inference/__tests__/connections-label.test.ts +9 -28
  44. package/src/providers/inference/auth.ts +1 -2
  45. package/src/providers/inference/backfill.ts +7 -5
  46. package/src/providers/inference/connections.ts +54 -30
  47. package/src/runtime/channel-reply-delivery.ts +83 -24
  48. package/src/runtime/channel-retry-sweep.ts +4 -0
  49. package/src/runtime/{slack-no-response.ts → no-response.ts} +26 -6
  50. package/src/runtime/routes/__tests__/conversation-query-routes.test.ts +43 -3
  51. package/src/runtime/routes/__tests__/inference-provider-connection-routes.test.ts +61 -24
  52. package/src/runtime/routes/conversation-query-routes.ts +18 -1
  53. package/src/runtime/routes/inbound-stages/background-dispatch.ts +1 -1
  54. package/src/runtime/routes/inference-provider-connection-routes.ts +4 -3
  55. package/src/runtime/routes/platform-routes.ts +82 -1
  56. package/src/runtime/slack-reply-session.ts +1 -1
  57. package/src/workspace/migrations/125-repoint-managed-connections-to-vellum.ts +99 -0
  58. package/src/workspace/migrations/registry.ts +2 -0
@@ -97,16 +97,14 @@ describe("connection CRUD label defaults", () => {
97
97
  });
98
98
 
99
99
  describe("seedCanonicalConnections labels", () => {
100
- test("first boot seeds default labels on all managed connections", () => {
100
+ test("first boot seeds the default label on the vellum connection", () => {
101
101
  const db = bootDb();
102
102
  seedCanonicalConnections(db);
103
103
 
104
104
  const conns = listConnections(db);
105
105
  const byName = Object.fromEntries(conns.map((c) => [c.name, c]));
106
106
 
107
- expect(byName["anthropic-managed"]?.label).toBe("Anthropic");
108
- expect(byName["openai-managed"]?.label).toBe("OpenAI");
109
- expect(byName["gemini-managed"]?.label).toBe("Google Gemini");
107
+ expect(byName["vellum"]?.label).toBe("Vellum");
110
108
  });
111
109
 
112
110
  test("second boot preserves user-customized label", () => {
@@ -114,33 +112,16 @@ describe("seedCanonicalConnections labels", () => {
114
112
  seedCanonicalConnections(db);
115
113
 
116
114
  // User customizes the label.
117
- updateConnection(db, "anthropic-managed", {
115
+ updateConnection(db, "vellum", {
118
116
  auth: { type: "platform" },
119
- label: "Work Anthropic",
117
+ label: "Work Vellum",
120
118
  });
121
119
 
122
120
  // Reboot.
123
121
  seedCanonicalConnections(db);
124
122
 
125
- const conn = getConnection(db, "anthropic-managed");
126
- expect(conn?.label).toBe("Work Anthropic");
127
- });
128
-
129
- test("second boot backfills default label when existing row has null label", () => {
130
- const db = bootDb();
131
-
132
- // `bootDb()` runs migration 243 which already inserted the three
133
- // canonical rows with `label=null` (the label column was added by 244
134
- // and defaults NULL for pre-existing rows). This matches the state
135
- // every pre-label install carries forward into the boot that ships
136
- // the label seed.
137
- const before = getConnection(db, "anthropic-managed");
138
- expect(before?.label).toBeNull();
139
-
140
- seedCanonicalConnections(db);
141
-
142
- const after = getConnection(db, "anthropic-managed");
143
- expect(after?.label).toBe("Anthropic");
123
+ const conn = getConnection(db, "vellum");
124
+ expect(conn?.label).toBe("Work Vellum");
144
125
  });
145
126
 
146
127
  test("backfill fills null label on subsequent boot", () => {
@@ -148,7 +129,7 @@ describe("seedCanonicalConnections labels", () => {
148
129
  seedCanonicalConnections(db);
149
130
 
150
131
  // User clears the label (PATCH label: null).
151
- updateConnection(db, "openai-managed", {
132
+ updateConnection(db, "vellum", {
152
133
  auth: { type: "platform" },
153
134
  label: null,
154
135
  });
@@ -160,7 +141,7 @@ describe("seedCanonicalConnections labels", () => {
160
141
  // present for everyone.
161
142
  seedCanonicalConnections(db);
162
143
 
163
- const conn = getConnection(db, "openai-managed");
164
- expect(conn?.label).toBe("OpenAI");
144
+ const conn = getConnection(db, "vellum");
145
+ expect(conn?.label).toBe("Vellum");
165
146
  });
166
147
  });
@@ -117,8 +117,7 @@ export const ProviderConnectionSchema = z
117
117
  createdAt: z.number().int(),
118
118
  updatedAt: z.number().int(),
119
119
  /**
120
- * Whether this row is a Vellum-managed connection (`anthropic-managed`,
121
- * `openai-managed`, `gemini-managed`). Derived from
120
+ * Whether this row is the Vellum-managed connection (`vellum`). Derived from
122
121
  * `MANAGED_CONNECTION_NAMES` in `connections.ts` at serialize time; the
123
122
  * DB column does not exist. Clients use this to render the read-only
124
123
  * "Vellum" badge + view-only editor and to disable the delete affordance
@@ -21,18 +21,17 @@ import { loadRawConfig, saveRawConfig } from "../../config/loader.js";
21
21
  import type { DrizzleDb } from "../../persistence/db-connection.js";
22
22
  import { credentialKey } from "../../security/credential-key.js";
23
23
  import { getLogger } from "../../util/logger.js";
24
+ import { MANAGED_ROUTABLE_PROVIDERS } from "../vellum-model-routing.js";
24
25
  import {
25
26
  createConnection,
26
27
  getConnection,
27
28
  PROVIDERS_REQUIRING_BASE_URL_AND_MODELS,
28
29
  seedCanonicalConnections,
30
+ VELLUM_MANAGED_CONNECTION_NAME,
29
31
  } from "./connections.js";
30
32
 
31
33
  const log = getLogger("provider-connections-backfill");
32
34
 
33
- // Providers that support the managed (platform) auth type.
34
- const MANAGED_PROVIDERS = new Set(["anthropic", "openai", "gemini"]);
35
-
36
35
  /**
37
36
  * Seed canonical provider_connections and backfill any legacy config locations
38
37
  * that pre-date the connection field.
@@ -168,8 +167,11 @@ function ensureProviderConnection(
168
167
 
169
168
  let connectionName: string;
170
169
 
171
- if (globalMode === "managed" && MANAGED_PROVIDERS.has(provider)) {
172
- connectionName = `${provider}-managed`;
170
+ if (globalMode === "managed" && MANAGED_ROUTABLE_PROVIDERS.has(provider)) {
171
+ // All managed-routable providers share the single provider-agnostic
172
+ // `vellum` connection; the upstream is recovered per-request from the
173
+ // profile's `provider` field.
174
+ connectionName = VELLUM_MANAGED_CONNECTION_NAME;
173
175
  } else {
174
176
  // "your-own" path (or provider not managed-supported): ensure a
175
177
  // personal connection exists. Ollama is keyless, so it gets
@@ -3,7 +3,9 @@ import { z } from "zod";
3
3
 
4
4
  import type { DrizzleDb } from "../../persistence/db-connection.js";
5
5
  import { providerConnections } from "../../persistence/schema/inference.js";
6
+ import { getLogger } from "../../util/logger.js";
6
7
  import { clearConnectionProviderCache } from "../registry.js";
8
+ import { VELLUM_MANAGED_PROVIDER } from "../vellum-model-routing.js";
7
9
  import {
8
10
  type Auth,
9
11
  AuthSchema,
@@ -15,6 +17,8 @@ import {
15
17
  VALID_CONNECTION_PROVIDERS,
16
18
  } from "./auth.js";
17
19
 
20
+ const log = getLogger("providers/inference/connections");
21
+
18
22
  // ---------------------------------------------------------------------------
19
23
  // Helpers
20
24
  // ---------------------------------------------------------------------------
@@ -316,6 +320,30 @@ export function deleteConnection(
316
320
  // Seed canonical connections (upsert, used at boot time)
317
321
  // ---------------------------------------------------------------------------
318
322
 
323
+ /**
324
+ * Name of the single Vellum-managed connection. Provider-agnostic: it stores
325
+ * the `vellum` sentinel instead of a real upstream, and the upstream is chosen
326
+ * per-request from the resolving profile's provider (see
327
+ * `resolveProviderFromConnection` / `isVellumManagedConnection`). This replaces
328
+ * the former per-provider `*-managed` connections.
329
+ */
330
+ export const VELLUM_MANAGED_CONNECTION_NAME = "vellum";
331
+
332
+ /**
333
+ * The former per-provider `*-managed` connection names, now collapsed into the
334
+ * single `vellum` connection. They are no longer seeded, but existing installs
335
+ * (and fresh installs seeded by migration 243) may still carry these rows until
336
+ * a follow-up migration deletes them. They are filtered out of the connection
337
+ * list so the UI never shows the pre-consolidation duplicates.
338
+ */
339
+ export const LEGACY_MANAGED_CONNECTION_NAMES: ReadonlySet<string> = new Set([
340
+ "anthropic-managed",
341
+ "openai-managed",
342
+ "gemini-managed",
343
+ "fireworks-managed",
344
+ "together-managed",
345
+ ]);
346
+
319
347
  const CANONICAL_CONNECTIONS: Array<{
320
348
  name: string;
321
349
  provider: string;
@@ -323,40 +351,16 @@ const CANONICAL_CONNECTIONS: Array<{
323
351
  label: string;
324
352
  }> = [
325
353
  {
326
- name: "anthropic-managed",
327
- provider: "anthropic",
354
+ name: VELLUM_MANAGED_CONNECTION_NAME,
355
+ provider: VELLUM_MANAGED_PROVIDER,
328
356
  auth: { type: "platform" },
329
- label: "Anthropic",
330
- },
331
- {
332
- name: "openai-managed",
333
- provider: "openai",
334
- auth: { type: "platform" },
335
- label: "OpenAI",
336
- },
337
- {
338
- name: "gemini-managed",
339
- provider: "gemini",
340
- auth: { type: "platform" },
341
- label: "Google Gemini",
342
- },
343
- {
344
- name: "fireworks-managed",
345
- provider: "fireworks",
346
- auth: { type: "platform" },
347
- label: "Fireworks",
348
- },
349
- {
350
- name: "together-managed",
351
- provider: "together",
352
- auth: { type: "platform" },
353
- label: "Together AI",
357
+ label: "Vellum",
354
358
  },
355
359
  ];
356
360
 
357
361
  /**
358
- * Names of the canonical Vellum-managed connections. These are seeded on every
359
- * daemon boot via `seedCanonicalConnections` and represent the platform-managed
362
+ * Names of the canonical Vellum-managed connections. Seeded on every daemon
363
+ * boot via `seedCanonicalConnections` and representing the platform-managed
360
364
  * inference route. They are write-protected at the route layer:
361
365
  * - DELETE is blocked outright (would resurrect on next boot anyway, but
362
366
  * blocking prevents a confusing delete → re-appear loop).
@@ -374,7 +378,7 @@ export const MANAGED_CONNECTION_NAMES: ReadonlySet<string> = new Set(
374
378
  );
375
379
 
376
380
  /**
377
- * Upsert the three canonical connections on every boot. Existing rows are
381
+ * Upsert the canonical connections on every boot. Existing rows are
378
382
  * updated to the latest provider/auth values so Vellum can push connection
379
383
  * changes to customers in new releases.
380
384
  *
@@ -388,6 +392,26 @@ export const MANAGED_CONNECTION_NAMES: ReadonlySet<string> = new Set(
388
392
  export function seedCanonicalConnections(db: DrizzleDb): void {
389
393
  const now = Date.now();
390
394
  for (const { name, provider, auth, label } of CANONICAL_CONNECTIONS) {
395
+ // Never clobber a pre-existing user connection that happens to share this
396
+ // canonical name. `vellum` was not a reserved name before consolidation, so
397
+ // an install may already have a BYOK connection keyed `vellum` (with a real
398
+ // provider + api_key auth). Upserting it here would silently rewrite it to
399
+ // the platform-auth sentinel and make it managed/write-protected, breaking
400
+ // any profile that references the user's connection. Only seed/refresh when
401
+ // the row is absent or already our sentinel (same provider).
402
+ const existing = db
403
+ .select({ provider: providerConnections.provider })
404
+ .from(providerConnections)
405
+ .where(eq(providerConnections.name, name))
406
+ .get();
407
+ if (existing && existing.provider !== provider) {
408
+ log.warn(
409
+ { name, existingProvider: existing.provider },
410
+ `Skipping canonical seed for "${name}": a user-owned connection already claims this name.`,
411
+ );
412
+ continue;
413
+ }
414
+
391
415
  db.insert(providerConnections)
392
416
  .values({
393
417
  name,
@@ -12,6 +12,10 @@ import { getLogger } from "../util/logger.js";
12
12
  import type { ChannelDeliveryResult } from "./gateway-client.js";
13
13
  import { deliverChannelReply } from "./gateway-client.js";
14
14
  import type { RuntimeAttachmentMetadata } from "./http-types.js";
15
+ import {
16
+ containsNoResponseMarker,
17
+ stripNoResponseMarkers,
18
+ } from "./no-response.js";
15
19
 
16
20
  const log = getLogger("channel-reply-delivery");
17
21
 
@@ -50,40 +54,69 @@ function sleep(ms: number): Promise<void> {
50
54
  return new Promise((resolve) => setTimeout(resolve, ms));
51
55
  }
52
56
 
53
- const NO_RESPONSE_RE = /^\s*<no_response\s*\/?>\s*$/;
54
-
55
- /** Returns true when any segment is a `<no_response/>` sentinel. */
57
+ /** Returns true when any segment carries a `<no_response/>` sentinel. */
56
58
  function hasNoResponseMarker(textSegments: string[]): boolean {
57
- return textSegments.some((s) => NO_RESPONSE_RE.test(s));
59
+ return textSegments.some(containsNoResponseMarker);
58
60
  }
59
61
 
60
62
  function toDeliverableTextSegments(
61
63
  textSegments: string[],
62
64
  fallbackText?: string,
63
65
  ): string[] {
66
+ // Hide the sentinel itself, never the content around it: strip every
67
+ // occurrence from mixed segments so a reply the model wrapped around a
68
+ // stray <no_response/> still reaches the user, and the raw sentinel never
69
+ // leaks into the channel as visible text.
64
70
  const nonEmptySegments = textSegments
65
- .map(stripVellumLinks)
66
- .filter(
67
- (segment) => segment.trim().length > 0 && !NO_RESPONSE_RE.test(segment),
68
- );
71
+ .map((segment) => {
72
+ const stripped = stripVellumLinks(segment);
73
+ return containsNoResponseMarker(stripped)
74
+ ? stripNoResponseMarkers(stripped)
75
+ : stripped;
76
+ })
77
+ .filter((segment) => segment.trim().length > 0);
69
78
  if (nonEmptySegments.length > 0) return nonEmptySegments;
70
79
  // If the only text was <no_response/>, treat as intentional silence —
71
80
  // do not fall back to fallbackText.
72
81
  if (hasNoResponseMarker(textSegments)) return [];
73
- if (typeof fallbackText === "string" && fallbackText.trim().length > 0) {
74
- return [fallbackText];
82
+ if (typeof fallbackText === "string") {
83
+ const fallback = stripNoResponseMarkers(fallbackText);
84
+ if (fallback.length > 0) return [fallback];
75
85
  }
76
86
  return [];
77
87
  }
78
88
 
89
+ /**
90
+ * A reply worth delivering on its own merits: real text after sentinel
91
+ * stripping, or attachments. A bare `<no_response/>` row does NOT count —
92
+ * even when it carries attachments, since `deliverRenderedReplyViaCallback`
93
+ * suppresses attachment delivery for marker rows; counting such a row as
94
+ * real would stop the turn scan on a row that delivers nothing.
95
+ */
96
+ function hasRealDeliverableReply(
97
+ rendered: RenderedHistoryContent,
98
+ attachments: RuntimeAttachmentMetadata[],
99
+ ): boolean {
100
+ if (
101
+ toDeliverableTextSegments(rendered.textSegments, rendered.text).length > 0
102
+ ) {
103
+ return true;
104
+ }
105
+ return attachments.length > 0 && !hasNoResponseMarker(rendered.textSegments);
106
+ }
107
+
108
+ /**
109
+ * A reply whose delivery is a terminal outcome: real content, or a bare
110
+ * `<no_response/>` sentinel whose "delivery" is deliberate silence. The
111
+ * unbounded fallback scan stops at either so a silence marker can never be
112
+ * skipped in favor of re-delivering an older turn's reply.
113
+ */
79
114
  function hasDeliverableReply(
80
115
  rendered: RenderedHistoryContent,
81
116
  attachments: RuntimeAttachmentMetadata[],
82
117
  ): boolean {
83
118
  return (
84
- toDeliverableTextSegments(rendered.textSegments, rendered.text).length >
85
- 0 ||
86
- attachments.length > 0 ||
119
+ hasRealDeliverableReply(rendered, attachments) ||
87
120
  hasNoResponseMarker(rendered.textSegments)
88
121
  );
89
122
  }
@@ -284,16 +317,27 @@ export function findAssistantReplyMessageIdForTurn(
284
317
  }
285
318
  }
286
319
 
320
+ let sentinelRowId: string | undefined;
287
321
  for (let i = turnEndIndex - 1; i > userIndex; i--) {
288
322
  const msg = msgs[i];
289
323
  if (msg.role === "assistant") {
290
324
  const { rendered, replyAttachments } = readPersistedAssistantReply(msg);
291
- if (hasDeliverableReply(rendered, replyAttachments)) {
325
+ if (hasRealDeliverableReply(rendered, replyAttachments)) {
292
326
  return msg.id;
293
327
  }
328
+ if (
329
+ sentinelRowId === undefined &&
330
+ hasNoResponseMarker(rendered.textSegments)
331
+ ) {
332
+ sentinelRowId = msg.id;
333
+ }
294
334
  }
295
335
  }
296
- return undefined;
336
+ // Silence means the turn produced no real reply text anywhere — not "the
337
+ // last row was a sentinel". Only when no row in the turn carries real
338
+ // content does the bare <no_response/> row become the reply; delivering it
339
+ // suppresses all output as deliberate silence.
340
+ return sentinelRowId;
297
341
  }
298
342
 
299
343
  async function deliverPersistedAssistantMessageViaCallback(
@@ -302,8 +346,10 @@ async function deliverPersistedAssistantMessageViaCallback(
302
346
  callbackUrl: string,
303
347
  assistantId: string | undefined,
304
348
  options: DeliverReplyOptions | undefined,
349
+ preRead?: ReturnType<typeof readPersistedAssistantReply>,
305
350
  ): Promise<boolean> {
306
- const { rendered, replyAttachments } = readPersistedAssistantReply(msg);
351
+ const { rendered, replyAttachments } =
352
+ preRead ?? readPersistedAssistantReply(msg);
307
353
  if (!hasDeliverableReply(rendered, replyAttachments)) {
308
354
  return false;
309
355
  }
@@ -355,14 +401,27 @@ export async function deliverReplyViaCallback(
355
401
  `Target assistant reply message not found: ${options.messageId}`,
356
402
  );
357
403
  }
358
- await deliverPersistedAssistantMessageViaCallback(
359
- msg,
360
- externalChatId,
361
- callbackUrl,
362
- assistantId,
363
- options,
364
- );
365
- return;
404
+ // The targeted row is usually the turn's final assistant row, but a bare
405
+ // <no_response/> or tool-only final row is not necessarily the turn's
406
+ // reply — the model may have written the real reply in an earlier row of
407
+ // the same turn. When the turn boundary is known, defer to the turn scan
408
+ // below; it returns this same sentinel row (deliberate silence) only when
409
+ // no row in the turn carries real content.
410
+ const preRead = readPersistedAssistantReply(msg);
411
+ if (
412
+ hasRealDeliverableReply(preRead.rendered, preRead.replyAttachments) ||
413
+ !options.sinceMessageId
414
+ ) {
415
+ await deliverPersistedAssistantMessageViaCallback(
416
+ msg,
417
+ externalChatId,
418
+ callbackUrl,
419
+ assistantId,
420
+ options,
421
+ preRead,
422
+ );
423
+ return;
424
+ }
366
425
  }
367
426
 
368
427
  if (options?.sinceMessageId) {
@@ -436,6 +436,10 @@ export async function sweepFailedEvents(
436
436
  assistantId,
437
437
  {
438
438
  messageId: replyMessageId,
439
+ // The stored reply id may point at a bare <no_response/> row from
440
+ // the turn's final message; the turn boundary lets delivery fall
441
+ // through to the real reply written earlier in the same turn.
442
+ ...(event.messageId ? { sinceMessageId: event.messageId } : {}),
439
443
  startFromSegment: event.deliveredSegmentCount,
440
444
  ...(priorStreamMessageTs ? { messageTs: priorStreamMessageTs } : {}),
441
445
  onSegmentDelivered: (count) =>
@@ -1,10 +1,12 @@
1
1
  /**
2
2
  * The assistant emits a `<no_response/>` sentinel to signal that a turn should
3
- * produce no user-visible reply. Slack surfaces that begin work on the first
4
- * deliverable text (the thinking-status indicator and native reply streaming)
5
- * must not act on a partially-arrived sentinel: a slow stream can surface just
6
- * `<` or `<no_resp` before the rest lands, and starting on that leaks a stray
7
- * message that later resolves to empty.
3
+ * produce no user-visible reply. Channel delivery must hide the sentinel
4
+ * itself without hiding real content it happens to coexist with. Slack
5
+ * surfaces that begin work on the first deliverable text (the thinking-status
6
+ * indicator and native reply streaming) additionally must not act on a
7
+ * partially-arrived sentinel: a slow stream can surface just `<` or `<no_resp`
8
+ * before the rest lands, and starting on that leaks a stray message that
9
+ * later resolves to empty.
8
10
  */
9
11
 
10
12
  /** Matches a message whose entire content is the sentinel. */
@@ -13,6 +15,24 @@ const NO_RESPONSE_ONLY_RE = /^\s*<no_response\s*\/?>\s*$/i;
13
15
  /** Matches every sentinel occurrence for stripping it out of mixed content. */
14
16
  export const NO_RESPONSE_INLINE_RE = /<no_response\s*\/?>/gi;
15
17
 
18
+ /**
19
+ * Detection variant without the `g` flag: a `g`-flagged regex is stateful
20
+ * under `.test()` (it resumes from `lastIndex`), so reusing
21
+ * {@link NO_RESPONSE_INLINE_RE} for detection would alternate between
22
+ * matches and misses across calls.
23
+ */
24
+ const NO_RESPONSE_MARKER_RE = new RegExp(NO_RESPONSE_INLINE_RE.source, "i");
25
+
26
+ /** Whether `text` contains the sentinel anywhere. */
27
+ export function containsNoResponseMarker(text: string): boolean {
28
+ return NO_RESPONSE_MARKER_RE.test(text);
29
+ }
30
+
31
+ /** Removes every sentinel occurrence, trimming the leftover whitespace. */
32
+ export function stripNoResponseMarkers(text: string): string {
33
+ return text.replace(NO_RESPONSE_INLINE_RE, "").trim();
34
+ }
35
+
16
36
  const NO_RESPONSE_SENTINEL_FORMS = [
17
37
  "<no_response/>",
18
38
  "<no_response />",
@@ -43,5 +63,5 @@ export function hasDeliverableAssistantText(text: string): boolean {
43
63
  if (trimmed.length === 0) return false;
44
64
  if (NO_RESPONSE_ONLY_RE.test(trimmed)) return false;
45
65
  if (isPotentialNoResponsePrefix(trimmed)) return false;
46
- return trimmed.replace(NO_RESPONSE_INLINE_RE, "").trim().length > 0;
66
+ return stripNoResponseMarkers(trimmed).length > 0;
47
67
  }
@@ -65,6 +65,7 @@ import {
65
65
  llmRequestLogs,
66
66
  memoryV2ActivationLogs,
67
67
  messages,
68
+ providerConnections,
68
69
  } from "../../../persistence/schema/index.js";
69
70
  import {
70
71
  backfillMemoryV2ActivationMessageId,
@@ -825,6 +826,16 @@ describe("PUT /v1/config/llm/profiles/:name", () => {
825
826
  });
826
827
 
827
828
  test("auto-derives provider_connection when omitted from body (Any active)", async () => {
829
+ // Start from a clean connection slate — provider_connections persists
830
+ // across tests in this file, so a leaked openai-personal would otherwise
831
+ // win the derivation.
832
+ getDb().delete(providerConnections).run();
833
+ // The single Vellum-managed connection serves managed-routable providers.
834
+ createConnection(getDb(), {
835
+ name: "vellum",
836
+ provider: "vellum",
837
+ auth: { type: "platform" },
838
+ });
828
839
  // Seed an existing binding so the test starts from a non-empty state.
829
840
  (
830
841
  rawConfigFixture.llm as {
@@ -851,9 +862,38 @@ describe("PUT /v1/config/llm/profiles/:name", () => {
851
862
  }
852
863
  ).profiles.custom;
853
864
 
854
- // The canonical "openai-managed" connection exists in the test DB;
855
- // the route auto-derives it when the UI omits provider_connection.
856
- expect(savedProfile.provider_connection).toBe("openai-managed");
865
+ // No personal openai connection exists, so the route auto-derives the
866
+ // single Vellum-managed connection for this managed-routable provider.
867
+ expect(savedProfile.provider_connection).toBe("vellum");
868
+ });
869
+
870
+ test("Any active derivation skips orphaned legacy *-managed rows", async () => {
871
+ getDb().delete(providerConnections).run();
872
+ // Upgraded workspaces may still carry a legacy openai-managed row (hidden
873
+ // from the list route, deleted by a follow-up migration). It must not be
874
+ // auto-picked — the derivation should bind to `vellum` instead.
875
+ createConnection(getDb(), {
876
+ name: "openai-managed",
877
+ provider: "openai",
878
+ auth: { type: "platform" },
879
+ });
880
+ createConnection(getDb(), {
881
+ name: "vellum",
882
+ provider: "vellum",
883
+ auth: { type: "platform" },
884
+ });
885
+
886
+ await replaceProfileRoute.handler({
887
+ pathParams: { name: "custom" },
888
+ body: { provider: "openai", model: "gpt-5.5" },
889
+ });
890
+
891
+ const savedProfile = (
892
+ savedRawConfig?.llm as {
893
+ profiles: Record<string, Record<string, unknown>>;
894
+ }
895
+ ).profiles.custom;
896
+ expect(savedProfile.provider_connection).toBe("vellum");
857
897
  });
858
898
 
859
899
  test("auto-derives provider_connection for BYOK provider (Any active)", async () => {