@vellumai/assistant 0.8.9-dev.202606091853.fbaa2ae → 0.8.9-dev.202606091926.ebb2d62

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/openapi.yaml CHANGED
@@ -15945,6 +15945,8 @@ paths:
15945
15945
  - assistant
15946
15946
  content:
15947
15947
  type: string
15948
+ deprecated: true
15949
+ description: "Deprecated: superseded by contentBlocks. Flat plain-text body (joined text segments)."
15948
15950
  timestamp:
15949
15951
  type: string
15950
15952
  attachments:
@@ -16304,14 +16306,22 @@ paths:
16304
16306
  type: array
16305
16307
  items:
16306
16308
  type: string
16309
+ deprecated: true
16310
+ description: "Deprecated: superseded by contentBlocks. Text segments split by tool-call boundaries."
16307
16311
  thinkingSegments:
16308
16312
  type: array
16309
16313
  items:
16310
16314
  type: string
16315
+ deprecated: true
16316
+ description: "Deprecated: superseded by contentBlocks. Reasoning text extracted from thinking blocks."
16311
16317
  contentOrder:
16312
16318
  type: array
16313
16319
  items:
16314
16320
  type: string
16321
+ deprecated: true
16322
+ description:
16323
+ 'Deprecated: superseded by contentBlocks. Positional "<type>:<index>" content ordering (e.g. "text:0",
16324
+ "thinking:1").'
16315
16325
  contentBlocks:
16316
16326
  type: array
16317
16327
  items:
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@vellumai/assistant",
3
- "version": "0.8.9-dev.202606091853.fbaa2ae",
3
+ "version": "0.8.9-dev.202606091926.ebb2d62",
4
4
  "license": "MIT",
5
5
  "type": "module",
6
6
  "exports": {
@@ -74,6 +74,8 @@ function projectModel(model: CatalogModel): Record<string, unknown> {
74
74
  projected.longContextMode = model.longContextMode;
75
75
  if (model.supportsThinking !== undefined)
76
76
  projected.supportsThinking = model.supportsThinking;
77
+ if (model.adaptiveThinkingOnly !== undefined)
78
+ projected.adaptiveThinkingOnly = model.adaptiveThinkingOnly;
77
79
  if (model.supportsCaching !== undefined)
78
80
  projected.supportsCaching = model.supportsCaching;
79
81
  if (model.supportsVision !== undefined)
@@ -54,7 +54,10 @@ mock.module("../memory/llm-request-log-store.js", () => ({
54
54
  }));
55
55
 
56
56
  // ── Imports (after mocks) ─────────────────────────────────────────────────────
57
- import { buildPersistedAssistantContent } from "../daemon/conversation-agent-loop-handlers.js";
57
+ import {
58
+ buildPersistedAssistantContent,
59
+ stampThinkingTiming,
60
+ } from "../daemon/conversation-agent-loop-handlers.js";
58
61
  import type { ToolActivityMetadata } from "../daemon/message-types/web-activity.js";
59
62
  import type { ContentBlock } from "../providers/types.js";
60
63
 
@@ -182,3 +185,74 @@ describe("buildPersistedAssistantContent — native activityMetadata", () => {
182
185
  expect(block._activityMetadata).toBeUndefined();
183
186
  });
184
187
  });
188
+
189
+ describe("stampThinkingTiming", () => {
190
+ test("stamps internal timing onto thinking blocks by position", () => {
191
+ // GIVEN a turn that interleaves text and two thinking blocks AND the
192
+ // per-block timing captured while streaming (one entry per thinking block,
193
+ // in stream order)
194
+ const content = [
195
+ { type: "thinking", thinking: "first", signature: "s1" },
196
+ { type: "text", text: "answer" },
197
+ { type: "thinking", thinking: "second", signature: "s2" },
198
+ ] as unknown as ContentBlock[];
199
+ const timings = [
200
+ { startedAt: 100, completedAt: 250 },
201
+ { startedAt: 400, completedAt: 480 },
202
+ ];
203
+
204
+ // WHEN the content is stamped before persistence
205
+ const stamped = stampThinkingTiming(content, timings) as unknown as Array<
206
+ Record<string, unknown>
207
+ >;
208
+
209
+ // THEN each thinking block carries the `_`-prefixed timing for its position
210
+ expect(stamped[0]).toMatchObject({
211
+ type: "thinking",
212
+ _startedAt: 100,
213
+ _completedAt: 250,
214
+ });
215
+ expect(stamped[2]).toMatchObject({
216
+ type: "thinking",
217
+ _startedAt: 400,
218
+ _completedAt: 480,
219
+ });
220
+ // AND the interleaved text block is left untouched
221
+ expect(stamped[1]).toEqual({ type: "text", text: "answer" });
222
+ });
223
+
224
+ test("leaves thinking blocks unstamped when no timing was captured", () => {
225
+ // GIVEN thinking content but an empty timing list (thinking streaming was
226
+ // disabled, so no per-block timing was recorded this turn)
227
+ const content = [
228
+ { type: "thinking", thinking: "first", signature: "s1" },
229
+ ] as unknown as ContentBlock[];
230
+
231
+ // WHEN the content is stamped with no timing
232
+ const stamped = stampThinkingTiming(content, []);
233
+
234
+ // THEN the original content is returned unchanged so the UI hides duration,
235
+ // exactly as a tool call with no timing
236
+ expect(stamped).toBe(content);
237
+ expect(stamped[0]).not.toHaveProperty("_startedAt");
238
+ });
239
+
240
+ test("stamps only the thinking blocks that have a matching timing entry", () => {
241
+ // GIVEN two thinking blocks but only one captured timing entry (e.g. the
242
+ // second block opened after the timing array was already finalized)
243
+ const content = [
244
+ { type: "thinking", thinking: "first", signature: "s1" },
245
+ { type: "thinking", thinking: "second", signature: "s2" },
246
+ ] as unknown as ContentBlock[];
247
+ const timings = [{ startedAt: 100, completedAt: 250 }];
248
+
249
+ // WHEN the content is stamped
250
+ const stamped = stampThinkingTiming(content, timings) as unknown as Array<
251
+ Record<string, unknown>
252
+ >;
253
+
254
+ // THEN the first block is stamped and the unmatched second block is left as-is
255
+ expect(stamped[0]).toMatchObject({ _startedAt: 100, _completedAt: 250 });
256
+ expect(stamped[1]).not.toHaveProperty("_startedAt");
257
+ });
258
+ });
@@ -57,17 +57,8 @@ mock.module("../util/logger.js", () => ({
57
57
  getLogger: () => makeLoggerStub(),
58
58
  }));
59
59
 
60
- // ---------------------------------------------------------------------------
61
- // Feature flag mock — controls whether managed-gemini-embeddings-enabled is on
62
- // ---------------------------------------------------------------------------
63
-
64
- let featureFlagEnabled = false;
65
-
66
60
  mock.module("../config/assistant-feature-flags.js", () => ({
67
- isAssistantFeatureFlagEnabled: (key: string) => {
68
- if (key === "managed-gemini-embeddings-enabled") return featureFlagEnabled;
69
- return true;
70
- },
61
+ isAssistantFeatureFlagEnabled: () => true,
71
62
  clearFeatureFlagOverridesCache: () => {},
72
63
  initFeatureFlagOverrides: async () => {},
73
64
  getAssistantFeatureFlagDefaults: () => ({}),
@@ -119,8 +110,6 @@ describe("managed Gemini embedding defaults (via loadConfig)", () => {
119
110
  setStorePathForTesting(join(WORKSPACE_DIR, "keys.enc"));
120
111
  invalidateConfigCache();
121
112
 
122
- // Reset mock state
123
- featureFlagEnabled = false;
124
113
  originalIsPlatform = process.env.IS_PLATFORM;
125
114
  delete process.env.IS_PLATFORM;
126
115
  });
@@ -137,10 +126,9 @@ describe("managed Gemini embedding defaults (via loadConfig)", () => {
137
126
  }
138
127
  });
139
128
 
140
- test("applies managed Gemini defaults when FF on + IS_PLATFORM + provider auto", () => {
129
+ test("applies managed Gemini defaults when IS_PLATFORM + provider auto", () => {
141
130
  writeConfig({});
142
131
 
143
- featureFlagEnabled = true;
144
132
  process.env.IS_PLATFORM = "true";
145
133
 
146
134
  const config = loadConfig();
@@ -162,22 +150,9 @@ describe("managed Gemini embedding defaults (via loadConfig)", () => {
162
150
  expect(qdrantRaw.vectorSize).toBe(3072);
163
151
  });
164
152
 
165
- test("does NOT apply when feature flag is OFF", () => {
166
- writeConfig({});
167
-
168
- featureFlagEnabled = false;
169
- process.env.IS_PLATFORM = "true";
170
-
171
- const config = loadConfig();
172
-
173
- expect(config.memory.embeddings.provider).toBe("auto");
174
- expect(config.memory.qdrant.vectorSize).toBe(384);
175
- });
176
-
177
153
  test("does NOT apply when IS_PLATFORM is not set", () => {
178
154
  writeConfig({});
179
155
 
180
- featureFlagEnabled = true;
181
156
  delete process.env.IS_PLATFORM;
182
157
 
183
158
  const config = loadConfig();
@@ -191,7 +166,6 @@ describe("managed Gemini embedding defaults (via loadConfig)", () => {
191
166
  memory: { embeddings: { provider: "local" } },
192
167
  });
193
168
 
194
- featureFlagEnabled = true;
195
169
  process.env.IS_PLATFORM = "true";
196
170
 
197
171
  const config = loadConfig();
@@ -204,7 +178,6 @@ describe("managed Gemini embedding defaults (via loadConfig)", () => {
204
178
  memory: { embeddings: { provider: "openai" } },
205
179
  });
206
180
 
207
- featureFlagEnabled = true;
208
181
  process.env.IS_PLATFORM = "true";
209
182
 
210
183
  const config = loadConfig();
@@ -219,7 +192,6 @@ describe("managed Gemini embedding defaults (via loadConfig)", () => {
219
192
  },
220
193
  });
221
194
 
222
- featureFlagEnabled = true;
223
195
  process.env.IS_PLATFORM = "true";
224
196
 
225
197
  const config = loadConfig();
@@ -235,7 +207,6 @@ describe("managed Gemini embedding defaults (via loadConfig)", () => {
235
207
  memory: { embeddings: { provider: "ollama" } },
236
208
  });
237
209
 
238
- featureFlagEnabled = true;
239
210
  process.env.IS_PLATFORM = "true";
240
211
 
241
212
  const config = loadConfig();
@@ -245,7 +216,6 @@ describe("managed Gemini embedding defaults (via loadConfig)", () => {
245
216
  test("is idempotent — second loadConfig is a no-op after migration", () => {
246
217
  writeConfig({});
247
218
 
248
- featureFlagEnabled = true;
249
219
  process.env.IS_PLATFORM = "true";
250
220
 
251
221
  const config = loadConfig();
@@ -274,7 +244,6 @@ describe("managed Gemini embedding defaults (via loadConfig)", () => {
274
244
  },
275
245
  });
276
246
 
277
- featureFlagEnabled = true;
278
247
  process.env.IS_PLATFORM = "true";
279
248
 
280
249
  const config = loadConfig();
@@ -295,22 +264,9 @@ describe("managed Gemini embedding defaults (via loadConfig)", () => {
295
264
  expect(qdrantRaw.onDisk).toBe(false);
296
265
  });
297
266
 
298
- test("does NOT apply when both FF off and IS_PLATFORM not set", () => {
299
- writeConfig({});
300
-
301
- featureFlagEnabled = false;
302
- delete process.env.IS_PLATFORM;
303
-
304
- const config = loadConfig();
305
-
306
- expect(config.memory.embeddings.provider).toBe("auto");
307
- expect(config.memory.qdrant.vectorSize).toBe(384);
308
- });
309
-
310
267
  test("applies when IS_PLATFORM is '1'", () => {
311
268
  writeConfig({});
312
269
 
313
- featureFlagEnabled = true;
314
270
  process.env.IS_PLATFORM = "1";
315
271
 
316
272
  const config = loadConfig();
@@ -165,10 +165,11 @@ describe("CES flags do not affect unrelated flags", () => {
165
165
  setOverridesForTesting(overrides);
166
166
  const config = makeConfig();
167
167
 
168
- // account-deletion defaults to true in the registry and should stay true.
169
- expect(isAssistantFeatureFlagEnabled("account-deletion", config)).toBe(
170
- true,
171
- );
168
+ // Flags with defaultEnabled: true in the registry should stay true
169
+ // regardless of CES overrides.
170
+ expect(
171
+ isAssistantFeatureFlagEnabled("platform-features-in-local-mode", config),
172
+ ).toBe(true);
172
173
  });
173
174
 
174
175
  test("enabling all CES flags does not change unrelated fail-closed flags", () => {
@@ -2,8 +2,7 @@
2
2
  * Tests for managed proxy Gemini embedding backend selection.
3
3
  *
4
4
  * Verifies that selectEmbeddingBackend correctly routes through the
5
- * managed proxy when the feature flag is enabled and managed proxy
6
- * prerequisites are satisfied.
5
+ * managed proxy when managed proxy prerequisites are satisfied.
7
6
  */
8
7
 
9
8
  import { afterEach, beforeEach, describe, expect, mock, test } from "bun:test";
@@ -42,13 +41,9 @@ mock.module("../security/secure-keys.js", () => ({
42
41
  },
43
42
  }));
44
43
 
45
- // Feature flag mock
46
- const mockFeatureFlags: Record<string, boolean> = {};
47
-
44
+ // Feature flag mock — always returns true (flag is GA'ed)
48
45
  mock.module("../config/assistant-feature-flags.js", () => ({
49
- isAssistantFeatureFlagEnabled: (key: string, _config: unknown) => {
50
- return mockFeatureFlags[key] ?? false;
51
- },
46
+ isAssistantFeatureFlagEnabled: () => true,
52
47
  }));
53
48
 
54
49
  import type { AssistantConfig } from "../config/types.js";
@@ -75,14 +70,6 @@ function disableManagedProxy() {
75
70
  mockAssistantApiKey = null;
76
71
  }
77
72
 
78
- function enableFlag() {
79
- mockFeatureFlags["managed-gemini-embeddings-enabled"] = true;
80
- }
81
-
82
- function disableFlag() {
83
- mockFeatureFlags["managed-gemini-embeddings-enabled"] = false;
84
- }
85
-
86
73
  function makeConfig(
87
74
  overrides: {
88
75
  provider?: string;
@@ -116,7 +103,6 @@ function makeConfig(
116
103
 
117
104
  beforeEach(() => {
118
105
  disableManagedProxy();
119
- disableFlag();
120
106
  mockProviderKeys = {};
121
107
  clearEmbeddingBackendCache();
122
108
  });
@@ -126,9 +112,8 @@ afterEach(() => {
126
112
  });
127
113
 
128
114
  describe("managed proxy Gemini embedding selection", () => {
129
- test("selects managed proxy Gemini when flag enabled and proxy context available", async () => {
115
+ test("selects managed proxy Gemini when proxy context available", async () => {
130
116
  enableManagedProxy();
131
- enableFlag();
132
117
  const config = makeConfig();
133
118
 
134
119
  const { backend, reason } = await selectEmbeddingBackend(config);
@@ -141,7 +126,6 @@ describe("managed proxy Gemini embedding selection", () => {
141
126
 
142
127
  test("managed proxy backend uses default 3072 dimensions when geminiDimensions not set", async () => {
143
128
  enableManagedProxy();
144
- enableFlag();
145
129
  const config = makeConfig();
146
130
 
147
131
  const { backend } = await selectEmbeddingBackend(config);
@@ -155,7 +139,6 @@ describe("managed proxy Gemini embedding selection", () => {
155
139
 
156
140
  test("managed proxy backend uses explicit geminiDimensions when set", async () => {
157
141
  enableManagedProxy();
158
- enableFlag();
159
142
  const config = makeConfig({ geminiDimensions: 768 });
160
143
 
161
144
  const { backend } = await selectEmbeddingBackend(config);
@@ -168,7 +151,6 @@ describe("managed proxy Gemini embedding selection", () => {
168
151
 
169
152
  test("managed proxy backend uses managedBaseUrl (not direct Google API)", async () => {
170
153
  enableManagedProxy();
171
- enableFlag();
172
154
  const config = makeConfig();
173
155
 
174
156
  const { backend } = await selectEmbeddingBackend(config);
@@ -179,32 +161,19 @@ describe("managed proxy Gemini embedding selection", () => {
179
161
  expect(managedBaseUrl).toBe(`${PLATFORM_BASE}/v1/runtime-proxy/gemini`);
180
162
  });
181
163
 
182
- test("falls back to local when flag is disabled (no managed proxy)", async () => {
183
- enableManagedProxy();
184
- disableFlag();
185
- const config = makeConfig();
186
-
187
- const { backend } = await selectEmbeddingBackend(config);
188
-
189
- // With auto and no provider keys, falls through to local
190
- expect(backend).not.toBeNull();
191
- expect(backend!.provider).toBe("local");
192
- });
193
-
194
164
  test("falls back to local when managed proxy context unavailable", async () => {
195
165
  disableManagedProxy();
196
- enableFlag();
197
166
  const config = makeConfig();
198
167
 
199
168
  const { backend } = await selectEmbeddingBackend(config);
200
169
 
170
+ // With auto and no provider keys or proxy, falls through to local
201
171
  expect(backend).not.toBeNull();
202
172
  expect(backend!.provider).toBe("local");
203
173
  });
204
174
 
205
175
  test("selects managed proxy when provider is explicitly gemini", async () => {
206
176
  enableManagedProxy();
207
- enableFlag();
208
177
  const config = makeConfig({ provider: "gemini" });
209
178
 
210
179
  const { backend } = await selectEmbeddingBackend(config);
@@ -217,7 +186,6 @@ describe("managed proxy Gemini embedding selection", () => {
217
186
 
218
187
  test("does not use managed proxy when provider is explicitly local", async () => {
219
188
  enableManagedProxy();
220
- enableFlag();
221
189
  const config = makeConfig({ provider: "local" });
222
190
 
223
191
  const { backend } = await selectEmbeddingBackend(config);
@@ -228,7 +196,6 @@ describe("managed proxy Gemini embedding selection", () => {
228
196
 
229
197
  test("does not use managed proxy when provider is explicitly openai", async () => {
230
198
  enableManagedProxy();
231
- enableFlag();
232
199
  mockProviderKeys[credentialKey("openai", "api_key")] = "user-openai-key";
233
200
  const config = makeConfig({ provider: "openai" });
234
201
 
@@ -238,9 +205,8 @@ describe("managed proxy Gemini embedding selection", () => {
238
205
  expect(backend!.provider).toBe("openai");
239
206
  });
240
207
 
241
- test("direct Gemini key still works when flag is off", async () => {
208
+ test("direct Gemini key still works without managed proxy", async () => {
242
209
  disableManagedProxy();
243
- disableFlag();
244
210
  mockProviderKeys[credentialKey("gemini", "api_key")] = "user-gemini-key";
245
211
  const config = makeConfig({ provider: "gemini" });
246
212
 
@@ -53,6 +53,7 @@ interface ClientCatalogModel {
53
53
  longContextPricingThresholdTokens?: number;
54
54
  longContextMode?: "native-model" | "provider-request-option" | "unsupported";
55
55
  supportsThinking?: boolean;
56
+ adaptiveThinkingOnly?: boolean;
56
57
  supportsCaching?: boolean;
57
58
  supportsVision?: boolean;
58
59
  supportsToolUse?: boolean;
@@ -194,6 +195,9 @@ describe("LLM catalog parity: daemon vs client", () => {
194
195
  );
195
196
  expect(clientModel.longContextMode).toBe(daemonModel.longContextMode);
196
197
  expect(clientModel.supportsThinking).toBe(daemonModel.supportsThinking);
198
+ expect(clientModel.adaptiveThinkingOnly).toBe(
199
+ daemonModel.adaptiveThinkingOnly,
200
+ );
197
201
  expect(clientModel.supportsCaching).toBe(daemonModel.supportsCaching);
198
202
  expect(clientModel.supportsVision).toBe(daemonModel.supportsVision);
199
203
  expect(clientModel.supportsToolUse).toBe(daemonModel.supportsToolUse);
@@ -370,3 +370,81 @@ describe("persistence-layer secret redaction", () => {
370
370
  expect(textBlock?.text).toBe(text);
371
371
  });
372
372
  });
373
+
374
+ describe("thinking timing persistence", () => {
375
+ let state: EventHandlerState;
376
+
377
+ beforeEach(() => {
378
+ addMessageCalls.length = 0;
379
+ state = createEventHandlerState();
380
+ state.turnStartedAt = 1_700_000_000_000;
381
+ });
382
+
383
+ afterEach(() => {
384
+ addMessageCalls.length = 0;
385
+ });
386
+
387
+ function makeThinkingCompleteEvent(): Extract<
388
+ AgentEvent,
389
+ { type: "message_complete" }
390
+ > {
391
+ return {
392
+ type: "message_complete",
393
+ message: {
394
+ role: "assistant",
395
+ content: [
396
+ { type: "thinking", thinking: "let me reason", signature: "sig" },
397
+ { type: "text", text: "the answer" },
398
+ ],
399
+ },
400
+ };
401
+ }
402
+
403
+ test("stamps per-block thinking timing onto the persisted thinking block", async () => {
404
+ // GIVEN a turn whose row was reserved at llm_call_started
405
+ await handleLlmCallStarted(state, makeDeps());
406
+
407
+ // AND streaming then captured timing for one thinking block (startedAt when
408
+ // it opened, completedAt at its last reasoning delta). The reserve resets
409
+ // the accumulator, so deltas — and thus the timing — land after it.
410
+ state.currentThinkingTimestamps = [{ startedAt: 1000, completedAt: 1750 }];
411
+
412
+ // WHEN the message completes and content is persisted
413
+ await handleMessageComplete(state, makeDeps(), makeThinkingCompleteEvent());
414
+
415
+ // THEN the persisted thinking block carries the internal `_`-prefixed
416
+ // timing so a history reload can surface the duration + "Started at" hover
417
+ const persisted = lastPersisted("assistant");
418
+ const blocks = JSON.parse(persisted.content) as Array<{
419
+ type: string;
420
+ _startedAt?: number;
421
+ _completedAt?: number;
422
+ }>;
423
+ const thinkingBlock = blocks.find((b) => b.type === "thinking");
424
+ expect(thinkingBlock?._startedAt).toBe(1000);
425
+ expect(thinkingBlock?._completedAt).toBe(1750);
426
+ });
427
+
428
+ test("persists no thinking timing when none was captured this turn", async () => {
429
+ // GIVEN a turn that produced a thinking block but captured no timing
430
+ // (thinking streaming disabled, so the timing list stays empty)
431
+ expect(state.currentThinkingTimestamps).toEqual([]);
432
+
433
+ // WHEN the message completes and content is persisted
434
+ await handleLlmCallStarted(state, makeDeps());
435
+ await handleMessageComplete(state, makeDeps(), makeThinkingCompleteEvent());
436
+
437
+ // THEN the thinking block is persisted without timing, so the client hides
438
+ // the duration exactly as a tool call with no timing
439
+ const persisted = lastPersisted("assistant");
440
+ const blocks = JSON.parse(persisted.content) as Array<{
441
+ type: string;
442
+ _startedAt?: number;
443
+ _completedAt?: number;
444
+ }>;
445
+ const thinkingBlock = blocks.find((b) => b.type === "thinking");
446
+ expect(thinkingBlock).toBeDefined();
447
+ expect(thinkingBlock?._startedAt).toBeUndefined();
448
+ expect(thinkingBlock?._completedAt).toBeUndefined();
449
+ });
450
+ });