@vellumai/assistant 0.11.7-dev.202609010224.44cd29e → 0.11.7-dev.202609010322.2bc955e

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@vellumai/assistant",
3
- "version": "0.11.7-dev.202609010224.44cd29e",
3
+ "version": "0.11.7-dev.202609010322.2bc955e",
4
4
  "license": "MIT",
5
5
  "type": "module",
6
6
  "exports": {
@@ -175,6 +175,22 @@ describe("AgentLoop — call-site precedence", () => {
175
175
 
176
176
  test("call-site thinking wins over conversation default when callSite is set", async () => {
177
177
  setLlmConfig({
178
+ // Pinned to a model that can actually be told not to think. Without an
179
+ // active profile the call site resolves the shipped Balanced body, and
180
+ // the disabled shape asserted below only exists for models whose
181
+ // reasoning can be turned off: the normalizer drops a disabled thinking
182
+ // config for adaptive-thinking-only models (they always reason and 4xx
183
+ // the opt-out), so this test would assert the Balanced pin's capabilities
184
+ // rather than call-site precedence.
185
+ profiles: {
186
+ "test-anthropic": {
187
+ source: "user",
188
+ provider: "anthropic",
189
+ model: "claude-sonnet-5",
190
+ },
191
+ },
192
+ profileOrder: ["test-anthropic"],
193
+ activeProfile: "test-anthropic",
178
194
  default: {
179
195
  provider: "anthropic",
180
196
  model: "claude-default",
@@ -1118,7 +1118,7 @@ describe("code-owned default profiles — wire view and write normalization", ()
1118
1118
  });
1119
1119
  const response = (await getRoute.handler({})) as Record<string, any>;
1120
1120
  const wireBalanced = response.llm.profiles.balanced;
1121
- expect(wireBalanced.model).toBe("accounts/fireworks/models/glm-5p2");
1121
+ expect(wireBalanced.model).toBe("accounts/fireworks/models/glm-5p3-flash");
1122
1122
  expect(wireBalanced.provider).toBe("vellum");
1123
1123
  expect(wireBalanced.provider_connection).toBeUndefined();
1124
1124
  expect(wireBalanced.status).toBe("disabled");
@@ -1366,7 +1366,7 @@ describe("code-owned default profiles — echoes over stale on-disk bodies", ()
1366
1366
  const response = (await getRoute.handler({})) as Record<string, any>;
1367
1367
  // The wire view serves catalog content, not the stale body.
1368
1368
  expect(response.llm.profiles.balanced.model).toBe(
1369
- "accounts/fireworks/models/glm-5p2",
1369
+ "accounts/fireworks/models/glm-5p3-flash",
1370
1370
  );
1371
1371
  const result = await patchRoute.handler({
1372
1372
  body: { llm: { profiles: response.llm.profiles } },
@@ -24,15 +24,16 @@ import {
24
24
  type ProfileEntry,
25
25
  } from "../schemas/llm.js";
26
26
 
27
- // The `experiment-balanced-model-2026-08-06` arm repoints the model the managed
27
+ // The `experiment-balanced-model-2026-08-31` arm repoints the model the managed
28
28
  // Balanced profile resolves to. Both the runtime resolver and the client-facing
29
29
  // profile listing go through `defaultProfileBodyForProvider`, so both are
30
30
  // exercised here: a disagreement between them shows the user one model beside a
31
31
  // profile that runs another.
32
32
 
33
33
  const FLAG = BALANCED_MODEL_EXPERIMENT_FLAG_KEY;
34
- const SHIPPED_MODEL = "accounts/fireworks/models/glm-5p2";
35
- const GLM_MODEL = SHIPPED_MODEL;
34
+ const SHIPPED_MODEL = "accounts/fireworks/models/glm-5p3-flash";
35
+ const GLM_53_MODEL = "accounts/fireworks/models/glm-5p3";
36
+ const GLM_52_MODEL = "accounts/fireworks/models/glm-5p2";
36
37
 
37
38
  function setArm(value: boolean | string): void {
38
39
  setOverridesForTesting({ [FLAG]: value });
@@ -57,22 +58,22 @@ afterEach(() => {
57
58
  });
58
59
 
59
60
  describe("balanced-model experiment arms", () => {
60
- test("terra repoints mainAgent at gpt-5.6-terra on the managed connection", () => {
61
- setArm("terra");
61
+ test("glm-5p3 repoints mainAgent at GLM 5.3 on the managed connection", () => {
62
+ setArm("glm-5p3");
62
63
  const resolved = resolveCallSiteConfig(
63
64
  "mainAgent",
64
65
  llmWithActiveBalanced(),
65
66
  );
66
- expect(resolved.model).toBe("gpt-5.6-terra");
67
+ expect(resolved.model).toBe(GLM_53_MODEL);
67
68
  expect(resolved.provider).toBe("vellum");
68
- expect(getManagedUpstream("gpt-5.6-terra")).toBe("openai");
69
+ expect(getManagedUpstream(GLM_53_MODEL)).toBe("fireworks");
69
70
 
70
71
  const entry = resolveDefaultProfileForProvider(
71
72
  undefined,
72
73
  "balanced",
73
74
  managed,
74
75
  );
75
- expect(entry?.model).toBe("gpt-5.6-terra");
76
+ expect(entry?.model).toBe(GLM_53_MODEL);
76
77
  expect(entry?.provider).toBe("vellum");
77
78
  // Routing-identity providers derive their connection per request.
78
79
  expect(entry?.provider_connection).toBeUndefined();
@@ -85,12 +86,12 @@ describe("balanced-model experiment arms", () => {
85
86
  "mainAgent",
86
87
  llmWithActiveBalanced(),
87
88
  );
88
- expect(resolved.model).toBe(GLM_MODEL);
89
+ expect(resolved.model).toBe(GLM_52_MODEL);
89
90
  expect(resolved.provider).toBe("vellum");
90
91
 
91
- const upstream = getManagedUpstream(GLM_MODEL);
92
+ const upstream = getManagedUpstream(GLM_52_MODEL);
92
93
  expect(upstream).toBe("fireworks");
93
- const cap = catalogMaxOutputTokens(upstream as string, GLM_MODEL);
94
+ const cap = catalogMaxOutputTokens(upstream as string, GLM_52_MODEL);
94
95
  expect(cap).toBeDefined();
95
96
  expect(resolved.maxTokens).toBeLessThanOrEqual(cap as number);
96
97
  // The shipped token budget stands.
@@ -99,19 +100,26 @@ describe("balanced-model experiment arms", () => {
99
100
  );
100
101
  });
101
102
 
103
+ test("the glm-5p3-flash arm names the shipped pin", () => {
104
+ setArm("glm-5p3-flash");
105
+ expect(
106
+ resolveDefaultProfileForProvider(undefined, "balanced", managed)?.model,
107
+ ).toBe(SHIPPED_MODEL);
108
+ });
109
+
102
110
  test("an arm changes only the model of the managed balanced body", () => {
103
- setArm("terra");
111
+ setArm("glm-5p2");
104
112
  const entry = resolveDefaultProfileForProvider(
105
113
  undefined,
106
114
  "balanced",
107
115
  managed,
108
116
  );
109
117
  const shipped = CODE_DEFAULT_PROFILE_ENTRIES.balanced;
110
- expect(entry).toEqual({ ...shipped, model: "gpt-5.6-terra" });
118
+ expect(entry).toEqual({ ...shipped, model: GLM_52_MODEL });
111
119
  });
112
120
 
113
121
  test("the other default profiles are untouched by an arm", () => {
114
- setArm("terra");
122
+ setArm("glm-5p2");
115
123
  for (const key of [
116
124
  "quality-optimized",
117
125
  "cost-optimized",
@@ -124,7 +132,7 @@ describe("balanced-model experiment arms", () => {
124
132
  });
125
133
 
126
134
  test("every pinned arm model is managed-routable and in its upstream catalog", () => {
127
- for (const model of ["gpt-5.6-terra", GLM_MODEL]) {
135
+ for (const model of [SHIPPED_MODEL, GLM_53_MODEL, GLM_52_MODEL]) {
128
136
  const upstream = getManagedUpstream(model);
129
137
  expect(upstream).not.toBeNull();
130
138
  expect(isModelInCatalog(upstream as string, model)).toBe(true);
@@ -136,7 +144,10 @@ describe("balanced-model experiment fallbacks", () => {
136
144
  const shippedCases: [string, () => void][] = [
137
145
  ["the flag is unset", () => clearCachedOverrides()],
138
146
  ["no override is present", () => setOverridesForTesting({})],
147
+ // Arms of the retired 2026-08-06 experiment are no longer pinned, so a
148
+ // stale value cannot strand an install on a model this test no longer runs.
139
149
  ["the arm is control", () => setArm("control")],
150
+ ["the arm is terra", () => setArm("terra")],
140
151
  ["the arm is an unknown string", () => setArm("gpt-9-does-not-exist")],
141
152
  ["the arm is the empty string", () => setArm("")],
142
153
  ["the value is boolean true", () => setArm(true)],
@@ -162,11 +173,11 @@ describe("balanced-model experiment fallbacks", () => {
162
173
  });
163
174
  }
164
175
 
165
- test("the shipped catalog body is the control arm", () => {
176
+ test("the shipped catalog body is the glm-5p3-flash arm", () => {
166
177
  expect(CODE_DEFAULT_PROFILE_ENTRIES.balanced.model).toBe(SHIPPED_MODEL);
167
178
  });
168
179
 
169
- test("the registry default is control, which the override-only flag read assumes", () => {
180
+ test("the registry default names the shipped pin, which the override-only flag read assumes", () => {
170
181
  const registryPath = join(
171
182
  import.meta.dirname,
172
183
  "..",
@@ -188,14 +199,35 @@ describe("balanced-model experiment fallbacks", () => {
188
199
  const flag = registry.flags.find((entry) => entry.key === FLAG);
189
200
  expect(flag).toBeDefined();
190
201
  expect(flag?.scope).toBe("assistant");
191
- expect(flag?.defaultEnabled).toBe("control");
192
- expect(flag?.values).toEqual(["control", "terra", "glm-5p2"]);
202
+ expect(flag?.defaultEnabled).toBe("glm-5p3-flash");
203
+ expect(flag?.values).toEqual(["glm-5p3-flash", "glm-5p3", "glm-5p2"]);
204
+ });
205
+
206
+ test("the retired experiment's flag is gone from the registry", () => {
207
+ const registryPath = join(
208
+ import.meta.dirname,
209
+ "..",
210
+ "..",
211
+ "..",
212
+ "..",
213
+ "meta",
214
+ "feature-flags",
215
+ "feature-flag-registry.json",
216
+ );
217
+ const registry = JSON.parse(readFileSync(registryPath, "utf-8")) as {
218
+ flags: { key: string }[];
219
+ };
220
+ expect(
221
+ registry.flags.some(
222
+ (entry) => entry.key === "experiment-balanced-model-2026-08-06",
223
+ ),
224
+ ).toBe(false);
193
225
  });
194
226
  });
195
227
 
196
228
  describe("balanced-model experiment boundaries", () => {
197
229
  test("a user-owned balanced shadow wins over the arm", () => {
198
- setArm("terra");
230
+ setArm("glm-5p2");
199
231
  const profiles: Record<string, ProfileEntry> = {
200
232
  balanced: { source: "user", provider: "openai", model: "gpt-5.5" },
201
233
  };
@@ -209,7 +241,7 @@ describe("balanced-model experiment boundaries", () => {
209
241
  });
210
242
 
211
243
  test("a managed-source stub still resolves to the arm's body", () => {
212
- setArm("terra");
244
+ setArm("glm-5p2");
213
245
  const profiles: Record<string, ProfileEntry> = {
214
246
  balanced: { source: "managed", label: "My Balanced" },
215
247
  };
@@ -218,12 +250,12 @@ describe("balanced-model experiment boundaries", () => {
218
250
  "balanced",
219
251
  managed,
220
252
  );
221
- expect(entry?.model).toBe("gpt-5.6-terra");
253
+ expect(entry?.model).toBe(GLM_52_MODEL);
222
254
  expect(entry?.label).toBe("My Balanced");
223
255
  });
224
256
 
225
257
  test("the chatgpt and BYOK columns stay out of the experiment", () => {
226
- setArm("terra");
258
+ setArm("glm-5p2");
227
259
  for (const provider of ["anthropic", "openai", "chatgpt"] as const) {
228
260
  const armed = resolveDefaultProfileForProvider(undefined, "balanced", {
229
261
  provider,
@@ -234,37 +266,37 @@ describe("balanced-model experiment boundaries", () => {
234
266
  });
235
267
  expect(armed).toEqual(shipped as ProfileEntry);
236
268
  expect(armed?.provider).toBe(provider);
237
- setArm("terra");
269
+ setArm("glm-5p2");
238
270
  }
239
271
  });
240
272
 
241
273
  test("an install with no defaultProvider still picks up the arm", () => {
242
- setArm("terra");
274
+ setArm("glm-5p2");
243
275
  expect(
244
276
  resolveDefaultProfileForProvider(undefined, "balanced", null)?.model,
245
- ).toBe("gpt-5.6-terra");
277
+ ).toBe(GLM_52_MODEL);
246
278
  });
247
279
  });
248
280
 
249
281
  describe("client-facing profile listing", () => {
250
282
  test("reports the arm's model so the UI matches what runs", () => {
251
- setArm("terra");
283
+ setArm("glm-5p2");
252
284
  const profiles = getEffectiveProfilesForProvider(undefined, managed);
253
- expect(profiles.balanced?.model).toBe("gpt-5.6-terra");
285
+ expect(profiles.balanced?.model).toBe(GLM_52_MODEL);
254
286
  expect(profiles["quality-optimized"]?.model).toBe(
255
287
  CODE_DEFAULT_PROFILE_ENTRIES["quality-optimized"].model,
256
288
  );
257
289
  });
258
290
 
259
- test("reports the shipped model on the control arm", () => {
260
- setArm("control");
291
+ test("reports the shipped model on an arm this build does not know", () => {
292
+ setArm("terra");
261
293
  expect(
262
294
  getEffectiveProfilesForProvider(undefined, managed).balanced?.model,
263
295
  ).toBe(SHIPPED_MODEL);
264
296
  });
265
297
 
266
298
  test("agrees with the runtime resolver on every arm", () => {
267
- for (const arm of ["control", "terra", "glm-5p2", "nonsense"]) {
299
+ for (const arm of ["glm-5p3-flash", "glm-5p3", "glm-5p2", "nonsense"]) {
268
300
  setArm(arm);
269
301
  const listed = getEffectiveProfilesForProvider(undefined, managed)
270
302
  .balanced?.model;
@@ -59,9 +59,9 @@ describe("getEffectiveProfiles", () => {
59
59
  }
60
60
  });
61
61
 
62
- test("the managed Balanced profile routes GLM 5.2 through Fireworks", () => {
62
+ test("the managed Balanced profile routes GLM 5.3 Flash through Fireworks", () => {
63
63
  const balanced = CODE_DEFAULT_PROFILE_ENTRIES.balanced;
64
- expect(balanced.model).toBe("accounts/fireworks/models/glm-5p2");
64
+ expect(balanced.model).toBe("accounts/fireworks/models/glm-5p3-flash");
65
65
  expect(resolveRoutingIdentity(balanced.provider, balanced.model)).toEqual({
66
66
  connectionName: "vellum",
67
67
  expectedProvider: "fireworks",
@@ -1,33 +1,39 @@
1
1
  /**
2
2
  * Flag read for the managed Balanced profile's model A/B test.
3
3
  *
4
- * `experiment-balanced-model-2026-08-06` is a multivariate LaunchDarkly flag
4
+ * `experiment-balanced-model-2026-08-31` is a multivariate LaunchDarkly flag
5
5
  * whose arm repoints the model the managed (`vellum`) implementation of the
6
6
  * `balanced` default profile resolves to. The arm to model pins live beside
7
7
  * the profile bodies in `default-profile-catalog.ts`, which validates their
8
8
  * managed routability at load time; this module owns only the flag read.
9
9
  *
10
+ * A flag key per experiment, rather than new variations on the previous one,
11
+ * is what keeps a readout honest: the activation cohort query keys a user on
12
+ * their first-ever assignment for a flag, so every user the 2026-08-06 test
13
+ * already stamped would stay pinned to that arm forever and never enter a new
14
+ * arm on the same key.
15
+ *
10
16
  * The read goes straight to the override cache rather than through
11
17
  * `assistant-feature-flags.ts`. `feature-flag-cache.ts` is a stdlib-only leaf,
12
18
  * so the profile catalog stays free of the pino logger and the gateway IPC
13
19
  * client that resolver would pull into its import graph, and the registry
14
- * declares `defaultEnabled: "control"` for this flag, which is the same
15
- * shipped body an absent override already resolves to. That equivalence is
16
- * pinned by `__tests__/balanced-model-experiment.test.ts` so it cannot drift.
20
+ * declares `defaultEnabled: "glm-5p3-flash"` for this flag, which pins the same
21
+ * model the shipped body an absent override resolves to already carries. That
22
+ * equivalence is pinned by `__tests__/balanced-model-experiment.test.ts` so it
23
+ * cannot drift.
17
24
  */
18
25
 
19
26
  import { getCachedOverrides } from "./feature-flag-cache.js";
20
27
 
21
28
  export const BALANCED_MODEL_EXPERIMENT_FLAG_KEY =
22
- "experiment-balanced-model-2026-08-06";
29
+ "experiment-balanced-model-2026-08-31";
23
30
 
24
31
  /**
25
32
  * The experiment arm in force, or `undefined` when the flag resolves to
26
33
  * anything that is not an arm name: unset, a boolean, or the empty string.
27
- * `control` and any arm this build does not know are returned as-is and miss
28
- * the pin table, which is what keeps a stale or malformed LaunchDarkly value
29
- * on the shipped model rather than stranding an install on one that does not
30
- * exist.
34
+ * Any arm this build does not know is returned as-is and misses the pin table,
35
+ * which is what keeps a stale or malformed LaunchDarkly value on the shipped
36
+ * model rather than stranding an install on one that does not exist.
31
37
  */
32
38
  export function getBalancedModelExperimentArm(): string | undefined {
33
39
  const value = getCachedOverrides()?.[BALANCED_MODEL_EXPERIMENT_FLAG_KEY];
@@ -86,7 +86,7 @@ type ProfileImpls = Record<DefaultProfileKey, DefaultProfileTemplate>;
86
86
  */
87
87
  const VELLUM_PROFILE_IMPLS: ProfileImpls = {
88
88
  balanced: {
89
- model: "accounts/fireworks/models/glm-5p2",
89
+ model: "accounts/fireworks/models/glm-5p3-flash",
90
90
  provider: "vellum",
91
91
  source: "managed",
92
92
  label: "Balanced",
@@ -241,25 +241,26 @@ const BACKUP_PROFILE_IMPLS: Record<BackupProfileKey, DefaultProfileTemplate> = {
241
241
  };
242
242
 
243
243
  /**
244
- * Arm to managed model pin for the `experiment-balanced-model-2026-08-06` A/B
244
+ * Arm to managed model pin for the `experiment-balanced-model-2026-08-31` A/B
245
245
  * test (`balanced-model-experiment.ts` owns the flag read). An arm repoints
246
246
  * the model of the managed (`vellum`) implementation of `balanced` and nothing
247
247
  * else: effort, thinking, token budget, label and description all stay on the
248
248
  * shipped body, and the `chatgpt` and BYOK columns are untouched because those
249
249
  * installs run the provider their user chose and sit outside the experiment.
250
250
  *
251
- * `control` is absent by design. It, an arm this build does not know, and an
252
- * unset flag all resolve to the shipped body, so no LaunchDarkly value can
253
- * strand an install on a model that is not pinned here. A `Map` rather than an
254
- * object literal keeps that true for every string LaunchDarkly can send: the
255
- * arm is remote input, and an object lookup would resolve `constructor` or
256
- * `toString` to an inherited `Object.prototype` member instead of missing.
251
+ * An arm this build does not know, and an unset flag, both resolve to the
252
+ * shipped body, so no LaunchDarkly value can strand an install on a model that
253
+ * is not pinned here. A `Map` rather than an object literal keeps that true for
254
+ * every string LaunchDarkly can send: the arm is remote input, and an object
255
+ * lookup would resolve `constructor` or `toString` to an inherited
256
+ * `Object.prototype` member instead of missing.
257
257
  *
258
- * `glm-5p2` names the same model as the shipped pin and stays in the table so
259
- * the arm keeps its meaning if the shipped pin moves again.
258
+ * `glm-5p3-flash` names the same model as the shipped pin and stays in the
259
+ * table so the arm keeps its meaning if the shipped pin moves again.
260
260
  */
261
261
  const BALANCED_EXPERIMENT_MODELS = new Map<string, string>([
262
- ["terra", "gpt-5.6-terra"],
262
+ ["glm-5p3-flash", "accounts/fireworks/models/glm-5p3-flash"],
263
+ ["glm-5p3", "accounts/fireworks/models/glm-5p3"],
263
264
  ["glm-5p2", "accounts/fireworks/models/glm-5p2"],
264
265
  ]);
265
266
 
@@ -382,13 +382,13 @@
382
382
  "defaultEnabled": true
383
383
  },
384
384
  {
385
- "id": "experiment-balanced-model-2026-08-06",
385
+ "id": "experiment-balanced-model-2026-08-31",
386
386
  "scope": "assistant",
387
- "key": "experiment-balanced-model-2026-08-06",
387
+ "key": "experiment-balanced-model-2026-08-31",
388
388
  "label": "Experiment: Balanced Profile Model",
389
- "description": "Multivariate experiment on the model the managed Balanced inference profile resolves to. control = the shipped pin (accounts/fireworks/models/glm-5p2); terra = gpt-5.6-terra; glm-5p2 = accounts/fireworks/models/glm-5p2. Only the managed (vellum) implementation moves: ChatGPT-subscription and BYOK installs run the provider the user chose and are outside the experiment. Rollout and targeting are managed in the LaunchDarkly dashboard; any value that is not an arm name resolves to control.",
390
- "defaultEnabled": "control",
391
- "values": ["control", "terra", "glm-5p2"]
389
+ "description": "Multivariate experiment on the model the managed Balanced inference profile resolves to. glm-5p3-flash = the shipped pin (accounts/fireworks/models/glm-5p3-flash); glm-5p3 = accounts/fireworks/models/glm-5p3; glm-5p2 = accounts/fireworks/models/glm-5p2. Replaces experiment-balanced-model-2026-08-06, which ended on 2026-08-31: the activation cohort keys a user on their first-ever assignment per flag, so a new arm on the old key could never enrol a user the old test already stamped. Only the managed (vellum) implementation moves: ChatGPT-subscription and BYOK installs run the provider the user chose and are outside the experiment. Rollout and targeting are managed in the LaunchDarkly dashboard; any value that is not an arm name resolves to the shipped pin.",
390
+ "defaultEnabled": "glm-5p3-flash",
391
+ "values": ["glm-5p3-flash", "glm-5p3", "glm-5p2"]
392
392
  },
393
393
  {
394
394
  "id": "ios-avatar-app-icon",