@vellumai/assistant 0.11.10-staging.1 → 0.11.10-staging.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@vellumai/assistant",
3
- "version": "0.11.10-staging.1",
3
+ "version": "0.11.10-staging.2",
4
4
  "license": "MIT",
5
5
  "type": "module",
6
6
  "exports": {
@@ -412,12 +412,17 @@ describe("OpenAIResponsesProvider explicit prompt caching (GPT-5.6+)", () => {
412
412
  expect(JSON.stringify(messages)).not.toContain("prompt_cache_breakpoint");
413
413
  });
414
414
 
415
- test("catalog flags exactly the GPT-5.6 direct-openai rows", () => {
415
+ test("catalog flags GPT-5.6 and GPT-6 Astra direct-openai rows", () => {
416
416
  const openai = PROVIDER_CATALOG.find((p) => p.id === "openai");
417
417
  const flagged = (openai?.models ?? [])
418
418
  .filter((m) => m.supportsPromptCacheBreakpoints)
419
419
  .map((m) => m.id)
420
420
  .sort();
421
- expect(flagged).toEqual(["gpt-5.6-luna", "gpt-5.6-sol", "gpt-5.6-terra"]);
421
+ expect(flagged).toEqual([
422
+ "gpt-5.6-luna",
423
+ "gpt-5.6-sol",
424
+ "gpt-5.6-terra",
425
+ "gpt-6-astra",
426
+ ]);
422
427
  });
423
428
  });
@@ -700,6 +700,36 @@ describe("OpenAIResponsesProvider", () => {
700
700
  });
701
701
  });
702
702
 
703
+ test('GPT-6 Astra effort: "max" is sent as reasoning.effort "max"', async () => {
704
+ const astraProvider = new OpenAIResponsesProvider("sk-test", "gpt-6-astra");
705
+ fakeStreamEvents = [textDeltaEvent("OK"), completedEvent(10, 2)];
706
+
707
+ await astraProvider.sendMessage(
708
+ [{ role: "user", content: [{ type: "text", text: "Hi" }] }],
709
+ { config: { effort: "max" } },
710
+ );
711
+
712
+ expect(lastStreamParams!.reasoning).toEqual({
713
+ effort: "max",
714
+ summary: "auto",
715
+ });
716
+ });
717
+
718
+ test('GPT-6 Astra effort: "none" snaps to the lowest accepted value', async () => {
719
+ const astraProvider = new OpenAIResponsesProvider("sk-test", "gpt-6-astra");
720
+ fakeStreamEvents = [textDeltaEvent("OK"), completedEvent(10, 2)];
721
+
722
+ await astraProvider.sendMessage(
723
+ [{ role: "user", content: [{ type: "text", text: "Hi" }] }],
724
+ { config: { effort: "none" } },
725
+ );
726
+
727
+ expect(lastStreamParams!.reasoning).toEqual({
728
+ effort: "low",
729
+ summary: "auto",
730
+ });
731
+ });
732
+
703
733
  test("no effort config means no reasoning in params", async () => {
704
734
  fakeStreamEvents = [textDeltaEvent("OK"), completedEvent(10, 2)];
705
735
 
@@ -845,6 +875,18 @@ describe("OpenAIResponsesProvider", () => {
845
875
  expect(lastStreamParams!.text).toEqual({ verbosity: "high" });
846
876
  });
847
877
 
878
+ test("verbosity is forwarded for GPT-6 Astra", async () => {
879
+ const astraProvider = new OpenAIResponsesProvider("sk-test", "gpt-6-astra");
880
+ fakeStreamEvents = [textDeltaEvent("OK"), completedEvent(10, 2)];
881
+
882
+ await astraProvider.sendMessage(
883
+ [{ role: "user", content: [{ type: "text", text: "Hi" }] }],
884
+ { config: { verbosity: "low" } },
885
+ );
886
+
887
+ expect(lastStreamParams!.text).toEqual({ verbosity: "low" });
888
+ });
889
+
848
890
  test("verbosity is forwarded for GPT-5 fine-tune IDs", async () => {
849
891
  const ftProvider = new OpenAIResponsesProvider(
850
892
  "sk-test",
@@ -121,6 +121,35 @@ describe("resolvePricing", () => {
121
121
  expect(result.estimatedCostUsd).toBeCloseTo(5 + 6.25 + 1 + 45, 10);
122
122
  });
123
123
 
124
+ test("bills GPT-6 Astra at published short-context rates", () => {
125
+ const result = resolvePricingForUsage("openai", "gpt-6-astra", {
126
+ directInputTokens: 50_000,
127
+ outputTokens: 0,
128
+ cacheCreationInputTokens: 50_000,
129
+ cacheReadInputTokens: 50_000,
130
+ anthropicCacheCreation: null,
131
+ });
132
+
133
+ expect(result.pricingStatus).toBe("priced");
134
+ // 0.05M x $10 direct + 0.05M x $12.5 write + 0.05M x $1 read
135
+ expect(result.estimatedCostUsd).toBeCloseTo(0.5 + 0.625 + 0.05, 10);
136
+ });
137
+
138
+ test("bills GPT-6 Astra at the long-context tier above 272k", () => {
139
+ const result = resolvePricingForUsage("openai", "gpt-6-astra", {
140
+ directInputTokens: 500_000,
141
+ outputTokens: 1_000_000,
142
+ cacheCreationInputTokens: 500_000,
143
+ cacheReadInputTokens: 1_000_000,
144
+ anthropicCacheCreation: null,
145
+ });
146
+
147
+ expect(result.pricingStatus).toBe("priced");
148
+ // Tier rates: 0.5M x $20 direct + 0.5M x $25 write + 1M x $2 read
149
+ // + 1M x $75 output.
150
+ expect(result.estimatedCostUsd).toBeCloseTo(10 + 12.5 + 2 + 75, 10);
151
+ });
152
+
124
153
  test("uses OpenAI short-context tiers through 272k prompt tokens", () => {
125
154
  const result = resolvePricingForUsage("openai", "gpt-5.4", {
126
155
  directInputTokens: 272_000,
@@ -1215,6 +1215,11 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
1215
1215
  private receivedAudio = false;
1216
1216
  private detectedSpeech = false;
1217
1217
  private dispatchedTurn = false;
1218
+ // Loudest server-VAD chunk not attributed to assistant playback, on the
1219
+ // gate's own scale. Logged with a silent session's end so `no_speech` can
1220
+ // be told apart: a peak under the gate is a user who talked and was not
1221
+ // heard, a peak near the floor is a user who said nothing.
1222
+ private peakChunkAmplitude = 0;
1218
1223
  // The client declared a text input affordance on the start frame, so it can
1219
1224
  // take a turn without the microphone. Governs one thing only: whether a
1220
1225
  // missing speech-to-text leg is fatal to startup (see start()).
@@ -1883,6 +1888,25 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
1883
1888
  ),
1884
1889
  outcome: failed ? "failed" : "completed",
1885
1890
  });
1891
+ if (silenceReason !== null) {
1892
+ // The levels behind a silent session, on the gate's scale, since the
1893
+ // end event carries only the classification. A peak below `speechGate`
1894
+ // on a `no_speech` session is a microphone the gate could not hear.
1895
+ log.info(
1896
+ {
1897
+ sessionId: this.context.sessionId,
1898
+ reason,
1899
+ silenceReason,
1900
+ peakChunkAmplitude: Math.round(this.peakChunkAmplitude),
1901
+ noiseFloor:
1902
+ this.roomNoiseFloor.floor === null
1903
+ ? null
1904
+ : Math.round(this.roomNoiseFloor.floor),
1905
+ speechGate: Math.round(this.effectiveBaseThreshold()),
1906
+ },
1907
+ "Live-voice session ended without a turn",
1908
+ );
1909
+ }
1886
1910
 
1887
1911
  const shouldEmitSessionEndMetrics = this.state !== "failed";
1888
1912
  this.state = "closed";
@@ -2294,6 +2318,12 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
2294
2318
  if (energyClassification === "echo") {
2295
2319
  return;
2296
2320
  }
2321
+ // Measured past the echo gate, so a greeting heard through the speaker
2322
+ // cannot stand in for the user on a silent close.
2323
+ const meanAmplitude = pcm16MeanAmplitude(chunk);
2324
+ if (meanAmplitude > this.peakChunkAmplitude) {
2325
+ this.peakChunkAmplitude = meanAmplitude;
2326
+ }
2297
2327
 
2298
2328
  // Idle mic: hold silent chunks in the bounded pre-roll instead of
2299
2329
  // collecting or streaming them; flushed on speech onset so the
@@ -54,6 +54,7 @@ describe("isConnectionCompatibleWithModel", () => {
54
54
 
55
55
  test("oauth_subscription connection is compatible with a Codex model", () => {
56
56
  const conn = { auth: oauthAuth };
57
+ expect(isConnectionCompatibleWithModel(conn, "gpt-6-astra")).toBe(true);
57
58
  expect(isConnectionCompatibleWithModel(conn, "gpt-5.6-sol")).toBe(true);
58
59
  expect(isConnectionCompatibleWithModel(conn, "gpt-5.6-terra")).toBe(true);
59
60
  expect(isConnectionCompatibleWithModel(conn, "gpt-5.6-luna")).toBe(true);
@@ -403,6 +403,42 @@ const RAW_PROVIDER_CATALOG: ProviderCatalogEntry[] = [
403
403
  linkLabel: "Open OpenAI Platform",
404
404
  },
405
405
  models: [
406
+ // GPT-6 Astra. cacheRead is the 90% cached-read discount; cacheWrite
407
+ // is the 1.25x-input rate GPT-5.6+ bills for prompt tokens written to
408
+ // the cache (reported as `cache_write_tokens` in usage, tracked as
409
+ // `cacheCreationInputTokens`). Long-context (>272K input) is 2x input
410
+ // / 1.5x output / 2x cache-read+write for the whole request. Effort
411
+ // accepts low through max and rejects `none`.
412
+ {
413
+ id: "gpt-6-astra",
414
+ displayName: "GPT-6 Astra",
415
+ contextWindowTokens: 1050000,
416
+ maxOutputTokens: 128000,
417
+ longContextPricingThresholdTokens:
418
+ OPENAI_LONG_CONTEXT_PRICING_THRESHOLD_TOKENS,
419
+ supportsThinking: true,
420
+ supportsCaching: true,
421
+ supportsVision: true,
422
+ supportsToolUse: true,
423
+ supportsPromptCacheBreakpoints: true,
424
+ maxEffort: "max",
425
+ supportedEfforts: ["low", "medium", "high", "xhigh", "max"],
426
+ pricing: {
427
+ inputPer1mTokens: 10.0,
428
+ outputPer1mTokens: 50.0,
429
+ cacheWritePer1mTokens: 12.5,
430
+ cacheReadPer1mTokens: 1.0,
431
+ tiers: [
432
+ {
433
+ inputTokenThreshold: OPENAI_LONG_CONTEXT_PRICING_THRESHOLD_TOKENS,
434
+ inputPer1mTokens: 20,
435
+ outputPer1mTokens: 75,
436
+ cacheWritePer1mTokens: 25,
437
+ cacheReadPer1mTokens: 2,
438
+ },
439
+ ],
440
+ },
441
+ },
406
442
  // GPT-5.6 family (Sol / Terra / Luna). cacheRead is the 90% cached-read
407
443
  // discount; cacheWrite is the 1.25x-input rate GPT-5.6+ bills for
408
444
  // prompt tokens written to the cache (reported as `cache_write_tokens`
@@ -1309,6 +1345,72 @@ const RAW_PROVIDER_CATALOG: ProviderCatalogEntry[] = [
1309
1345
  },
1310
1346
  },
1311
1347
  // OpenAI
1348
+ // GPT-6 Astra. The `*-pro` slug is the same underlying model served
1349
+ // with `reasoning.mode: pro` at identical rates. cacheWrite is the
1350
+ // 1.25x-input rate GPT-5.6+ bills for prompt tokens written to the
1351
+ // cache. Long-context (>272K input) is 2x input / 1.5x output / 2x
1352
+ // cache-read+write for the whole request. Effort accepts low through
1353
+ // max and rejects `none`.
1354
+ {
1355
+ id: "openai/gpt-6-astra",
1356
+ displayName: "GPT-6 Astra",
1357
+ contextWindowTokens: 1050000,
1358
+ maxOutputTokens: 128000,
1359
+ longContextPricingThresholdTokens:
1360
+ OPENAI_LONG_CONTEXT_PRICING_THRESHOLD_TOKENS,
1361
+ supportsThinking: true,
1362
+ supportsCaching: true,
1363
+ supportsVision: true,
1364
+ supportsToolUse: true,
1365
+ supportsPromptCacheBreakpoints: true,
1366
+ maxEffort: "max",
1367
+ supportedEfforts: ["low", "medium", "high", "xhigh", "max"],
1368
+ pricing: {
1369
+ inputPer1mTokens: 10.0,
1370
+ outputPer1mTokens: 50.0,
1371
+ cacheWritePer1mTokens: 12.5,
1372
+ cacheReadPer1mTokens: 1.0,
1373
+ tiers: [
1374
+ {
1375
+ inputTokenThreshold: OPENAI_LONG_CONTEXT_PRICING_THRESHOLD_TOKENS,
1376
+ inputPer1mTokens: 20,
1377
+ outputPer1mTokens: 75,
1378
+ cacheWritePer1mTokens: 25,
1379
+ cacheReadPer1mTokens: 2,
1380
+ },
1381
+ ],
1382
+ },
1383
+ },
1384
+ {
1385
+ id: "openai/gpt-6-astra-pro",
1386
+ displayName: "GPT-6 Astra Pro",
1387
+ contextWindowTokens: 1050000,
1388
+ maxOutputTokens: 128000,
1389
+ longContextPricingThresholdTokens:
1390
+ OPENAI_LONG_CONTEXT_PRICING_THRESHOLD_TOKENS,
1391
+ supportsThinking: true,
1392
+ supportsCaching: true,
1393
+ supportsVision: true,
1394
+ supportsToolUse: true,
1395
+ supportsPromptCacheBreakpoints: true,
1396
+ maxEffort: "max",
1397
+ supportedEfforts: ["low", "medium", "high", "xhigh", "max"],
1398
+ pricing: {
1399
+ inputPer1mTokens: 10.0,
1400
+ outputPer1mTokens: 50.0,
1401
+ cacheWritePer1mTokens: 12.5,
1402
+ cacheReadPer1mTokens: 1.0,
1403
+ tiers: [
1404
+ {
1405
+ inputTokenThreshold: OPENAI_LONG_CONTEXT_PRICING_THRESHOLD_TOKENS,
1406
+ inputPer1mTokens: 20,
1407
+ outputPer1mTokens: 75,
1408
+ cacheWritePer1mTokens: 25,
1409
+ cacheReadPer1mTokens: 2,
1410
+ },
1411
+ ],
1412
+ },
1413
+ },
1312
1414
  // GPT-5.6 family (Sol / Terra / Luna). The `*-pro` slugs are the same
1313
1415
  // underlying models served with `reasoning.mode: pro` at identical
1314
1416
  // rates. cacheWrite is the 1.25x-input rate GPT-5.6+ bills for prompt
@@ -191,7 +191,7 @@ const log = getLogger("chat-completions");
191
191
  /** Wire-level reasoning_effort values. The OpenAI SDK type doesn't include
192
192
  * `"max"`, but Fireworks accepts it for DeepSeek V4; the assignment to
193
193
  * `params.reasoning_effort` casts through this union. */
194
- type ReasoningEffortWire = "none" | "low" | "medium" | "high" | "xhigh" | "max";
194
+ export type ReasoningEffortWire = "none" | "low" | "medium" | "high" | "xhigh" | "max";
195
195
 
196
196
  const REASONING_EFFORT_RANK: Record<ReasoningEffortWire, number> = {
197
197
  none: 0,
@@ -9,6 +9,7 @@
9
9
  * connection" profile.
10
10
  */
11
11
  export const CODEX_SUBSCRIPTION_MODEL_IDS: ReadonlySet<string> = new Set([
12
+ "gpt-6-astra",
12
13
  "gpt-5.6-sol",
13
14
  "gpt-5.6-terra",
14
15
  "gpt-5.6-luna",
@@ -8,7 +8,11 @@ import { extractRetryAfterMs } from "../../util/retry.js";
8
8
  import { clampProviderString } from "../content-block-size.js";
9
9
  import { fileBlockToProviderText } from "../file-block-text.js";
10
10
  import { base64Source, resolveMediaReferences } from "../media-resolve.js";
11
- import { PROMPT_CACHE_BREAKPOINT_MODEL_IDS } from "../model-catalog.js";
11
+ import {
12
+ modelEffortCeilings,
13
+ modelSupportedEfforts,
14
+ PROMPT_CACHE_BREAKPOINT_MODEL_IDS,
15
+ } from "../model-catalog.js";
12
16
  import { recordProviderRequestDiagnostics } from "../request-diagnostics.js";
13
17
  import { createStreamTimeout } from "../stream-timeout.js";
14
18
  import { createToolProgressEmitter } from "../tool-progress-events.js";
@@ -26,7 +30,12 @@ import {
26
30
  formatNormalizedOpenAIAPIError,
27
31
  normalizeOpenAIAPIError,
28
32
  } from "./api-error-normalization.js";
29
- import { detectOpenAICompatibleContextOverflow } from "./chat-completions-provider.js";
33
+ import {
34
+ clampReasoningEffort,
35
+ detectOpenAICompatibleContextOverflow,
36
+ type ReasoningEffortWire,
37
+ snapReasoningEffortToSupported,
38
+ } from "./chat-completions-provider.js";
30
39
  import { serializeToolResult } from "./orphaned-tool-result.js";
31
40
 
32
41
  const log = getLogger("openai-responses");
@@ -46,22 +55,63 @@ export interface OpenAIResponsesProviderOptions {
46
55
  }
47
56
 
48
57
  /** Map our internal effort values to the Responses API reasoning.effort parameter.
49
- * OpenAI caps at "xhigh", so our "max" tier collapses to "xhigh". `"none"` is
50
- * passed through explicitly because OpenAI defaults `reasoning.effort` to
51
- * "medium" when the field is omitted — the user's opt-out is only honored
52
- * when we send it on the wire. */
53
- const EFFORT_TO_REASONING_EFFORT: Record<
54
- string,
55
- "none" | "low" | "medium" | "high" | "xhigh"
56
- > = {
58
+ * `"max"` is emitted raw and then clamped to the model's catalog ceiling
59
+ * ({@link mapResponsesReasoningEffort}); models that omit `maxEffort`
60
+ * inherit OpenAI's historical `xhigh` cap. `"none"` is passed through
61
+ * explicitly because OpenAI defaults `reasoning.effort` to "medium" when
62
+ * the field is omitted, except on models whose `supportedEfforts` omit
63
+ * `none` (those snap to the lowest accepted value). */
64
+ const EFFORT_TO_REASONING_EFFORT: Record<string, ReasoningEffortWire> = {
57
65
  none: "none",
58
66
  low: "low",
59
67
  medium: "medium",
60
68
  high: "high",
61
69
  xhigh: "xhigh",
62
- max: "xhigh",
70
+ max: "max",
63
71
  };
64
72
 
73
+ const OPENAI_EFFORT_CEILINGS = modelEffortCeilings("openai");
74
+ const OPENROUTER_EFFORT_CEILINGS = modelEffortCeilings("openrouter");
75
+ const OPENAI_SUPPORTED_EFFORTS = modelSupportedEfforts("openai");
76
+ const OPENROUTER_SUPPORTED_EFFORTS = modelSupportedEfforts("openrouter");
77
+
78
+ function effortCeilingForModel(model: string): "high" | "xhigh" | "max" {
79
+ return (
80
+ OPENAI_EFFORT_CEILINGS.get(model) ??
81
+ OPENROUTER_EFFORT_CEILINGS.get(model) ??
82
+ "xhigh"
83
+ );
84
+ }
85
+
86
+ function supportedEffortsForModel(
87
+ model: string,
88
+ ): readonly ("low" | "medium" | "high" | "xhigh" | "max")[] | undefined {
89
+ return (
90
+ OPENAI_SUPPORTED_EFFORTS.get(model) ??
91
+ OPENROUTER_SUPPORTED_EFFORTS.get(model)
92
+ );
93
+ }
94
+
95
+ /** Translate a Vellum effort value onto the Responses wire for `model`. */
96
+ function mapResponsesReasoningEffort(
97
+ effort: string,
98
+ model: string,
99
+ ): ReasoningEffortWire | undefined {
100
+ const raw = EFFORT_TO_REASONING_EFFORT[effort];
101
+ if (!raw) {
102
+ return undefined;
103
+ }
104
+ const supported = supportedEffortsForModel(model);
105
+ if (raw === "none") {
106
+ if (supported && supported.length > 0) {
107
+ return supported[0];
108
+ }
109
+ return "none";
110
+ }
111
+ const clamped = clampReasoningEffort(raw, effortCeilingForModel(model));
112
+ return supported ? snapReasoningEffortToSupported(clamped, supported) : clamped;
113
+ }
114
+
65
115
  /** Values accepted by the Responses API `text.verbosity` parameter. */
66
116
  const VALID_VERBOSITIES = new Set<string>(["low", "medium", "high"]);
67
117
 
@@ -102,15 +152,15 @@ export function mapNeutralToolChoiceForResponses(
102
152
  }
103
153
  }
104
154
 
105
- /** `text.verbosity` is a GPT-5-series-only parameter. Older models on the
155
+ /** `text.verbosity` is a GPT-5/GPT-6-series parameter. Older models on the
106
156
  * Responses API (o-series, etc.) reject unknown wire fields with HTTP 400, so
107
157
  * gate forwarding by model name here. The retry layer can't make this call
108
158
  * because verbosity defaults to "medium" in the LLM schema, so every
109
159
  * callSite-resolved request would otherwise carry it regardless of model.
110
160
  * Also matches OpenAI fine-tune IDs of the form `ft:gpt-5.x:org::id` so users
111
- * on GPT-5 fine-tunes keep explicit verbosity control. */
161
+ * on GPT-5/GPT-6 fine-tunes keep explicit verbosity control. */
112
162
  function modelSupportsVerbosity(model: string): boolean {
113
- return /^(ft:)?gpt-5(\b|[-.])/i.test(model);
163
+ return /^(ft:)?gpt-[56](\b|[-.])/i.test(model);
114
164
  }
115
165
 
116
166
  /** Loosely-typed Responses stream event to avoid `any` while the SDK types settle. */
@@ -295,7 +345,7 @@ export class OpenAIResponsesProvider implements Provider {
295
345
  }
296
346
 
297
347
  const reasoningEffort = effort
298
- ? EFFORT_TO_REASONING_EFFORT[effort]
348
+ ? mapResponsesReasoningEffort(effort, effectiveModel)
299
349
  : undefined;
300
350
  if (reasoningEffort) {
301
351
  // Request a human-readable reasoning summary whenever the model will