smoltalk 0.9.0 → 0.10.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (86) hide show
  1. package/README.md +175 -7
  2. package/dist/classes/message/AssistantMessage.d.ts +2 -0
  3. package/dist/classes/message/UserMessage.d.ts +21 -0
  4. package/dist/classes/message/UserMessage.js +3 -0
  5. package/dist/classes/message/contentParts.d.ts +71 -2
  6. package/dist/classes/message/contentParts.js +6 -0
  7. package/dist/classes/message/index.d.ts +5 -2
  8. package/dist/classes/message/index.js +7 -0
  9. package/dist/classes/message/renderers/AnthropicRenderer.d.ts +2 -1
  10. package/dist/classes/message/renderers/AnthropicRenderer.js +3 -0
  11. package/dist/classes/message/renderers/GoogleRenderer.d.ts +2 -1
  12. package/dist/classes/message/renderers/GoogleRenderer.js +3 -0
  13. package/dist/classes/message/renderers/JSONRenderer.d.ts +2 -1
  14. package/dist/classes/message/renderers/JSONRenderer.js +4 -0
  15. package/dist/classes/message/renderers/OpenAIChatRenderer.d.ts +8 -1
  16. package/dist/classes/message/renderers/OpenAIChatRenderer.js +18 -0
  17. package/dist/classes/message/renderers/OpenAIResponsesRenderer.d.ts +2 -1
  18. package/dist/classes/message/renderers/OpenAIResponsesRenderer.js +3 -0
  19. package/dist/classes/message/renderers/PartRenderer.d.ts +3 -2
  20. package/dist/classes/message/renderers/PartRenderer.js +3 -0
  21. package/dist/client.js +1 -0
  22. package/dist/clients/anthropic.js +1 -1
  23. package/dist/clients/baseClient.d.ts +13 -1
  24. package/dist/clients/baseClient.js +36 -7
  25. package/dist/clients/google.js +1 -1
  26. package/dist/clients/ollama.js +1 -1
  27. package/dist/clients/openai.d.ts +2 -1
  28. package/dist/clients/openai.js +15 -3
  29. package/dist/clients/openaiCompat.d.ts +2 -0
  30. package/dist/clients/openaiCompat.js +5 -0
  31. package/dist/clients/openaiResponses.js +1 -1
  32. package/dist/clients/resolveAttachments.d.ts +8 -4
  33. package/dist/clients/resolveAttachments.js +101 -50
  34. package/dist/embed.d.ts +4 -0
  35. package/dist/files.d.ts +1 -1
  36. package/dist/files.js +1 -1
  37. package/dist/image/google.js +2 -2
  38. package/dist/image/openai.js +3 -3
  39. package/dist/image.d.ts +1 -1
  40. package/dist/index.d.ts +10 -2
  41. package/dist/index.js +7 -1
  42. package/dist/model.d.ts +15 -4
  43. package/dist/model.js +72 -12
  44. package/dist/models.d.ts +204 -22
  45. package/dist/models.js +211 -30
  46. package/dist/speech/baseSpeechClient.d.ts +36 -0
  47. package/dist/speech/baseSpeechClient.js +117 -0
  48. package/dist/speech/google.d.ts +6 -0
  49. package/dist/speech/google.js +54 -0
  50. package/dist/speech/groq.d.ts +11 -0
  51. package/dist/speech/groq.js +19 -0
  52. package/dist/speech/openai.d.ts +14 -0
  53. package/dist/speech/openai.js +51 -0
  54. package/dist/speech/openaiCompat.d.ts +13 -0
  55. package/dist/speech/openaiCompat.js +22 -0
  56. package/dist/speech.d.ts +45 -0
  57. package/dist/speech.js +63 -0
  58. package/dist/transcription/baseTranscriptionClient.d.ts +36 -0
  59. package/dist/transcription/baseTranscriptionClient.js +133 -0
  60. package/dist/transcription/google.d.ts +6 -0
  61. package/dist/transcription/google.js +56 -0
  62. package/dist/transcription/groq.d.ts +10 -0
  63. package/dist/transcription/groq.js +17 -0
  64. package/dist/transcription/openai.d.ts +11 -0
  65. package/dist/transcription/openai.js +67 -0
  66. package/dist/transcription/openaiCompat.d.ts +13 -0
  67. package/dist/transcription/openaiCompat.js +22 -0
  68. package/dist/transcription.d.ts +54 -0
  69. package/dist/transcription.js +64 -0
  70. package/dist/types/tokenUsage.d.ts +4 -0
  71. package/dist/types/tokenUsage.js +4 -0
  72. package/dist/types.d.ts +4 -0
  73. package/dist/util/attachments.d.ts +1 -1
  74. package/dist/util/audioMime.d.ts +26 -0
  75. package/dist/util/audioMime.js +74 -0
  76. package/dist/util/{imageRef.d.ts → blobRef.d.ts} +9 -9
  77. package/dist/util/{imageRef.js → blobRef.js} +6 -13
  78. package/dist/util/googleAudioUsage.d.ts +14 -0
  79. package/dist/util/googleAudioUsage.js +52 -0
  80. package/dist/util/mime.d.ts +21 -0
  81. package/dist/util/mime.js +54 -0
  82. package/dist/util/modalities.d.ts +6 -2
  83. package/dist/util/modalities.js +13 -15
  84. package/dist/util/provider.d.ts +3 -0
  85. package/dist/util/provider.js +3 -1
  86. package/package.json +1 -1
package/dist/model.js CHANGED
@@ -1,6 +1,7 @@
1
- import { getModel, isTextModel, ModelNameSchema } from "./models.js";
1
+ import { getModel, getModelForProvider, isSpeechToTextModel, isTextToSpeechModel, ModelNameSchema, } from "./models.js";
2
2
  import { SmolError } from "./smolError.js";
3
3
  import { round } from "./util/util.js";
4
+ const TOKEN_COST_UNIT = 1_000_000;
4
5
  export class Model {
5
6
  model;
6
7
  provider;
@@ -24,8 +25,30 @@ export class Model {
24
25
  return modelInfo ? modelInfo.provider : undefined;
25
26
  }
26
27
  calculateCost(usage) {
27
- const model = getModel(this.model, this.modelData);
28
- if (!model || !isTextModel(model)) {
28
+ let model;
29
+ if (this.provider !== undefined) {
30
+ model = getModelForProvider(this.provider, this.model, this.modelData);
31
+ }
32
+ else {
33
+ model = getModel(this.model, this.modelData);
34
+ }
35
+ if (!model) {
36
+ return null;
37
+ }
38
+ // This token engine prices text generation and token-billed audio models
39
+ // (e.g. Gemini TTS). Image and embeddings models have their own cost paths,
40
+ // so they are never priced here even if they carry text-token rates.
41
+ if (model.type === "image" || model.type === "embeddings") {
42
+ return null;
43
+ }
44
+ // BaseModel token-rate fields, read structurally across the model union.
45
+ const rates = model;
46
+ // Price only models that carry at least one token rate; those without
47
+ // (per-minute STT, per-char TTS) return null so their dedicated helpers apply.
48
+ if (rates.inputTokenCost === undefined &&
49
+ rates.outputTokenCost === undefined &&
50
+ rates.inputAudioTokenCost === undefined &&
51
+ rates.outputAudioTokenCost === undefined) {
29
52
  return null;
30
53
  }
31
54
  const cachedTokens = usage.cachedInputTokens ?? 0;
@@ -33,10 +56,17 @@ export class Model {
33
56
  // Disjoint buckets. If a discount price isn't defined for this model,
34
57
  // the tokens were still billed by the provider — charge them at the
35
58
  // full input rate so totalCost stays honest.
36
- const cachedRate = model.cachedInputTokenCost ?? model.inputTokenCost ?? 0;
37
- const cacheCreationRate = model.cacheCreationInputTokenCost ?? model.inputTokenCost ?? 0;
38
- const inputCost = round((usage.inputTokens * (model.inputTokenCost || 0)) / 1_000_000, 6);
39
- const outputCost = round((usage.outputTokens * (model.outputTokenCost || 0)) / 1_000_000, 6);
59
+ const cachedRate = rates.cachedInputTokenCost ?? rates.inputTokenCost ?? 0;
60
+ const cacheCreationRate = rates.cacheCreationInputTokenCost ?? rates.inputTokenCost ?? 0;
61
+ const inputCost = round((usage.inputTokens * (rates.inputTokenCost || 0)) / TOKEN_COST_UNIT, 6);
62
+ const outputCost = round((usage.outputTokens * (rates.outputTokenCost || 0)) / TOKEN_COST_UNIT, 6);
63
+ const audioInTokens = usage.inputAudioTokens ?? 0;
64
+ const audioOutTokens = usage.outputAudioTokens ?? 0;
65
+ // Fall back to the text rate if no audio rate is defined so the total stays honest.
66
+ const audioInRate = rates.inputAudioTokenCost ?? rates.inputTokenCost ?? 0;
67
+ const audioOutRate = rates.outputAudioTokenCost ?? rates.outputTokenCost ?? 0;
68
+ const audioInCost = round((audioInTokens * audioInRate) / TOKEN_COST_UNIT, 6);
69
+ const audioOutCost = round((audioOutTokens * audioOutRate) / TOKEN_COST_UNIT, 6);
40
70
  // Only expose cachedInputCost / cacheCreationInputCost when the model
41
71
  // actually has a distinct discount price. Otherwise, fold those dollars
42
72
  // into inputCost so the user isn't misled by a $0 cached field.
@@ -45,7 +75,7 @@ export class Model {
45
75
  let foldedInputDollars = 0;
46
76
  if (cachedTokens > 0) {
47
77
  const dollars = (cachedTokens * cachedRate) / 1_000_000;
48
- if (model.cachedInputTokenCost != null) {
78
+ if (rates.cachedInputTokenCost != null) {
49
79
  cachedInputCost = round(dollars, 6);
50
80
  }
51
81
  else {
@@ -54,21 +84,22 @@ export class Model {
54
84
  }
55
85
  if (cacheCreationTokens > 0) {
56
86
  const dollars = (cacheCreationTokens * cacheCreationRate) / 1_000_000;
57
- if (model.cacheCreationInputTokenCost != null) {
87
+ if (rates.cacheCreationInputTokenCost != null) {
58
88
  cacheCreationInputCost = round(dollars, 6);
59
89
  }
60
90
  else {
61
91
  foldedInputDollars += dollars;
62
92
  }
63
93
  }
64
- const finalInputCost = round(inputCost + foldedInputDollars, 6);
94
+ const finalInputCost = round(inputCost + foldedInputDollars + audioInCost, 6);
95
+ const finalOutputCost = round(outputCost + audioOutCost, 6);
65
96
  const totalCost = round(finalInputCost +
66
- outputCost +
97
+ finalOutputCost +
67
98
  (cachedInputCost || 0) +
68
99
  (cacheCreationInputCost || 0), 6);
69
100
  return {
70
101
  inputCost: finalInputCost,
71
- outputCost,
102
+ outputCost: finalOutputCost,
72
103
  cachedInputCost,
73
104
  cacheCreationInputCost,
74
105
  totalCost,
@@ -88,3 +119,32 @@ export class Model {
88
119
  return new Model(model, provider, modelData);
89
120
  }
90
121
  }
122
+ /**
123
+ * Per-minute STT pricing from a registry entry. Returns undefined (cost
124
+ * omitted, no error) when the model, rate, or duration is unknown — a rate of
125
+ * 0 still yields a present zero cost.
126
+ */
127
+ export function calculateTranscriptionCost(model, durationSeconds) {
128
+ if (model === undefined || !isSpeechToTextModel(model)) {
129
+ return undefined;
130
+ }
131
+ if (model.perMinuteCost === undefined || durationSeconds === undefined || durationSeconds === null) {
132
+ return undefined;
133
+ }
134
+ // Providers may bill a minimum duration regardless of actual length
135
+ // (e.g. Groq rounds up to 10s), so a shorter clip isn't understated.
136
+ const billedSeconds = Math.max(durationSeconds, model.minimumBillableSeconds ?? 0);
137
+ const inputCost = round((billedSeconds / 60) * model.perMinuteCost, 6);
138
+ return { inputCost, outputCost: 0, totalCost: inputCost, currency: "USD" };
139
+ }
140
+ /** Per-code-point TTS pricing from a registry entry; same omission semantics. */
141
+ export function calculateSpeechCost(model, charCount) {
142
+ if (model === undefined || !isTextToSpeechModel(model)) {
143
+ return undefined;
144
+ }
145
+ if (model.perCharacterCost === undefined) {
146
+ return undefined;
147
+ }
148
+ const inputCost = round(charCount * model.perCharacterCost, 6);
149
+ return { inputCost, outputCost: 0, totalCost: inputCost, currency: "USD" };
150
+ }
package/dist/models.d.ts CHANGED
@@ -1,6 +1,6 @@
1
1
  import { z } from "zod";
2
2
  import { type ModelDataBlob, type HostedTool, type HostedToolPrice } from "./modelData.js";
3
- export declare const providers: readonly ["ollama", "openai", "openai-responses", "anthropic", "google", "replicate", "modal", "openrouter", "deepinfra", "litellm", "openai-compat"];
3
+ export declare const providers: readonly ["ollama", "openai", "openai-responses", "anthropic", "google", "replicate", "modal", "openrouter", "deepinfra", "litellm", "openai-compat", "groq"];
4
4
  export declare const ProviderSchema: z.ZodEnum<{
5
5
  openai: "openai";
6
6
  anthropic: "anthropic";
@@ -13,6 +13,7 @@ export declare const ProviderSchema: z.ZodEnum<{
13
13
  deepinfra: "deepinfra";
14
14
  litellm: "litellm";
15
15
  "openai-compat": "openai-compat";
16
+ groq: "groq";
16
17
  }>;
17
18
  export type Provider = z.infer<typeof ProviderSchema>;
18
19
  export type BaseModel = {
@@ -23,6 +24,8 @@ export type BaseModel = {
23
24
  cachedInputTokenCost?: number;
24
25
  cacheCreationInputTokenCost?: number;
25
26
  outputTokenCost?: number;
27
+ inputAudioTokenCost?: number;
28
+ outputAudioTokenCost?: number;
26
29
  disabled?: boolean;
27
30
  costUnit?: "tokens" | "characters" | "minutes";
28
31
  knowledge?: string;
@@ -34,6 +37,25 @@ export type BaseModel = {
34
37
  export type SpeechToTextModel = BaseModel & {
35
38
  type: "speech-to-text";
36
39
  perMinuteCost?: number;
40
+ /** Provider's minimum billable duration in seconds (e.g. Groq bills >= 10s). */
41
+ minimumBillableSeconds?: number;
42
+ /** Canonical MIME types accepted after alias normalization through AUDIO_FORMATS. */
43
+ supportedMimeTypes?: readonly string[];
44
+ /** Provider upload cap in bytes. */
45
+ maxBytes?: number;
46
+ };
47
+ export type TextToSpeechModel = BaseModel & {
48
+ type: "text-to-speech";
49
+ perCharacterCost?: number;
50
+ /** Input cap in Unicode code points. */
51
+ maxInputChars?: number;
52
+ /** Accepted values for the speed option. */
53
+ speedRange?: {
54
+ min: number;
55
+ max: number;
56
+ };
57
+ /** Output formats the provider can render for this model. */
58
+ formats?: readonly string[];
37
59
  };
38
60
  export type ImageModel = BaseModel & {
39
61
  type: "image";
@@ -73,10 +95,11 @@ export type TextModel = BaseModel & {
73
95
  input: string[];
74
96
  output: string[];
75
97
  };
98
+ /** Audio-input constraints when this multimodal model is used for transcription. */
99
+ supportedMimeTypes?: readonly string[];
100
+ maxBytes?: number;
76
101
  structuredOutput?: boolean;
77
102
  temperatureSupported?: boolean;
78
- inputAudioTokenCost?: number;
79
- outputAudioTokenCost?: number;
80
103
  /** Pricing that applies above a context-size threshold (e.g. Gemini >200k). */
81
104
  longContext?: {
82
105
  thresholdTokens: number;
@@ -91,12 +114,81 @@ export type EmbeddingsModel = {
91
114
  provider: string;
92
115
  tokenCost?: number;
93
116
  };
94
- export type ModelType = SpeechToTextModel | TextModel | EmbeddingsModel | ImageModel;
117
+ export type ModelType = SpeechToTextModel | TextToSpeechModel | TextModel | EmbeddingsModel | ImageModel;
95
118
  export declare const speechToTextModels: readonly [{
96
119
  readonly type: "speech-to-text";
97
- readonly modelName: "whisper-web";
120
+ readonly modelName: "whisper-1";
98
121
  readonly perMinuteCost: 0.006;
99
122
  readonly provider: "openai";
123
+ readonly supportedMimeTypes: readonly ["audio/flac", "audio/mpeg", "audio/mp4", "audio/m4a", "audio/ogg", "audio/wav", "audio/webm"];
124
+ readonly maxBytes: number;
125
+ }, {
126
+ readonly type: "speech-to-text";
127
+ readonly modelName: "whisper-large-v3";
128
+ readonly provider: "groq";
129
+ readonly perMinuteCost: 0.00185;
130
+ readonly minimumBillableSeconds: 10;
131
+ readonly supportedMimeTypes: readonly ["audio/flac", "audio/mpeg", "audio/mp4", "audio/m4a", "audio/ogg", "audio/wav", "audio/webm"];
132
+ readonly maxBytes: number;
133
+ }, {
134
+ readonly type: "speech-to-text";
135
+ readonly modelName: "whisper-large-v3-turbo";
136
+ readonly provider: "groq";
137
+ readonly perMinuteCost: 0.000667;
138
+ readonly minimumBillableSeconds: 10;
139
+ readonly supportedMimeTypes: readonly ["audio/flac", "audio/mpeg", "audio/mp4", "audio/m4a", "audio/ogg", "audio/wav", "audio/webm"];
140
+ readonly maxBytes: number;
141
+ }];
142
+ export declare const textToSpeechModels: readonly [{
143
+ readonly type: "text-to-speech";
144
+ readonly modelName: "tts-1";
145
+ readonly perCharacterCost: 0.000015;
146
+ readonly provider: "openai";
147
+ readonly maxInputChars: 4096;
148
+ readonly speedRange: {
149
+ readonly min: 0.25;
150
+ readonly max: 4;
151
+ };
152
+ readonly formats: readonly ["mp3", "opus", "aac", "flac", "wav", "pcm"];
153
+ }, {
154
+ readonly type: "text-to-speech";
155
+ readonly modelName: "tts-1-hd";
156
+ readonly perCharacterCost: 0.00003;
157
+ readonly provider: "openai";
158
+ readonly maxInputChars: 4096;
159
+ readonly speedRange: {
160
+ readonly min: 0.25;
161
+ readonly max: 4;
162
+ };
163
+ readonly formats: readonly ["mp3", "opus", "aac", "flac", "wav", "pcm"];
164
+ }, {
165
+ readonly type: "text-to-speech";
166
+ readonly modelName: "canopylabs/orpheus-v1-english";
167
+ readonly provider: "groq";
168
+ readonly perCharacterCost: 0.000022;
169
+ readonly maxInputChars: 200;
170
+ readonly formats: readonly ["wav"];
171
+ }, {
172
+ readonly type: "text-to-speech";
173
+ readonly modelName: "canopylabs/orpheus-arabic-saudi";
174
+ readonly provider: "groq";
175
+ readonly perCharacterCost: 0.00004;
176
+ readonly maxInputChars: 200;
177
+ readonly formats: readonly ["wav"];
178
+ }, {
179
+ readonly type: "text-to-speech";
180
+ readonly modelName: "gemini-2.5-flash-preview-tts";
181
+ readonly provider: "google";
182
+ readonly inputTokenCost: 0.5;
183
+ readonly outputAudioTokenCost: 10;
184
+ readonly formats: readonly ["pcm", "wav"];
185
+ }, {
186
+ readonly type: "text-to-speech";
187
+ readonly modelName: "gemini-2.5-pro-preview-tts";
188
+ readonly provider: "google";
189
+ readonly inputTokenCost: 1;
190
+ readonly outputAudioTokenCost: 20;
191
+ readonly formats: readonly ["pcm", "wav"];
100
192
  }];
101
193
  export declare const textModels: readonly [{
102
194
  readonly type: "text";
@@ -810,16 +902,16 @@ export declare const textModels: readonly [{
810
902
  }, {
811
903
  readonly type: "text";
812
904
  readonly modelName: "gpt-5.6-terra";
813
- readonly description: "GPT-5.6 Terra balances capability and cost — competitive with GPT-5.5 at roughly half the price. 1M context window. Standard pricing for ≤272K tokens, 2x input/1.5x output for >272K. Knowledge cutoff: February 2026.";
905
+ readonly description: "GPT-5.6 Terra balances capability and cost. 1M context window. Standard pricing for ≤272K tokens, 2x input/1.5x output for >272K. Reflects the July 30, 2026 price cut. Knowledge cutoff: February 2026.";
814
906
  readonly maxInputTokens: 1050000;
815
907
  readonly maxOutputTokens: 128000;
816
- readonly inputTokenCost: 2.5;
817
- readonly cachedInputTokenCost: 0.25;
818
- readonly outputTokenCost: 15;
908
+ readonly inputTokenCost: 2;
909
+ readonly cachedInputTokenCost: 0.2;
910
+ readonly outputTokenCost: 12;
819
911
  readonly longContext: {
820
- readonly inputTokenCost: 5;
821
- readonly cachedInputTokenCost: 0.5;
822
- readonly outputTokenCost: 22.5;
912
+ readonly inputTokenCost: 4;
913
+ readonly cachedInputTokenCost: 0.4;
914
+ readonly outputTokenCost: 18;
823
915
  readonly thresholdTokens: 200000;
824
916
  };
825
917
  readonly reasoning: {
@@ -844,16 +936,16 @@ export declare const textModels: readonly [{
844
936
  }, {
845
937
  readonly type: "text";
846
938
  readonly modelName: "gpt-5.6-luna";
847
- readonly description: "GPT-5.6 Luna is the fast, most affordable member of the GPT-5.6 family. 1M context window. Standard pricing for ≤272K tokens, 2x input/1.5x output for >272K. Knowledge cutoff: February 2026.";
939
+ readonly description: "GPT-5.6 Luna is the fast, most affordable member of the GPT-5.6 family. 1M context window. Standard pricing for ≤272K tokens, 2x input/1.5x output for >272K. Reflects the July 30, 2026 price cut (input/output down ~80%). Knowledge cutoff: February 2026.";
848
940
  readonly maxInputTokens: 1050000;
849
941
  readonly maxOutputTokens: 128000;
850
- readonly inputTokenCost: 1;
851
- readonly cachedInputTokenCost: 0.1;
852
- readonly outputTokenCost: 6;
942
+ readonly inputTokenCost: 0.2;
943
+ readonly cachedInputTokenCost: 0.02;
944
+ readonly outputTokenCost: 1.2;
853
945
  readonly longContext: {
854
- readonly inputTokenCost: 2;
855
- readonly cachedInputTokenCost: 0.2;
856
- readonly outputTokenCost: 9;
946
+ readonly inputTokenCost: 0.4;
947
+ readonly cachedInputTokenCost: 0.04;
948
+ readonly outputTokenCost: 1.8;
857
949
  readonly thresholdTokens: 200000;
858
950
  };
859
951
  readonly reasoning: {
@@ -938,10 +1030,39 @@ export declare const textModels: readonly [{
938
1030
  readonly temperatureSupported: true;
939
1031
  readonly disabled: true;
940
1032
  readonly provider: "google";
1033
+ }, {
1034
+ readonly type: "text";
1035
+ readonly modelName: "gemini-3.6-flash";
1036
+ readonly description: "Latest Gemini 3.6 Flash model (GA July 21, 2026). Supersedes Gemini 3.5 Flash with better token efficiency and agentic planning at a lower output price ($7.50 vs $9.00/1M). 1M context window, 64K output. Context caching: $0.15/1M read.";
1037
+ readonly maxInputTokens: 1048576;
1038
+ readonly maxOutputTokens: 65536;
1039
+ readonly inputTokenCost: 1.5;
1040
+ readonly cachedInputTokenCost: 0.15;
1041
+ readonly outputTokenCost: 7.5;
1042
+ readonly inputAudioTokenCost: 1.5;
1043
+ readonly reasoning: {
1044
+ readonly levels: readonly ["minimal", "low", "medium", "high"];
1045
+ readonly defaultLevel: "high";
1046
+ readonly canDisable: false;
1047
+ readonly outputsThinking: true;
1048
+ readonly outputsSignatures: true;
1049
+ };
1050
+ readonly modalities: {
1051
+ readonly input: readonly ["text", "image", "video", "audio", "pdf"];
1052
+ readonly output: readonly ["text"];
1053
+ };
1054
+ readonly knowledge: "2026-03";
1055
+ readonly releaseDate: "2026-07-21";
1056
+ readonly lastUpdated: "2026-07-21";
1057
+ readonly family: "gemini-flash";
1058
+ readonly openWeights: false;
1059
+ readonly structuredOutput: true;
1060
+ readonly temperatureSupported: true;
1061
+ readonly provider: "google";
941
1062
  }, {
942
1063
  readonly type: "text";
943
1064
  readonly modelName: "gemini-3.5-flash";
944
- readonly description: "Latest Gemini 3.5 Flash model (GA May 2026). Outperforms Gemini 3.1 Pro on coding and agentic suites at 4x the speed. 1M context window, 64K output. Context caching: $0.15/1M read.";
1065
+ readonly description: "Gemini 3.5 Flash (GA May 2026). Superseded by gemini-3.6-flash. Outperforms Gemini 3.1 Pro on coding and agentic suites at 4x the speed. 1M context window, 64K output. Context caching: $0.15/1M read.";
945
1066
  readonly maxInputTokens: 1048576;
946
1067
  readonly maxOutputTokens: 65536;
947
1068
  readonly inputTokenCost: 1.5;
@@ -997,10 +1118,39 @@ export declare const textModels: readonly [{
997
1118
  readonly structuredOutput: true;
998
1119
  readonly temperatureSupported: true;
999
1120
  readonly provider: "google";
1121
+ }, {
1122
+ readonly type: "text";
1123
+ readonly modelName: "gemini-3.5-flash-lite";
1124
+ readonly description: "Most cost-effective Gemini model (GA July 21, 2026). Supersedes gemini-3.1-flash-lite. Built for high-volume, low-reasoning work (search, document processing, translation). Thinking support, 1M context window, 64K output.";
1125
+ readonly maxInputTokens: 1048576;
1126
+ readonly maxOutputTokens: 65536;
1127
+ readonly inputTokenCost: 0.3;
1128
+ readonly cachedInputTokenCost: 0.03;
1129
+ readonly outputTokenCost: 2.5;
1130
+ readonly inputAudioTokenCost: 0.5;
1131
+ readonly reasoning: {
1132
+ readonly levels: readonly ["minimal", "low", "medium", "high"];
1133
+ readonly defaultLevel: "minimal";
1134
+ readonly canDisable: false;
1135
+ readonly outputsThinking: true;
1136
+ readonly outputsSignatures: true;
1137
+ };
1138
+ readonly modalities: {
1139
+ readonly input: readonly ["text", "image", "video", "audio", "pdf"];
1140
+ readonly output: readonly ["text"];
1141
+ };
1142
+ readonly knowledge: "2025-01";
1143
+ readonly releaseDate: "2026-07-21";
1144
+ readonly lastUpdated: "2026-07-21";
1145
+ readonly family: "gemini-flash-lite";
1146
+ readonly openWeights: false;
1147
+ readonly structuredOutput: true;
1148
+ readonly temperatureSupported: true;
1149
+ readonly provider: "google";
1000
1150
  }, {
1001
1151
  readonly type: "text";
1002
1152
  readonly modelName: "gemini-3.1-flash-lite";
1003
- readonly description: "Most cost-effective Gemini 3.1 model (GA). Thinking support, 1M context window, 64K output. 2.5x faster TTFA and 45% faster output than 2.5 Flash.";
1153
+ readonly description: "Cost-effective Gemini 3.1 model (GA). Superseded by gemini-3.5-flash-lite. Thinking support, 1M context window, 64K output. 2.5x faster TTFA and 45% faster output than 2.5 Flash.";
1004
1154
  readonly maxInputTokens: 1048576;
1005
1155
  readonly maxOutputTokens: 65536;
1006
1156
  readonly inputTokenCost: 0.25;
@@ -1103,6 +1253,8 @@ export declare const textModels: readonly [{
1103
1253
  readonly input: readonly ["text", "image", "audio", "video", "pdf"];
1104
1254
  readonly output: readonly ["text"];
1105
1255
  };
1256
+ readonly supportedMimeTypes: readonly ["audio/wav", "audio/mpeg", "audio/aac", "audio/ogg", "audio/flac", "audio/aiff"];
1257
+ readonly maxBytes: 14000000;
1106
1258
  readonly knowledge: "2025-01";
1107
1259
  readonly releaseDate: "2025-06-17";
1108
1260
  readonly lastUpdated: "2025-06-17";
@@ -1558,6 +1710,21 @@ export declare const textModels: readonly [{
1558
1710
  readonly structuredOutput: true;
1559
1711
  readonly temperatureSupported: false;
1560
1712
  readonly provider: "openai-responses";
1713
+ }, {
1714
+ readonly type: "text";
1715
+ readonly modelName: "gpt-audio-1.5";
1716
+ readonly description: "OpenAI GA audio chat model (Chat Completions). Text+audio in, text+audio out.";
1717
+ readonly provider: "openai";
1718
+ readonly modalities: {
1719
+ readonly input: readonly ["text", "audio"];
1720
+ readonly output: readonly ["text", "audio"];
1721
+ };
1722
+ readonly inputTokenCost: 2.5;
1723
+ readonly outputTokenCost: 10;
1724
+ readonly inputAudioTokenCost: 32;
1725
+ readonly outputAudioTokenCost: 64;
1726
+ readonly maxInputTokens: 128000;
1727
+ readonly maxOutputTokens: 16384;
1561
1728
  }];
1562
1729
  export declare const imageModels: readonly [{
1563
1730
  readonly type: "image";
@@ -1610,6 +1777,7 @@ export declare const embeddingsModels: EmbeddingsModel[];
1610
1777
  export type TextModelName = (typeof textModels)[number]["modelName"];
1611
1778
  export type ImageModelName = (typeof imageModels)[number]["modelName"];
1612
1779
  export type SpeechToTextModelName = (typeof speechToTextModels)[number]["modelName"];
1780
+ export type TextToSpeechModelName = (typeof textToSpeechModels)[number]["modelName"];
1613
1781
  export type EmbeddingsModelName = (typeof embeddingsModels)[number]["modelName"];
1614
1782
  export type ModelName = string;
1615
1783
  export declare const hostedTools: HostedTool[];
@@ -1630,12 +1798,19 @@ export declare function getRegisteredModelData(): ModelDataBlob | null;
1630
1798
  */
1631
1799
  export declare function getAllModels(requestData?: ModelDataBlob): ModelType[];
1632
1800
  export declare function getModel(modelName: ModelName, requestData?: ModelDataBlob): ModelType | undefined;
1801
+ /**
1802
+ * Like `getModel`, but also matches on `provider`. Use this whenever a
1803
+ * modelName may collide across providers (the merge key everywhere else in
1804
+ * this module is `provider:modelName`) — plain `getModel` returns whichever
1805
+ * matching entry comes first and can silently pick the wrong provider.
1806
+ */
1807
+ export declare function getModelForProvider(provider: string, modelName: ModelName, requestData?: ModelDataBlob): ModelType | undefined;
1633
1808
  /**
1634
1809
  * Whether a model is known to accept the given input modality ("image", "pdf", …).
1635
1810
  * Returns undefined when the model is unknown or carries no `modalities` data —
1636
1811
  * callers should treat undefined as "don't gate".
1637
1812
  */
1638
- export declare function modelSupportsInputModality(modelName: ModelName, modality: string, requestData?: ModelDataBlob): boolean | undefined;
1813
+ export declare function modelSupportsInputModality(modelName: ModelName, modality: string, requestData?: ModelDataBlob, provider?: string): boolean | undefined;
1639
1814
  export declare function getHostedTools(opts?: {
1640
1815
  provider?: string;
1641
1816
  model?: string;
@@ -1647,5 +1822,12 @@ export declare function hostedToolPricingFor(tool: HostedTool, model?: string):
1647
1822
  export declare function isImageModel(model: ModelType): model is ImageModel;
1648
1823
  export declare function isTextModel(model: ModelType): model is TextModel;
1649
1824
  export declare function isSpeechToTextModel(model: ModelType): model is SpeechToTextModel;
1825
+ export declare function isTextToSpeechModel(model: ModelType): model is TextToSpeechModel;
1826
+ /** Audio-input constraints, readable off either a dedicated STT model or a
1827
+ * multimodal text model. Empty for any other model type. */
1828
+ export declare function audioInputConstraints(model: ModelType): {
1829
+ maxBytes?: number;
1830
+ supportedMimeTypes?: readonly string[];
1831
+ };
1650
1832
  export declare function isEmbeddingsModel(model: ModelType): model is EmbeddingsModel;
1651
1833
  export declare const ModelNameSchema: z.ZodString;