smoltalk 0.8.4 → 0.10.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (75) hide show
  1. package/README.md +145 -18
  2. package/dist/classes/ToolCall.js +18 -10
  3. package/dist/classes/message/AssistantMessage.d.ts +2 -0
  4. package/dist/classes/message/ToolMessage.js +13 -10
  5. package/dist/classes/message/UserMessage.d.ts +21 -0
  6. package/dist/classes/message/UserMessage.js +3 -0
  7. package/dist/classes/message/contentParts.d.ts +71 -2
  8. package/dist/classes/message/contentParts.js +6 -0
  9. package/dist/classes/message/index.d.ts +5 -2
  10. package/dist/classes/message/index.js +7 -0
  11. package/dist/classes/message/renderers/AnthropicRenderer.d.ts +2 -1
  12. package/dist/classes/message/renderers/AnthropicRenderer.js +3 -0
  13. package/dist/classes/message/renderers/GoogleRenderer.d.ts +2 -1
  14. package/dist/classes/message/renderers/GoogleRenderer.js +3 -0
  15. package/dist/classes/message/renderers/JSONRenderer.d.ts +2 -1
  16. package/dist/classes/message/renderers/JSONRenderer.js +4 -0
  17. package/dist/classes/message/renderers/OpenAIChatRenderer.d.ts +8 -1
  18. package/dist/classes/message/renderers/OpenAIChatRenderer.js +18 -0
  19. package/dist/classes/message/renderers/OpenAIResponsesRenderer.d.ts +2 -1
  20. package/dist/classes/message/renderers/OpenAIResponsesRenderer.js +3 -0
  21. package/dist/classes/message/renderers/PartRenderer.d.ts +3 -2
  22. package/dist/classes/message/renderers/PartRenderer.js +3 -0
  23. package/dist/client.js +1 -0
  24. package/dist/clients/anthropic.js +1 -1
  25. package/dist/clients/baseClient.d.ts +13 -1
  26. package/dist/clients/baseClient.js +36 -7
  27. package/dist/clients/google.d.ts +2 -0
  28. package/dist/clients/google.js +125 -3
  29. package/dist/clients/ollama.js +1 -1
  30. package/dist/clients/openai.d.ts +2 -1
  31. package/dist/clients/openai.js +15 -3
  32. package/dist/clients/openaiCompat.d.ts +2 -0
  33. package/dist/clients/openaiCompat.js +5 -0
  34. package/dist/clients/openaiResponses.js +1 -1
  35. package/dist/clients/resolveAttachments.d.ts +8 -4
  36. package/dist/clients/resolveAttachments.js +101 -50
  37. package/dist/embed.d.ts +4 -0
  38. package/dist/files.d.ts +1 -1
  39. package/dist/files.js +1 -1
  40. package/dist/image/google.js +2 -2
  41. package/dist/image/openai.js +3 -3
  42. package/dist/image.d.ts +1 -1
  43. package/dist/index.d.ts +10 -2
  44. package/dist/index.js +7 -1
  45. package/dist/model.d.ts +15 -4
  46. package/dist/model.js +48 -7
  47. package/dist/models.d.ts +143 -19
  48. package/dist/models.js +137 -30
  49. package/dist/speech/baseSpeechClient.d.ts +31 -0
  50. package/dist/speech/baseSpeechClient.js +98 -0
  51. package/dist/speech/openai.d.ts +6 -0
  52. package/dist/speech/openai.js +39 -0
  53. package/dist/speech.d.ts +40 -0
  54. package/dist/speech.js +57 -0
  55. package/dist/transcription/baseTranscriptionClient.d.ts +31 -0
  56. package/dist/transcription/baseTranscriptionClient.js +107 -0
  57. package/dist/transcription/openai.d.ts +6 -0
  58. package/dist/transcription/openai.js +59 -0
  59. package/dist/transcription.d.ts +51 -0
  60. package/dist/transcription.js +58 -0
  61. package/dist/types/tokenUsage.d.ts +4 -0
  62. package/dist/types/tokenUsage.js +4 -0
  63. package/dist/types.d.ts +3 -0
  64. package/dist/util/attachments.d.ts +1 -1
  65. package/dist/util/audioMime.d.ts +9 -0
  66. package/dist/util/audioMime.js +33 -0
  67. package/dist/util/{imageRef.d.ts → blobRef.d.ts} +9 -9
  68. package/dist/util/{imageRef.js → blobRef.js} +6 -13
  69. package/dist/util/mime.d.ts +21 -0
  70. package/dist/util/mime.js +52 -0
  71. package/dist/util/modalities.d.ts +6 -2
  72. package/dist/util/modalities.js +13 -15
  73. package/dist/util/provider.d.ts +2 -0
  74. package/dist/util/provider.js +1 -1
  75. package/package.json +1 -1
@@ -2,7 +2,7 @@ import OpenAI from "openai";
2
2
  import { toFile } from "openai/uploads";
3
3
  import { success, failure } from "../types/result.js";
4
4
  import { getModel, isImageModel } from "../models.js";
5
- import { normalizeImageRef } from "../util/imageRef.js";
5
+ import { normalizeBlob } from "../util/blobRef.js";
6
6
  import { COST_DECIMAL_PLACES, omitUndefined, round, tokenCost, } from "../util/util.js";
7
7
  export async function openaiImage(input, config, apiKey) {
8
8
  try {
@@ -51,11 +51,11 @@ function buildBaseParams(config, prompt) {
51
51
  }
52
52
  async function callEdit(client, baseParams, normalized) {
53
53
  const imageFiles = await Promise.all((normalized.images ?? []).map(async (ref, i) => {
54
- const n = await normalizeImageRef(ref);
54
+ const n = await normalizeBlob(ref);
55
55
  return toFileFor(n, `image-${i}`);
56
56
  }));
57
57
  const maskFile = normalized.mask
58
- ? await toFileFor(await normalizeImageRef(normalized.mask), "mask")
58
+ ? await toFileFor(await normalizeBlob(normalized.mask), "mask")
59
59
  : undefined;
60
60
  return client.images.edit(omitUndefined({
61
61
  ...baseParams,
package/dist/image.d.ts CHANGED
@@ -2,7 +2,7 @@ import type { ModelDataBlob } from "./modelData.js";
2
2
  import { Result } from "./types/result.js";
3
3
  import { TokenUsage } from "./types/tokenUsage.js";
4
4
  import { CostEstimate } from "./types/costEstimate.js";
5
- import { ImageRef } from "./util/imageRef.js";
5
+ import { ImageRef } from "./util/blobRef.js";
6
6
  export { ImageRef };
7
7
  export type ImageInput = string | {
8
8
  prompt: string;
package/dist/index.d.ts CHANGED
@@ -13,8 +13,16 @@ export * from "./embed.js";
13
13
  export * from "./image.js";
14
14
  export { uploadFile, deleteFile, registerFileProvider, DEFAULT_UPLOAD_BYTES } from "./files.js";
15
15
  export type { UploadFileOptions, FileProviderContext, FileProvider } from "./files.js";
16
- export { normalizeImageRef, loadBlob } from "./util/imageRef.js";
17
- export type { BlobRef } from "./util/imageRef.js";
16
+ export { normalizeBlob, loadBlob } from "./util/blobRef.js";
17
+ export type { BlobRef } from "./util/blobRef.js";
18
18
  export { getLogger, EgonLog } from "./util/logger.js";
19
19
  export type { LogLevel } from "./util/logger.js";
20
20
  export { redactAttachments } from "./util/redact.js";
21
+ export { transcribe, registerTranscriptionProvider, DEFAULT_TRANSCRIBE_BYTES, } from "./transcription.js";
22
+ export type { TranscribeOptions, TranscriptionSegment, TranscriptionWord, TranscriptionResult, TranscriptionClientClass, } from "./transcription.js";
23
+ export { BaseTranscriptionClient } from "./transcription/baseTranscriptionClient.js";
24
+ export type { TranscriptionClientConfig } from "./transcription/baseTranscriptionClient.js";
25
+ export { speak, registerSpeechProvider, } from "./speech.js";
26
+ export type { SpeakOptions, PcmAudioMetadata, SpeechResult, SpeechClientClass, } from "./speech.js";
27
+ export { BaseSpeechClient } from "./speech/baseSpeechClient.js";
28
+ export type { SpeechClientConfig } from "./speech/baseSpeechClient.js";
package/dist/index.js CHANGED
@@ -13,6 +13,12 @@ export * from "./embed.js";
13
13
  export * from "./image.js";
14
14
  // Explicit (not `export *`) so the test-only `_resetForTests` stays off the public surface.
15
15
  export { uploadFile, deleteFile, registerFileProvider, DEFAULT_UPLOAD_BYTES } from "./files.js";
16
- export { normalizeImageRef, loadBlob } from "./util/imageRef.js";
16
+ export { normalizeBlob, loadBlob } from "./util/blobRef.js";
17
17
  export { getLogger, EgonLog } from "./util/logger.js";
18
18
  export { redactAttachments } from "./util/redact.js";
19
+ // Explicit (not `export *`) so internal factories and test helpers stay private.
20
+ export { transcribe, registerTranscriptionProvider, DEFAULT_TRANSCRIBE_BYTES, } from "./transcription.js";
21
+ export { BaseTranscriptionClient } from "./transcription/baseTranscriptionClient.js";
22
+ // Explicit (not `export *`) so internal factories and test helpers stay private.
23
+ export { speak, registerSpeechProvider, } from "./speech.js";
24
+ export { BaseSpeechClient } from "./speech/baseSpeechClient.js";
package/dist/model.d.ts CHANGED
@@ -1,19 +1,22 @@
1
- import { ModelName, Provider } from "./models.js";
1
+ import { ModelName, ModelType } from "./models.js";
2
2
  import { ModelLike } from "./types.js";
3
3
  import type { ModelDataBlob } from "./modelData.js";
4
+ import type { CostEstimate } from "./types/costEstimate.js";
4
5
  export declare class Model {
5
6
  private model;
6
7
  private provider?;
7
8
  private modelData?;
8
- constructor(model: ModelName, provider?: Provider, modelData?: ModelDataBlob);
9
+ constructor(model: ModelName, provider?: string, modelData?: ModelDataBlob);
9
10
  getModel(): ModelName;
10
- getProvider(): Provider | undefined;
11
+ getProvider(): string | undefined;
11
12
  private lookupProvider;
12
13
  calculateCost(usage: {
13
14
  inputTokens: number;
14
15
  outputTokens: number;
15
16
  cachedInputTokens?: number;
16
17
  cacheCreationInputTokens?: number;
18
+ inputAudioTokens?: number;
19
+ outputAudioTokens?: number;
17
20
  }): {
18
21
  inputCost: number;
19
22
  outputCost: number;
@@ -24,5 +27,13 @@ export declare class Model {
24
27
  } | null;
25
28
  toString(): string;
26
29
  toJSON(): ModelName;
27
- static create(model: ModelLike, provider?: Provider, modelData?: ModelDataBlob): Model;
30
+ static create(model: ModelLike, provider?: string, modelData?: ModelDataBlob): Model;
28
31
  }
32
+ /**
33
+ * Per-minute STT pricing from a registry entry. Returns undefined (cost
34
+ * omitted, no error) when the model, rate, or duration is unknown — a rate of
35
+ * 0 still yields a present zero cost.
36
+ */
37
+ export declare function calculateTranscriptionCost(model: ModelType | undefined, durationSeconds: number | undefined): CostEstimate | undefined;
38
+ /** Per-code-point TTS pricing from a registry entry; same omission semantics. */
39
+ export declare function calculateSpeechCost(model: ModelType | undefined, charCount: number): CostEstimate | undefined;
package/dist/model.js CHANGED
@@ -1,6 +1,7 @@
1
- import { getModel, isTextModel, ModelNameSchema } from "./models.js";
1
+ import { getModel, getModelForProvider, isSpeechToTextModel, isTextModel, isTextToSpeechModel, ModelNameSchema, } from "./models.js";
2
2
  import { SmolError } from "./smolError.js";
3
3
  import { round } from "./util/util.js";
4
+ const TOKEN_COST_UNIT = 1_000_000;
4
5
  export class Model {
5
6
  model;
6
7
  provider;
@@ -24,7 +25,13 @@ export class Model {
24
25
  return modelInfo ? modelInfo.provider : undefined;
25
26
  }
26
27
  calculateCost(usage) {
27
- const model = getModel(this.model, this.modelData);
28
+ let model;
29
+ if (this.provider !== undefined) {
30
+ model = getModelForProvider(this.provider, this.model, this.modelData);
31
+ }
32
+ else {
33
+ model = getModel(this.model, this.modelData);
34
+ }
28
35
  if (!model || !isTextModel(model)) {
29
36
  return null;
30
37
  }
@@ -35,8 +42,15 @@ export class Model {
35
42
  // full input rate so totalCost stays honest.
36
43
  const cachedRate = model.cachedInputTokenCost ?? model.inputTokenCost ?? 0;
37
44
  const cacheCreationRate = model.cacheCreationInputTokenCost ?? model.inputTokenCost ?? 0;
38
- const inputCost = round((usage.inputTokens * (model.inputTokenCost || 0)) / 1_000_000, 6);
39
- const outputCost = round((usage.outputTokens * (model.outputTokenCost || 0)) / 1_000_000, 6);
45
+ const inputCost = round((usage.inputTokens * (model.inputTokenCost || 0)) / TOKEN_COST_UNIT, 6);
46
+ const outputCost = round((usage.outputTokens * (model.outputTokenCost || 0)) / TOKEN_COST_UNIT, 6);
47
+ const audioInTokens = usage.inputAudioTokens ?? 0;
48
+ const audioOutTokens = usage.outputAudioTokens ?? 0;
49
+ // Fall back to the text rate if no audio rate is defined so the total stays honest.
50
+ const audioInRate = model.inputAudioTokenCost ?? model.inputTokenCost ?? 0;
51
+ const audioOutRate = model.outputAudioTokenCost ?? model.outputTokenCost ?? 0;
52
+ const audioInCost = round((audioInTokens * audioInRate) / TOKEN_COST_UNIT, 6);
53
+ const audioOutCost = round((audioOutTokens * audioOutRate) / TOKEN_COST_UNIT, 6);
40
54
  // Only expose cachedInputCost / cacheCreationInputCost when the model
41
55
  // actually has a distinct discount price. Otherwise, fold those dollars
42
56
  // into inputCost so the user isn't misled by a $0 cached field.
@@ -61,14 +75,15 @@ export class Model {
61
75
  foldedInputDollars += dollars;
62
76
  }
63
77
  }
64
- const finalInputCost = round(inputCost + foldedInputDollars, 6);
78
+ const finalInputCost = round(inputCost + foldedInputDollars + audioInCost, 6);
79
+ const finalOutputCost = round(outputCost + audioOutCost, 6);
65
80
  const totalCost = round(finalInputCost +
66
- outputCost +
81
+ finalOutputCost +
67
82
  (cachedInputCost || 0) +
68
83
  (cacheCreationInputCost || 0), 6);
69
84
  return {
70
85
  inputCost: finalInputCost,
71
- outputCost,
86
+ outputCost: finalOutputCost,
72
87
  cachedInputCost,
73
88
  cacheCreationInputCost,
74
89
  totalCost,
@@ -88,3 +103,29 @@ export class Model {
88
103
  return new Model(model, provider, modelData);
89
104
  }
90
105
  }
106
+ /**
107
+ * Per-minute STT pricing from a registry entry. Returns undefined (cost
108
+ * omitted, no error) when the model, rate, or duration is unknown — a rate of
109
+ * 0 still yields a present zero cost.
110
+ */
111
+ export function calculateTranscriptionCost(model, durationSeconds) {
112
+ if (model === undefined || !isSpeechToTextModel(model)) {
113
+ return undefined;
114
+ }
115
+ if (model.perMinuteCost === undefined || durationSeconds === undefined || durationSeconds === null) {
116
+ return undefined;
117
+ }
118
+ const inputCost = round((durationSeconds / 60) * model.perMinuteCost, 6);
119
+ return { inputCost, outputCost: 0, totalCost: inputCost, currency: "USD" };
120
+ }
121
+ /** Per-code-point TTS pricing from a registry entry; same omission semantics. */
122
+ export function calculateSpeechCost(model, charCount) {
123
+ if (model === undefined || !isTextToSpeechModel(model)) {
124
+ return undefined;
125
+ }
126
+ if (model.perCharacterCost === undefined) {
127
+ return undefined;
128
+ }
129
+ const inputCost = round(charCount * model.perCharacterCost, 6);
130
+ return { inputCost, outputCost: 0, totalCost: inputCost, currency: "USD" };
131
+ }
package/dist/models.d.ts CHANGED
@@ -34,6 +34,23 @@ export type BaseModel = {
34
34
  export type SpeechToTextModel = BaseModel & {
35
35
  type: "speech-to-text";
36
36
  perMinuteCost?: number;
37
+ /** Canonical MIME types accepted after alias normalization through AUDIO_FORMATS. */
38
+ supportedMimeTypes?: readonly string[];
39
+ /** Provider upload cap in bytes. */
40
+ maxBytes?: number;
41
+ };
42
+ export type TextToSpeechModel = BaseModel & {
43
+ type: "text-to-speech";
44
+ perCharacterCost?: number;
45
+ /** Input cap in Unicode code points. */
46
+ maxInputChars?: number;
47
+ /** Accepted values for the speed option. */
48
+ speedRange?: {
49
+ min: number;
50
+ max: number;
51
+ };
52
+ /** Output formats the provider can render for this model. */
53
+ formats?: readonly string[];
37
54
  };
38
55
  export type ImageModel = BaseModel & {
39
56
  type: "image";
@@ -91,12 +108,37 @@ export type EmbeddingsModel = {
91
108
  provider: string;
92
109
  tokenCost?: number;
93
110
  };
94
- export type ModelType = SpeechToTextModel | TextModel | EmbeddingsModel | ImageModel;
111
+ export type ModelType = SpeechToTextModel | TextToSpeechModel | TextModel | EmbeddingsModel | ImageModel;
95
112
  export declare const speechToTextModels: readonly [{
96
113
  readonly type: "speech-to-text";
97
- readonly modelName: "whisper-web";
114
+ readonly modelName: "whisper-1";
98
115
  readonly perMinuteCost: 0.006;
99
116
  readonly provider: "openai";
117
+ readonly supportedMimeTypes: readonly ["audio/flac", "audio/mpeg", "audio/mp4", "audio/m4a", "audio/ogg", "audio/wav", "audio/webm"];
118
+ readonly maxBytes: number;
119
+ }];
120
+ export declare const textToSpeechModels: readonly [{
121
+ readonly type: "text-to-speech";
122
+ readonly modelName: "tts-1";
123
+ readonly perCharacterCost: 0.000015;
124
+ readonly provider: "openai";
125
+ readonly maxInputChars: 4096;
126
+ readonly speedRange: {
127
+ readonly min: 0.25;
128
+ readonly max: 4;
129
+ };
130
+ readonly formats: readonly ["mp3", "opus", "aac", "flac", "wav", "pcm"];
131
+ }, {
132
+ readonly type: "text-to-speech";
133
+ readonly modelName: "tts-1-hd";
134
+ readonly perCharacterCost: 0.00003;
135
+ readonly provider: "openai";
136
+ readonly maxInputChars: 4096;
137
+ readonly speedRange: {
138
+ readonly min: 0.25;
139
+ readonly max: 4;
140
+ };
141
+ readonly formats: readonly ["mp3", "opus", "aac", "flac", "wav", "pcm"];
100
142
  }];
101
143
  export declare const textModels: readonly [{
102
144
  readonly type: "text";
@@ -810,16 +852,16 @@ export declare const textModels: readonly [{
810
852
  }, {
811
853
  readonly type: "text";
812
854
  readonly modelName: "gpt-5.6-terra";
813
- readonly description: "GPT-5.6 Terra balances capability and cost — competitive with GPT-5.5 at roughly half the price. 1M context window. Standard pricing for ≤272K tokens, 2x input/1.5x output for >272K. Knowledge cutoff: February 2026.";
855
+ readonly description: "GPT-5.6 Terra balances capability and cost. 1M context window. Standard pricing for ≤272K tokens, 2x input/1.5x output for >272K. Reflects the July 30, 2026 price cut. Knowledge cutoff: February 2026.";
814
856
  readonly maxInputTokens: 1050000;
815
857
  readonly maxOutputTokens: 128000;
816
- readonly inputTokenCost: 2.5;
817
- readonly cachedInputTokenCost: 0.25;
818
- readonly outputTokenCost: 15;
858
+ readonly inputTokenCost: 2;
859
+ readonly cachedInputTokenCost: 0.2;
860
+ readonly outputTokenCost: 12;
819
861
  readonly longContext: {
820
- readonly inputTokenCost: 5;
821
- readonly cachedInputTokenCost: 0.5;
822
- readonly outputTokenCost: 22.5;
862
+ readonly inputTokenCost: 4;
863
+ readonly cachedInputTokenCost: 0.4;
864
+ readonly outputTokenCost: 18;
823
865
  readonly thresholdTokens: 200000;
824
866
  };
825
867
  readonly reasoning: {
@@ -844,16 +886,16 @@ export declare const textModels: readonly [{
844
886
  }, {
845
887
  readonly type: "text";
846
888
  readonly modelName: "gpt-5.6-luna";
847
- readonly description: "GPT-5.6 Luna is the fast, most affordable member of the GPT-5.6 family. 1M context window. Standard pricing for ≤272K tokens, 2x input/1.5x output for >272K. Knowledge cutoff: February 2026.";
889
+ readonly description: "GPT-5.6 Luna is the fast, most affordable member of the GPT-5.6 family. 1M context window. Standard pricing for ≤272K tokens, 2x input/1.5x output for >272K. Reflects the July 30, 2026 price cut (input/output down ~80%). Knowledge cutoff: February 2026.";
848
890
  readonly maxInputTokens: 1050000;
849
891
  readonly maxOutputTokens: 128000;
850
- readonly inputTokenCost: 1;
851
- readonly cachedInputTokenCost: 0.1;
852
- readonly outputTokenCost: 6;
892
+ readonly inputTokenCost: 0.2;
893
+ readonly cachedInputTokenCost: 0.02;
894
+ readonly outputTokenCost: 1.2;
853
895
  readonly longContext: {
854
- readonly inputTokenCost: 2;
855
- readonly cachedInputTokenCost: 0.2;
856
- readonly outputTokenCost: 9;
896
+ readonly inputTokenCost: 0.4;
897
+ readonly cachedInputTokenCost: 0.04;
898
+ readonly outputTokenCost: 1.8;
857
899
  readonly thresholdTokens: 200000;
858
900
  };
859
901
  readonly reasoning: {
@@ -938,10 +980,39 @@ export declare const textModels: readonly [{
938
980
  readonly temperatureSupported: true;
939
981
  readonly disabled: true;
940
982
  readonly provider: "google";
983
+ }, {
984
+ readonly type: "text";
985
+ readonly modelName: "gemini-3.6-flash";
986
+ readonly description: "Latest Gemini 3.6 Flash model (GA July 21, 2026). Supersedes Gemini 3.5 Flash with better token efficiency and agentic planning at a lower output price ($7.50 vs $9.00/1M). 1M context window, 64K output. Context caching: $0.15/1M read.";
987
+ readonly maxInputTokens: 1048576;
988
+ readonly maxOutputTokens: 65536;
989
+ readonly inputTokenCost: 1.5;
990
+ readonly cachedInputTokenCost: 0.15;
991
+ readonly outputTokenCost: 7.5;
992
+ readonly inputAudioTokenCost: 1.5;
993
+ readonly reasoning: {
994
+ readonly levels: readonly ["minimal", "low", "medium", "high"];
995
+ readonly defaultLevel: "high";
996
+ readonly canDisable: false;
997
+ readonly outputsThinking: true;
998
+ readonly outputsSignatures: true;
999
+ };
1000
+ readonly modalities: {
1001
+ readonly input: readonly ["text", "image", "video", "audio", "pdf"];
1002
+ readonly output: readonly ["text"];
1003
+ };
1004
+ readonly knowledge: "2026-03";
1005
+ readonly releaseDate: "2026-07-21";
1006
+ readonly lastUpdated: "2026-07-21";
1007
+ readonly family: "gemini-flash";
1008
+ readonly openWeights: false;
1009
+ readonly structuredOutput: true;
1010
+ readonly temperatureSupported: true;
1011
+ readonly provider: "google";
941
1012
  }, {
942
1013
  readonly type: "text";
943
1014
  readonly modelName: "gemini-3.5-flash";
944
- readonly description: "Latest Gemini 3.5 Flash model (GA May 2026). Outperforms Gemini 3.1 Pro on coding and agentic suites at 4x the speed. 1M context window, 64K output. Context caching: $0.15/1M read.";
1015
+ readonly description: "Gemini 3.5 Flash (GA May 2026). Superseded by gemini-3.6-flash. Outperforms Gemini 3.1 Pro on coding and agentic suites at 4x the speed. 1M context window, 64K output. Context caching: $0.15/1M read.";
945
1016
  readonly maxInputTokens: 1048576;
946
1017
  readonly maxOutputTokens: 65536;
947
1018
  readonly inputTokenCost: 1.5;
@@ -997,10 +1068,39 @@ export declare const textModels: readonly [{
997
1068
  readonly structuredOutput: true;
998
1069
  readonly temperatureSupported: true;
999
1070
  readonly provider: "google";
1071
+ }, {
1072
+ readonly type: "text";
1073
+ readonly modelName: "gemini-3.5-flash-lite";
1074
+ readonly description: "Most cost-effective Gemini model (GA July 21, 2026). Supersedes gemini-3.1-flash-lite. Built for high-volume, low-reasoning work (search, document processing, translation). Thinking support, 1M context window, 64K output.";
1075
+ readonly maxInputTokens: 1048576;
1076
+ readonly maxOutputTokens: 65536;
1077
+ readonly inputTokenCost: 0.3;
1078
+ readonly cachedInputTokenCost: 0.03;
1079
+ readonly outputTokenCost: 2.5;
1080
+ readonly inputAudioTokenCost: 0.5;
1081
+ readonly reasoning: {
1082
+ readonly levels: readonly ["minimal", "low", "medium", "high"];
1083
+ readonly defaultLevel: "minimal";
1084
+ readonly canDisable: false;
1085
+ readonly outputsThinking: true;
1086
+ readonly outputsSignatures: true;
1087
+ };
1088
+ readonly modalities: {
1089
+ readonly input: readonly ["text", "image", "video", "audio", "pdf"];
1090
+ readonly output: readonly ["text"];
1091
+ };
1092
+ readonly knowledge: "2025-01";
1093
+ readonly releaseDate: "2026-07-21";
1094
+ readonly lastUpdated: "2026-07-21";
1095
+ readonly family: "gemini-flash-lite";
1096
+ readonly openWeights: false;
1097
+ readonly structuredOutput: true;
1098
+ readonly temperatureSupported: true;
1099
+ readonly provider: "google";
1000
1100
  }, {
1001
1101
  readonly type: "text";
1002
1102
  readonly modelName: "gemini-3.1-flash-lite";
1003
- readonly description: "Most cost-effective Gemini 3.1 model (GA). Thinking support, 1M context window, 64K output. 2.5x faster TTFA and 45% faster output than 2.5 Flash.";
1103
+ readonly description: "Cost-effective Gemini 3.1 model (GA). Superseded by gemini-3.5-flash-lite. Thinking support, 1M context window, 64K output. 2.5x faster TTFA and 45% faster output than 2.5 Flash.";
1004
1104
  readonly maxInputTokens: 1048576;
1005
1105
  readonly maxOutputTokens: 65536;
1006
1106
  readonly inputTokenCost: 0.25;
@@ -1558,6 +1658,21 @@ export declare const textModels: readonly [{
1558
1658
  readonly structuredOutput: true;
1559
1659
  readonly temperatureSupported: false;
1560
1660
  readonly provider: "openai-responses";
1661
+ }, {
1662
+ readonly type: "text";
1663
+ readonly modelName: "gpt-audio-1.5";
1664
+ readonly description: "OpenAI GA audio chat model (Chat Completions). Text+audio in, text+audio out.";
1665
+ readonly provider: "openai";
1666
+ readonly modalities: {
1667
+ readonly input: readonly ["text", "audio"];
1668
+ readonly output: readonly ["text", "audio"];
1669
+ };
1670
+ readonly inputTokenCost: 2.5;
1671
+ readonly outputTokenCost: 10;
1672
+ readonly inputAudioTokenCost: 32;
1673
+ readonly outputAudioTokenCost: 64;
1674
+ readonly maxInputTokens: 128000;
1675
+ readonly maxOutputTokens: 16384;
1561
1676
  }];
1562
1677
  export declare const imageModels: readonly [{
1563
1678
  readonly type: "image";
@@ -1610,6 +1725,7 @@ export declare const embeddingsModels: EmbeddingsModel[];
1610
1725
  export type TextModelName = (typeof textModels)[number]["modelName"];
1611
1726
  export type ImageModelName = (typeof imageModels)[number]["modelName"];
1612
1727
  export type SpeechToTextModelName = (typeof speechToTextModels)[number]["modelName"];
1728
+ export type TextToSpeechModelName = (typeof textToSpeechModels)[number]["modelName"];
1613
1729
  export type EmbeddingsModelName = (typeof embeddingsModels)[number]["modelName"];
1614
1730
  export type ModelName = string;
1615
1731
  export declare const hostedTools: HostedTool[];
@@ -1630,12 +1746,19 @@ export declare function getRegisteredModelData(): ModelDataBlob | null;
1630
1746
  */
1631
1747
  export declare function getAllModels(requestData?: ModelDataBlob): ModelType[];
1632
1748
  export declare function getModel(modelName: ModelName, requestData?: ModelDataBlob): ModelType | undefined;
1749
+ /**
1750
+ * Like `getModel`, but also matches on `provider`. Use this whenever a
1751
+ * modelName may collide across providers (the merge key everywhere else in
1752
+ * this module is `provider:modelName`) — plain `getModel` returns whichever
1753
+ * matching entry comes first and can silently pick the wrong provider.
1754
+ */
1755
+ export declare function getModelForProvider(provider: string, modelName: ModelName, requestData?: ModelDataBlob): ModelType | undefined;
1633
1756
  /**
1634
1757
  * Whether a model is known to accept the given input modality ("image", "pdf", …).
1635
1758
  * Returns undefined when the model is unknown or carries no `modalities` data —
1636
1759
  * callers should treat undefined as "don't gate".
1637
1760
  */
1638
- export declare function modelSupportsInputModality(modelName: ModelName, modality: string, requestData?: ModelDataBlob): boolean | undefined;
1761
+ export declare function modelSupportsInputModality(modelName: ModelName, modality: string, requestData?: ModelDataBlob, provider?: string): boolean | undefined;
1639
1762
  export declare function getHostedTools(opts?: {
1640
1763
  provider?: string;
1641
1764
  model?: string;
@@ -1647,5 +1770,6 @@ export declare function hostedToolPricingFor(tool: HostedTool, model?: string):
1647
1770
  export declare function isImageModel(model: ModelType): model is ImageModel;
1648
1771
  export declare function isTextModel(model: ModelType): model is TextModel;
1649
1772
  export declare function isSpeechToTextModel(model: ModelType): model is SpeechToTextModel;
1773
+ export declare function isTextToSpeechModel(model: ModelType): model is TextToSpeechModel;
1650
1774
  export declare function isEmbeddingsModel(model: ModelType): model is EmbeddingsModel;
1651
1775
  export declare const ModelNameSchema: z.ZodString;