smoltalk 0.9.0 → 0.10.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (72) hide show
  1. package/README.md +141 -7
  2. package/dist/classes/message/AssistantMessage.d.ts +2 -0
  3. package/dist/classes/message/UserMessage.d.ts +21 -0
  4. package/dist/classes/message/UserMessage.js +3 -0
  5. package/dist/classes/message/contentParts.d.ts +71 -2
  6. package/dist/classes/message/contentParts.js +6 -0
  7. package/dist/classes/message/index.d.ts +5 -2
  8. package/dist/classes/message/index.js +7 -0
  9. package/dist/classes/message/renderers/AnthropicRenderer.d.ts +2 -1
  10. package/dist/classes/message/renderers/AnthropicRenderer.js +3 -0
  11. package/dist/classes/message/renderers/GoogleRenderer.d.ts +2 -1
  12. package/dist/classes/message/renderers/GoogleRenderer.js +3 -0
  13. package/dist/classes/message/renderers/JSONRenderer.d.ts +2 -1
  14. package/dist/classes/message/renderers/JSONRenderer.js +4 -0
  15. package/dist/classes/message/renderers/OpenAIChatRenderer.d.ts +8 -1
  16. package/dist/classes/message/renderers/OpenAIChatRenderer.js +18 -0
  17. package/dist/classes/message/renderers/OpenAIResponsesRenderer.d.ts +2 -1
  18. package/dist/classes/message/renderers/OpenAIResponsesRenderer.js +3 -0
  19. package/dist/classes/message/renderers/PartRenderer.d.ts +3 -2
  20. package/dist/classes/message/renderers/PartRenderer.js +3 -0
  21. package/dist/client.js +1 -0
  22. package/dist/clients/anthropic.js +1 -1
  23. package/dist/clients/baseClient.d.ts +13 -1
  24. package/dist/clients/baseClient.js +36 -7
  25. package/dist/clients/google.js +1 -1
  26. package/dist/clients/ollama.js +1 -1
  27. package/dist/clients/openai.d.ts +2 -1
  28. package/dist/clients/openai.js +15 -3
  29. package/dist/clients/openaiCompat.d.ts +2 -0
  30. package/dist/clients/openaiCompat.js +5 -0
  31. package/dist/clients/openaiResponses.js +1 -1
  32. package/dist/clients/resolveAttachments.d.ts +8 -4
  33. package/dist/clients/resolveAttachments.js +101 -50
  34. package/dist/embed.d.ts +4 -0
  35. package/dist/files.d.ts +1 -1
  36. package/dist/files.js +1 -1
  37. package/dist/image/google.js +2 -2
  38. package/dist/image/openai.js +3 -3
  39. package/dist/image.d.ts +1 -1
  40. package/dist/index.d.ts +10 -2
  41. package/dist/index.js +7 -1
  42. package/dist/model.d.ts +15 -4
  43. package/dist/model.js +48 -7
  44. package/dist/models.d.ts +143 -19
  45. package/dist/models.js +137 -30
  46. package/dist/speech/baseSpeechClient.d.ts +31 -0
  47. package/dist/speech/baseSpeechClient.js +98 -0
  48. package/dist/speech/openai.d.ts +6 -0
  49. package/dist/speech/openai.js +39 -0
  50. package/dist/speech.d.ts +40 -0
  51. package/dist/speech.js +57 -0
  52. package/dist/transcription/baseTranscriptionClient.d.ts +31 -0
  53. package/dist/transcription/baseTranscriptionClient.js +107 -0
  54. package/dist/transcription/openai.d.ts +6 -0
  55. package/dist/transcription/openai.js +59 -0
  56. package/dist/transcription.d.ts +51 -0
  57. package/dist/transcription.js +58 -0
  58. package/dist/types/tokenUsage.d.ts +4 -0
  59. package/dist/types/tokenUsage.js +4 -0
  60. package/dist/types.d.ts +3 -0
  61. package/dist/util/attachments.d.ts +1 -1
  62. package/dist/util/audioMime.d.ts +9 -0
  63. package/dist/util/audioMime.js +33 -0
  64. package/dist/util/{imageRef.d.ts → blobRef.d.ts} +9 -9
  65. package/dist/util/{imageRef.js → blobRef.js} +6 -13
  66. package/dist/util/mime.d.ts +21 -0
  67. package/dist/util/mime.js +52 -0
  68. package/dist/util/modalities.d.ts +6 -2
  69. package/dist/util/modalities.js +13 -15
  70. package/dist/util/provider.d.ts +2 -0
  71. package/dist/util/provider.js +1 -1
  72. package/package.json +1 -1
package/dist/models.d.ts CHANGED
@@ -34,6 +34,23 @@ export type BaseModel = {
34
34
  export type SpeechToTextModel = BaseModel & {
35
35
  type: "speech-to-text";
36
36
  perMinuteCost?: number;
37
+ /** Canonical MIME types accepted after alias normalization through AUDIO_FORMATS. */
38
+ supportedMimeTypes?: readonly string[];
39
+ /** Provider upload cap in bytes. */
40
+ maxBytes?: number;
41
+ };
42
+ export type TextToSpeechModel = BaseModel & {
43
+ type: "text-to-speech";
44
+ perCharacterCost?: number;
45
+ /** Input cap in Unicode code points. */
46
+ maxInputChars?: number;
47
+ /** Accepted values for the speed option. */
48
+ speedRange?: {
49
+ min: number;
50
+ max: number;
51
+ };
52
+ /** Output formats the provider can render for this model. */
53
+ formats?: readonly string[];
37
54
  };
38
55
  export type ImageModel = BaseModel & {
39
56
  type: "image";
@@ -91,12 +108,37 @@ export type EmbeddingsModel = {
91
108
  provider: string;
92
109
  tokenCost?: number;
93
110
  };
94
- export type ModelType = SpeechToTextModel | TextModel | EmbeddingsModel | ImageModel;
111
+ export type ModelType = SpeechToTextModel | TextToSpeechModel | TextModel | EmbeddingsModel | ImageModel;
95
112
  export declare const speechToTextModels: readonly [{
96
113
  readonly type: "speech-to-text";
97
- readonly modelName: "whisper-web";
114
+ readonly modelName: "whisper-1";
98
115
  readonly perMinuteCost: 0.006;
99
116
  readonly provider: "openai";
117
+ readonly supportedMimeTypes: readonly ["audio/flac", "audio/mpeg", "audio/mp4", "audio/m4a", "audio/ogg", "audio/wav", "audio/webm"];
118
+ readonly maxBytes: number;
119
+ }];
120
+ export declare const textToSpeechModels: readonly [{
121
+ readonly type: "text-to-speech";
122
+ readonly modelName: "tts-1";
123
+ readonly perCharacterCost: 0.000015;
124
+ readonly provider: "openai";
125
+ readonly maxInputChars: 4096;
126
+ readonly speedRange: {
127
+ readonly min: 0.25;
128
+ readonly max: 4;
129
+ };
130
+ readonly formats: readonly ["mp3", "opus", "aac", "flac", "wav", "pcm"];
131
+ }, {
132
+ readonly type: "text-to-speech";
133
+ readonly modelName: "tts-1-hd";
134
+ readonly perCharacterCost: 0.00003;
135
+ readonly provider: "openai";
136
+ readonly maxInputChars: 4096;
137
+ readonly speedRange: {
138
+ readonly min: 0.25;
139
+ readonly max: 4;
140
+ };
141
+ readonly formats: readonly ["mp3", "opus", "aac", "flac", "wav", "pcm"];
100
142
  }];
101
143
  export declare const textModels: readonly [{
102
144
  readonly type: "text";
@@ -810,16 +852,16 @@ export declare const textModels: readonly [{
810
852
  }, {
811
853
  readonly type: "text";
812
854
  readonly modelName: "gpt-5.6-terra";
813
- readonly description: "GPT-5.6 Terra balances capability and cost — competitive with GPT-5.5 at roughly half the price. 1M context window. Standard pricing for ≤272K tokens, 2x input/1.5x output for >272K. Knowledge cutoff: February 2026.";
855
+ readonly description: "GPT-5.6 Terra balances capability and cost. 1M context window. Standard pricing for ≤272K tokens, 2x input/1.5x output for >272K. Reflects the July 30, 2026 price cut. Knowledge cutoff: February 2026.";
814
856
  readonly maxInputTokens: 1050000;
815
857
  readonly maxOutputTokens: 128000;
816
- readonly inputTokenCost: 2.5;
817
- readonly cachedInputTokenCost: 0.25;
818
- readonly outputTokenCost: 15;
858
+ readonly inputTokenCost: 2;
859
+ readonly cachedInputTokenCost: 0.2;
860
+ readonly outputTokenCost: 12;
819
861
  readonly longContext: {
820
- readonly inputTokenCost: 5;
821
- readonly cachedInputTokenCost: 0.5;
822
- readonly outputTokenCost: 22.5;
862
+ readonly inputTokenCost: 4;
863
+ readonly cachedInputTokenCost: 0.4;
864
+ readonly outputTokenCost: 18;
823
865
  readonly thresholdTokens: 200000;
824
866
  };
825
867
  readonly reasoning: {
@@ -844,16 +886,16 @@ export declare const textModels: readonly [{
844
886
  }, {
845
887
  readonly type: "text";
846
888
  readonly modelName: "gpt-5.6-luna";
847
- readonly description: "GPT-5.6 Luna is the fast, most affordable member of the GPT-5.6 family. 1M context window. Standard pricing for ≤272K tokens, 2x input/1.5x output for >272K. Knowledge cutoff: February 2026.";
889
+ readonly description: "GPT-5.6 Luna is the fast, most affordable member of the GPT-5.6 family. 1M context window. Standard pricing for ≤272K tokens, 2x input/1.5x output for >272K. Reflects the July 30, 2026 price cut (input/output down ~80%). Knowledge cutoff: February 2026.";
848
890
  readonly maxInputTokens: 1050000;
849
891
  readonly maxOutputTokens: 128000;
850
- readonly inputTokenCost: 1;
851
- readonly cachedInputTokenCost: 0.1;
852
- readonly outputTokenCost: 6;
892
+ readonly inputTokenCost: 0.2;
893
+ readonly cachedInputTokenCost: 0.02;
894
+ readonly outputTokenCost: 1.2;
853
895
  readonly longContext: {
854
- readonly inputTokenCost: 2;
855
- readonly cachedInputTokenCost: 0.2;
856
- readonly outputTokenCost: 9;
896
+ readonly inputTokenCost: 0.4;
897
+ readonly cachedInputTokenCost: 0.04;
898
+ readonly outputTokenCost: 1.8;
857
899
  readonly thresholdTokens: 200000;
858
900
  };
859
901
  readonly reasoning: {
@@ -938,10 +980,39 @@ export declare const textModels: readonly [{
938
980
  readonly temperatureSupported: true;
939
981
  readonly disabled: true;
940
982
  readonly provider: "google";
983
+ }, {
984
+ readonly type: "text";
985
+ readonly modelName: "gemini-3.6-flash";
986
+ readonly description: "Latest Gemini 3.6 Flash model (GA July 21, 2026). Supersedes Gemini 3.5 Flash with better token efficiency and agentic planning at a lower output price ($7.50 vs $9.00/1M). 1M context window, 64K output. Context caching: $0.15/1M read.";
987
+ readonly maxInputTokens: 1048576;
988
+ readonly maxOutputTokens: 65536;
989
+ readonly inputTokenCost: 1.5;
990
+ readonly cachedInputTokenCost: 0.15;
991
+ readonly outputTokenCost: 7.5;
992
+ readonly inputAudioTokenCost: 1.5;
993
+ readonly reasoning: {
994
+ readonly levels: readonly ["minimal", "low", "medium", "high"];
995
+ readonly defaultLevel: "high";
996
+ readonly canDisable: false;
997
+ readonly outputsThinking: true;
998
+ readonly outputsSignatures: true;
999
+ };
1000
+ readonly modalities: {
1001
+ readonly input: readonly ["text", "image", "video", "audio", "pdf"];
1002
+ readonly output: readonly ["text"];
1003
+ };
1004
+ readonly knowledge: "2026-03";
1005
+ readonly releaseDate: "2026-07-21";
1006
+ readonly lastUpdated: "2026-07-21";
1007
+ readonly family: "gemini-flash";
1008
+ readonly openWeights: false;
1009
+ readonly structuredOutput: true;
1010
+ readonly temperatureSupported: true;
1011
+ readonly provider: "google";
941
1012
  }, {
942
1013
  readonly type: "text";
943
1014
  readonly modelName: "gemini-3.5-flash";
944
- readonly description: "Latest Gemini 3.5 Flash model (GA May 2026). Outperforms Gemini 3.1 Pro on coding and agentic suites at 4x the speed. 1M context window, 64K output. Context caching: $0.15/1M read.";
1015
+ readonly description: "Gemini 3.5 Flash (GA May 2026). Superseded by gemini-3.6-flash. Outperforms Gemini 3.1 Pro on coding and agentic suites at 4x the speed. 1M context window, 64K output. Context caching: $0.15/1M read.";
945
1016
  readonly maxInputTokens: 1048576;
946
1017
  readonly maxOutputTokens: 65536;
947
1018
  readonly inputTokenCost: 1.5;
@@ -997,10 +1068,39 @@ export declare const textModels: readonly [{
997
1068
  readonly structuredOutput: true;
998
1069
  readonly temperatureSupported: true;
999
1070
  readonly provider: "google";
1071
+ }, {
1072
+ readonly type: "text";
1073
+ readonly modelName: "gemini-3.5-flash-lite";
1074
+ readonly description: "Most cost-effective Gemini model (GA July 21, 2026). Supersedes gemini-3.1-flash-lite. Built for high-volume, low-reasoning work (search, document processing, translation). Thinking support, 1M context window, 64K output.";
1075
+ readonly maxInputTokens: 1048576;
1076
+ readonly maxOutputTokens: 65536;
1077
+ readonly inputTokenCost: 0.3;
1078
+ readonly cachedInputTokenCost: 0.03;
1079
+ readonly outputTokenCost: 2.5;
1080
+ readonly inputAudioTokenCost: 0.5;
1081
+ readonly reasoning: {
1082
+ readonly levels: readonly ["minimal", "low", "medium", "high"];
1083
+ readonly defaultLevel: "minimal";
1084
+ readonly canDisable: false;
1085
+ readonly outputsThinking: true;
1086
+ readonly outputsSignatures: true;
1087
+ };
1088
+ readonly modalities: {
1089
+ readonly input: readonly ["text", "image", "video", "audio", "pdf"];
1090
+ readonly output: readonly ["text"];
1091
+ };
1092
+ readonly knowledge: "2025-01";
1093
+ readonly releaseDate: "2026-07-21";
1094
+ readonly lastUpdated: "2026-07-21";
1095
+ readonly family: "gemini-flash-lite";
1096
+ readonly openWeights: false;
1097
+ readonly structuredOutput: true;
1098
+ readonly temperatureSupported: true;
1099
+ readonly provider: "google";
1000
1100
  }, {
1001
1101
  readonly type: "text";
1002
1102
  readonly modelName: "gemini-3.1-flash-lite";
1003
- readonly description: "Most cost-effective Gemini 3.1 model (GA). Thinking support, 1M context window, 64K output. 2.5x faster TTFA and 45% faster output than 2.5 Flash.";
1103
+ readonly description: "Cost-effective Gemini 3.1 model (GA). Superseded by gemini-3.5-flash-lite. Thinking support, 1M context window, 64K output. 2.5x faster TTFA and 45% faster output than 2.5 Flash.";
1004
1104
  readonly maxInputTokens: 1048576;
1005
1105
  readonly maxOutputTokens: 65536;
1006
1106
  readonly inputTokenCost: 0.25;
@@ -1558,6 +1658,21 @@ export declare const textModels: readonly [{
1558
1658
  readonly structuredOutput: true;
1559
1659
  readonly temperatureSupported: false;
1560
1660
  readonly provider: "openai-responses";
1661
+ }, {
1662
+ readonly type: "text";
1663
+ readonly modelName: "gpt-audio-1.5";
1664
+ readonly description: "OpenAI GA audio chat model (Chat Completions). Text+audio in, text+audio out.";
1665
+ readonly provider: "openai";
1666
+ readonly modalities: {
1667
+ readonly input: readonly ["text", "audio"];
1668
+ readonly output: readonly ["text", "audio"];
1669
+ };
1670
+ readonly inputTokenCost: 2.5;
1671
+ readonly outputTokenCost: 10;
1672
+ readonly inputAudioTokenCost: 32;
1673
+ readonly outputAudioTokenCost: 64;
1674
+ readonly maxInputTokens: 128000;
1675
+ readonly maxOutputTokens: 16384;
1561
1676
  }];
1562
1677
  export declare const imageModels: readonly [{
1563
1678
  readonly type: "image";
@@ -1610,6 +1725,7 @@ export declare const embeddingsModels: EmbeddingsModel[];
1610
1725
  export type TextModelName = (typeof textModels)[number]["modelName"];
1611
1726
  export type ImageModelName = (typeof imageModels)[number]["modelName"];
1612
1727
  export type SpeechToTextModelName = (typeof speechToTextModels)[number]["modelName"];
1728
+ export type TextToSpeechModelName = (typeof textToSpeechModels)[number]["modelName"];
1613
1729
  export type EmbeddingsModelName = (typeof embeddingsModels)[number]["modelName"];
1614
1730
  export type ModelName = string;
1615
1731
  export declare const hostedTools: HostedTool[];
@@ -1630,12 +1746,19 @@ export declare function getRegisteredModelData(): ModelDataBlob | null;
1630
1746
  */
1631
1747
  export declare function getAllModels(requestData?: ModelDataBlob): ModelType[];
1632
1748
  export declare function getModel(modelName: ModelName, requestData?: ModelDataBlob): ModelType | undefined;
1749
+ /**
1750
+ * Like `getModel`, but also matches on `provider`. Use this whenever a
1751
+ * modelName may collide across providers (the merge key everywhere else in
1752
+ * this module is `provider:modelName`) — plain `getModel` returns whichever
1753
+ * matching entry comes first and can silently pick the wrong provider.
1754
+ */
1755
+ export declare function getModelForProvider(provider: string, modelName: ModelName, requestData?: ModelDataBlob): ModelType | undefined;
1633
1756
  /**
1634
1757
  * Whether a model is known to accept the given input modality ("image", "pdf", …).
1635
1758
  * Returns undefined when the model is unknown or carries no `modalities` data —
1636
1759
  * callers should treat undefined as "don't gate".
1637
1760
  */
1638
- export declare function modelSupportsInputModality(modelName: ModelName, modality: string, requestData?: ModelDataBlob): boolean | undefined;
1761
+ export declare function modelSupportsInputModality(modelName: ModelName, modality: string, requestData?: ModelDataBlob, provider?: string): boolean | undefined;
1639
1762
  export declare function getHostedTools(opts?: {
1640
1763
  provider?: string;
1641
1764
  model?: string;
@@ -1647,5 +1770,6 @@ export declare function hostedToolPricingFor(tool: HostedTool, model?: string):
1647
1770
  export declare function isImageModel(model: ModelType): model is ImageModel;
1648
1771
  export declare function isTextModel(model: ModelType): model is TextModel;
1649
1772
  export declare function isSpeechToTextModel(model: ModelType): model is SpeechToTextModel;
1773
+ export declare function isTextToSpeechModel(model: ModelType): model is TextToSpeechModel;
1650
1774
  export declare function isEmbeddingsModel(model: ModelType): model is EmbeddingsModel;
1651
1775
  export declare const ModelNameSchema: z.ZodString;
package/dist/models.js CHANGED
@@ -17,20 +17,35 @@ export const ProviderSchema = z.enum(providers);
17
17
  export const speechToTextModels = [
18
18
  {
19
19
  type: "speech-to-text",
20
- modelName: "whisper-web",
20
+ modelName: "whisper-1",
21
21
  perMinuteCost: 0.006,
22
22
  provider: "openai",
23
+ supportedMimeTypes: [
24
+ "audio/flac", "audio/mpeg", "audio/mp4", "audio/m4a", "audio/ogg",
25
+ "audio/wav", "audio/webm",
26
+ ],
27
+ maxBytes: 25 * 1024 * 1024,
28
+ },
29
+ ];
30
+ export const textToSpeechModels = [
31
+ {
32
+ type: "text-to-speech",
33
+ modelName: "tts-1",
34
+ perCharacterCost: 0.000015,
35
+ provider: "openai",
36
+ maxInputChars: 4096,
37
+ speedRange: { min: 0.25, max: 4 },
38
+ formats: ["mp3", "opus", "aac", "flac", "wav", "pcm"],
39
+ },
40
+ {
41
+ type: "text-to-speech",
42
+ modelName: "tts-1-hd",
43
+ perCharacterCost: 0.00003,
44
+ provider: "openai",
45
+ maxInputChars: 4096,
46
+ speedRange: { min: 0.25, max: 4 },
47
+ formats: ["mp3", "opus", "aac", "flac", "wav", "pcm"],
23
48
  },
24
- // not a speech to text model?
25
- /* {
26
- type: "speech-to-text",
27
- modelName: "gpt-4o-audio-preview",
28
- description:
29
- "This is a preview release of the GPT-4o Audio models. These models accept audio inputs and outputs, and can be used in the Chat Completions REST API. Learn more. The knowledge cutoff for GPT-4o Audio models is October, 2023.",
30
- inputTokenCost: 2.5,
31
- outputTokenCost: 10,
32
- provider: "openai",
33
- }, */
34
49
  ];
35
50
  export const textModels = [
36
51
  {
@@ -771,16 +786,16 @@ export const textModels = [
771
786
  {
772
787
  type: "text",
773
788
  modelName: "gpt-5.6-terra",
774
- description: "GPT-5.6 Terra balances capability and cost — competitive with GPT-5.5 at roughly half the price. 1M context window. Standard pricing for ≤272K tokens, 2x input/1.5x output for >272K. Knowledge cutoff: February 2026.",
789
+ description: "GPT-5.6 Terra balances capability and cost. 1M context window. Standard pricing for ≤272K tokens, 2x input/1.5x output for >272K. Reflects the July 30, 2026 price cut. Knowledge cutoff: February 2026.",
775
790
  maxInputTokens: 1050000,
776
791
  maxOutputTokens: 128000,
777
- inputTokenCost: 2.5,
778
- cachedInputTokenCost: 0.25,
779
- outputTokenCost: 15,
792
+ inputTokenCost: 2,
793
+ cachedInputTokenCost: 0.2,
794
+ outputTokenCost: 12,
780
795
  longContext: {
781
- inputTokenCost: 5,
782
- cachedInputTokenCost: 0.5,
783
- outputTokenCost: 22.5,
796
+ inputTokenCost: 4,
797
+ cachedInputTokenCost: 0.4,
798
+ outputTokenCost: 18,
784
799
  thresholdTokens: 200000,
785
800
  },
786
801
  reasoning: {
@@ -806,16 +821,16 @@ export const textModels = [
806
821
  {
807
822
  type: "text",
808
823
  modelName: "gpt-5.6-luna",
809
- description: "GPT-5.6 Luna is the fast, most affordable member of the GPT-5.6 family. 1M context window. Standard pricing for ≤272K tokens, 2x input/1.5x output for >272K. Knowledge cutoff: February 2026.",
824
+ description: "GPT-5.6 Luna is the fast, most affordable member of the GPT-5.6 family. 1M context window. Standard pricing for ≤272K tokens, 2x input/1.5x output for >272K. Reflects the July 30, 2026 price cut (input/output down ~80%). Knowledge cutoff: February 2026.",
810
825
  maxInputTokens: 1050000,
811
826
  maxOutputTokens: 128000,
812
- inputTokenCost: 1,
813
- cachedInputTokenCost: 0.1,
814
- outputTokenCost: 6,
827
+ inputTokenCost: 0.2,
828
+ cachedInputTokenCost: 0.02,
829
+ outputTokenCost: 1.2,
815
830
  longContext: {
816
- inputTokenCost: 2,
817
- cachedInputTokenCost: 0.2,
818
- outputTokenCost: 9,
831
+ inputTokenCost: 0.4,
832
+ cachedInputTokenCost: 0.04,
833
+ outputTokenCost: 1.8,
819
834
  thresholdTokens: 200000,
820
835
  },
821
836
  reasoning: {
@@ -903,10 +918,40 @@ export const textModels = [
903
918
  disabled: true,
904
919
  provider: "google",
905
920
  },
921
+ {
922
+ type: "text",
923
+ modelName: "gemini-3.6-flash",
924
+ description: "Latest Gemini 3.6 Flash model (GA July 21, 2026). Supersedes Gemini 3.5 Flash with better token efficiency and agentic planning at a lower output price ($7.50 vs $9.00/1M). 1M context window, 64K output. Context caching: $0.15/1M read.",
925
+ maxInputTokens: 1048576,
926
+ maxOutputTokens: 65536,
927
+ inputTokenCost: 1.5,
928
+ cachedInputTokenCost: 0.15,
929
+ outputTokenCost: 7.5,
930
+ inputAudioTokenCost: 1.5,
931
+ reasoning: {
932
+ levels: ["minimal", "low", "medium", "high"],
933
+ defaultLevel: "high",
934
+ canDisable: false,
935
+ outputsThinking: true,
936
+ outputsSignatures: true,
937
+ },
938
+ modalities: {
939
+ input: ["text", "image", "video", "audio", "pdf"],
940
+ output: ["text"],
941
+ },
942
+ knowledge: "2026-03",
943
+ releaseDate: "2026-07-21",
944
+ lastUpdated: "2026-07-21",
945
+ family: "gemini-flash",
946
+ openWeights: false,
947
+ structuredOutput: true,
948
+ temperatureSupported: true,
949
+ provider: "google",
950
+ },
906
951
  {
907
952
  type: "text",
908
953
  modelName: "gemini-3.5-flash",
909
- description: "Latest Gemini 3.5 Flash model (GA May 2026). Outperforms Gemini 3.1 Pro on coding and agentic suites at 4x the speed. 1M context window, 64K output. Context caching: $0.15/1M read.",
954
+ description: "Gemini 3.5 Flash (GA May 2026). Superseded by gemini-3.6-flash. Outperforms Gemini 3.1 Pro on coding and agentic suites at 4x the speed. 1M context window, 64K output. Context caching: $0.15/1M read.",
910
955
  maxInputTokens: 1048576,
911
956
  maxOutputTokens: 65536,
912
957
  inputTokenCost: 1.5,
@@ -964,10 +1009,40 @@ export const textModels = [
964
1009
  temperatureSupported: true,
965
1010
  provider: "google",
966
1011
  },
1012
+ {
1013
+ type: "text",
1014
+ modelName: "gemini-3.5-flash-lite",
1015
+ description: "Most cost-effective Gemini model (GA July 21, 2026). Supersedes gemini-3.1-flash-lite. Built for high-volume, low-reasoning work (search, document processing, translation). Thinking support, 1M context window, 64K output.",
1016
+ maxInputTokens: 1048576,
1017
+ maxOutputTokens: 65536,
1018
+ inputTokenCost: 0.3,
1019
+ cachedInputTokenCost: 0.03,
1020
+ outputTokenCost: 2.5,
1021
+ inputAudioTokenCost: 0.5,
1022
+ reasoning: {
1023
+ levels: ["minimal", "low", "medium", "high"],
1024
+ defaultLevel: "minimal",
1025
+ canDisable: false,
1026
+ outputsThinking: true,
1027
+ outputsSignatures: true,
1028
+ },
1029
+ modalities: {
1030
+ input: ["text", "image", "video", "audio", "pdf"],
1031
+ output: ["text"],
1032
+ },
1033
+ knowledge: "2025-01",
1034
+ releaseDate: "2026-07-21",
1035
+ lastUpdated: "2026-07-21",
1036
+ family: "gemini-flash-lite",
1037
+ openWeights: false,
1038
+ structuredOutput: true,
1039
+ temperatureSupported: true,
1040
+ provider: "google",
1041
+ },
967
1042
  {
968
1043
  type: "text",
969
1044
  modelName: "gemini-3.1-flash-lite",
970
- description: "Most cost-effective Gemini 3.1 model (GA). Thinking support, 1M context window, 64K output. 2.5x faster TTFA and 45% faster output than 2.5 Flash.",
1045
+ description: "Cost-effective Gemini 3.1 model (GA). Superseded by gemini-3.5-flash-lite. Thinking support, 1M context window, 64K output. 2.5x faster TTFA and 45% faster output than 2.5 Flash.",
971
1046
  maxInputTokens: 1048576,
972
1047
  maxOutputTokens: 65536,
973
1048
  inputTokenCost: 0.25,
@@ -1549,6 +1624,19 @@ export const textModels = [
1549
1624
  temperatureSupported: false,
1550
1625
  provider: "openai-responses",
1551
1626
  },
1627
+ {
1628
+ type: "text",
1629
+ modelName: "gpt-audio-1.5",
1630
+ description: "OpenAI GA audio chat model (Chat Completions). Text+audio in, text+audio out.",
1631
+ provider: "openai",
1632
+ modalities: { input: ["text", "audio"], output: ["text", "audio"] },
1633
+ inputTokenCost: 2.5,
1634
+ outputTokenCost: 10,
1635
+ inputAudioTokenCost: 32,
1636
+ outputAudioTokenCost: 64,
1637
+ maxInputTokens: 128000,
1638
+ maxOutputTokens: 16384,
1639
+ },
1552
1640
  ];
1553
1641
  export const imageModels = [
1554
1642
  {
@@ -1730,7 +1818,7 @@ export const hostedTools = [
1730
1818
  category: "maps_grounding",
1731
1819
  description: "Grounding with Google Maps (Gemini 3 only).",
1732
1820
  providerToolId: "google_maps",
1733
- models: ["gemini-3-pro-preview", "gemini-3.1-pro-preview", "gemini-3-flash-preview", "gemini-3.5-flash", "gemini-3.1-flash-lite"],
1821
+ models: ["gemini-3-pro-preview", "gemini-3.1-pro-preview", "gemini-3-flash-preview", "gemini-3.5-flash", "gemini-3.6-flash", "gemini-3.1-flash-lite", "gemini-3.5-flash-lite"],
1734
1822
  pricing: { unit: "per_call", note: "Gemini 3 family only; see Google pricing." },
1735
1823
  },
1736
1824
  {
@@ -1765,6 +1853,7 @@ function baselineModels() {
1765
1853
  ...textModels,
1766
1854
  ...imageModels,
1767
1855
  ...speechToTextModels,
1856
+ ...textToSpeechModels,
1768
1857
  ...registeredTextModels,
1769
1858
  ...embeddingsModels,
1770
1859
  ];
@@ -1790,13 +1879,28 @@ export function getAllModels(requestData) {
1790
1879
  export function getModel(modelName, requestData) {
1791
1880
  return getAllModels(requestData).find((model) => model.modelName === modelName);
1792
1881
  }
1882
+ /**
1883
+ * Like `getModel`, but also matches on `provider`. Use this whenever a
1884
+ * modelName may collide across providers (the merge key everywhere else in
1885
+ * this module is `provider:modelName`) — plain `getModel` returns whichever
1886
+ * matching entry comes first and can silently pick the wrong provider.
1887
+ */
1888
+ export function getModelForProvider(provider, modelName, requestData) {
1889
+ return getAllModels(requestData).find((model) => model.modelName === modelName && model.provider === provider);
1890
+ }
1793
1891
  /**
1794
1892
  * Whether a model is known to accept the given input modality ("image", "pdf", …).
1795
1893
  * Returns undefined when the model is unknown or carries no `modalities` data —
1796
1894
  * callers should treat undefined as "don't gate".
1797
1895
  */
1798
- export function modelSupportsInputModality(modelName, modality, requestData) {
1799
- const model = getModel(modelName, requestData);
1896
+ export function modelSupportsInputModality(modelName, modality, requestData, provider) {
1897
+ let model;
1898
+ if (provider !== undefined) {
1899
+ model = getModelForProvider(provider, modelName, requestData);
1900
+ }
1901
+ else {
1902
+ model = getModel(modelName, requestData);
1903
+ }
1800
1904
  if (!model || model.type !== "text") {
1801
1905
  return undefined;
1802
1906
  }
@@ -1868,6 +1972,9 @@ export function isTextModel(model) {
1868
1972
  export function isSpeechToTextModel(model) {
1869
1973
  return model.type === "speech-to-text";
1870
1974
  }
1975
+ export function isTextToSpeechModel(model) {
1976
+ return model.type === "text-to-speech";
1977
+ }
1871
1978
  export function isEmbeddingsModel(model) {
1872
1979
  return model.type === "embeddings";
1873
1980
  }
@@ -0,0 +1,31 @@
1
+ import type { ModelDataBlob } from "../modelData.js";
2
+ import { Result } from "../types/result.js";
3
+ import type { SpeechResult } from "../speech.js";
4
+ export type SpeechClientConfig = {
5
+ model: string;
6
+ /** Resolved provider name. */
7
+ provider: string;
8
+ /** Resolved API key; empty string when none was found. */
9
+ apiKey: string;
10
+ voice: string;
11
+ modelData?: ModelDataBlob;
12
+ /** Output format; provider-specific vocabulary (OpenAI: mp3/opus/aac/flac/wav/pcm). */
13
+ format?: string;
14
+ speed?: number;
15
+ metadata?: Record<string, unknown>;
16
+ };
17
+ /**
18
+ * Shared TTS behavior, mirroring BaseClient for text generation: the public
19
+ * speak() template method owns model-data-driven validation (char cap, speed
20
+ * range, format list), cost, and the single redacting/logging exception
21
+ * boundary. Subclasses implement only _speak(): SDK call + response mapping.
22
+ * A model with no registry entry skips validation — the provider is then the
23
+ * authority, matching how cost is silently omitted for unknown models.
24
+ */
25
+ export declare abstract class BaseSpeechClient {
26
+ protected config: SpeechClientConfig;
27
+ constructor(config: SpeechClientConfig);
28
+ speak(text: string): Promise<Result<SpeechResult>>;
29
+ /** Provider hook: SDK call + response mapping only; validation and cost live in the base. */
30
+ protected abstract _speak(text: string): Promise<Result<SpeechResult>>;
31
+ }
@@ -0,0 +1,98 @@
1
+ import { getModelForProvider, isTextToSpeechModel, } from "../models.js";
2
+ import { calculateSpeechCost } from "../model.js";
3
+ import { failure } from "../types/result.js";
4
+ import { redactSecret } from "../util/redact.js";
5
+ import { getLogger } from "../util/logger.js";
6
+ /** Validate the declarative TTS constraint block once before consuming it. */
7
+ function speechConstraintError(model) {
8
+ const maxInputChars = model.maxInputChars;
9
+ if (maxInputChars !== undefined &&
10
+ (typeof maxInputChars !== "number" ||
11
+ !Number.isInteger(maxInputChars) ||
12
+ maxInputChars <= 0)) {
13
+ return `Model "${model.modelName}" has an invalid maxInputChars value.`;
14
+ }
15
+ const speedRange = model.speedRange;
16
+ if (speedRange !== undefined) {
17
+ if (typeof speedRange !== "object" || speedRange === null) {
18
+ return `Model "${model.modelName}" has an invalid speedRange.`;
19
+ }
20
+ const min = speedRange.min;
21
+ const max = speedRange.max;
22
+ if (typeof min !== "number" ||
23
+ typeof max !== "number" ||
24
+ !Number.isFinite(min) ||
25
+ !Number.isFinite(max) ||
26
+ min > max) {
27
+ return `Model "${model.modelName}" has an invalid speedRange.`;
28
+ }
29
+ }
30
+ const formats = model.formats;
31
+ if (formats !== undefined &&
32
+ (!Array.isArray(formats) ||
33
+ !formats.every((format) => typeof format === "string"))) {
34
+ return `Model "${model.modelName}" has invalid formats.`;
35
+ }
36
+ return null;
37
+ }
38
+ /**
39
+ * Shared TTS behavior, mirroring BaseClient for text generation: the public
40
+ * speak() template method owns model-data-driven validation (char cap, speed
41
+ * range, format list), cost, and the single redacting/logging exception
42
+ * boundary. Subclasses implement only _speak(): SDK call + response mapping.
43
+ * A model with no registry entry skips validation — the provider is then the
44
+ * authority, matching how cost is silently omitted for unknown models.
45
+ */
46
+ export class BaseSpeechClient {
47
+ config;
48
+ constructor(config) {
49
+ this.config = config;
50
+ }
51
+ async speak(text) {
52
+ try {
53
+ const model = getModelForProvider(this.config.provider, this.config.model, this.config.modelData);
54
+ if (model !== undefined && !isTextToSpeechModel(model)) {
55
+ return failure(`Model "${this.config.model}" is not a text-to-speech model.`);
56
+ }
57
+ if (model !== undefined) {
58
+ const constraintError = speechConstraintError(model);
59
+ if (constraintError !== null) {
60
+ return failure(constraintError);
61
+ }
62
+ if (model.maxInputChars !== undefined && [...text].length > model.maxInputChars) {
63
+ return failure(`Input exceeds the ${model.maxInputChars}-character limit for model "${this.config.model}".`);
64
+ }
65
+ if (this.config.speed !== undefined && model.speedRange !== undefined) {
66
+ const { min, max } = model.speedRange;
67
+ if (!Number.isFinite(this.config.speed) || this.config.speed < min || this.config.speed > max) {
68
+ return failure(`speed must be a finite number in [${min}, ${max}].`);
69
+ }
70
+ }
71
+ if (this.config.format !== undefined &&
72
+ model.formats !== undefined &&
73
+ !model.formats.includes(this.config.format)) {
74
+ return failure(`Format "${this.config.format}" is not supported by model "${this.config.model}". ` +
75
+ `Supported: ${model.formats.join(", ")}.`);
76
+ }
77
+ }
78
+ const result = await this._speak(text);
79
+ if (!result.success) {
80
+ return result;
81
+ }
82
+ const cost = calculateSpeechCost(model, [...text].length);
83
+ if (cost !== undefined) {
84
+ result.value.cost = cost;
85
+ }
86
+ return result;
87
+ }
88
+ catch (err) {
89
+ let msg = "speak() failed";
90
+ if (err instanceof Error) {
91
+ msg = err.message;
92
+ }
93
+ const redacted = redactSecret(msg, this.config.apiKey);
94
+ getLogger().error("speak() provider failed:", redacted);
95
+ return failure(redacted);
96
+ }
97
+ }
98
+ }
@@ -0,0 +1,6 @@
1
+ import { Result } from "../types/result.js";
2
+ import { BaseSpeechClient } from "./baseSpeechClient.js";
3
+ import type { SpeechResult } from "../speech.js";
4
+ export declare class OpenAISpeechClient extends BaseSpeechClient {
5
+ protected _speak(text: string): Promise<Result<SpeechResult>>;
6
+ }