smoltalk 0.9.0 → 0.10.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +141 -7
- package/dist/classes/message/AssistantMessage.d.ts +2 -0
- package/dist/classes/message/UserMessage.d.ts +21 -0
- package/dist/classes/message/UserMessage.js +3 -0
- package/dist/classes/message/contentParts.d.ts +71 -2
- package/dist/classes/message/contentParts.js +6 -0
- package/dist/classes/message/index.d.ts +5 -2
- package/dist/classes/message/index.js +7 -0
- package/dist/classes/message/renderers/AnthropicRenderer.d.ts +2 -1
- package/dist/classes/message/renderers/AnthropicRenderer.js +3 -0
- package/dist/classes/message/renderers/GoogleRenderer.d.ts +2 -1
- package/dist/classes/message/renderers/GoogleRenderer.js +3 -0
- package/dist/classes/message/renderers/JSONRenderer.d.ts +2 -1
- package/dist/classes/message/renderers/JSONRenderer.js +4 -0
- package/dist/classes/message/renderers/OpenAIChatRenderer.d.ts +8 -1
- package/dist/classes/message/renderers/OpenAIChatRenderer.js +18 -0
- package/dist/classes/message/renderers/OpenAIResponsesRenderer.d.ts +2 -1
- package/dist/classes/message/renderers/OpenAIResponsesRenderer.js +3 -0
- package/dist/classes/message/renderers/PartRenderer.d.ts +3 -2
- package/dist/classes/message/renderers/PartRenderer.js +3 -0
- package/dist/client.js +1 -0
- package/dist/clients/anthropic.js +1 -1
- package/dist/clients/baseClient.d.ts +13 -1
- package/dist/clients/baseClient.js +36 -7
- package/dist/clients/google.js +1 -1
- package/dist/clients/ollama.js +1 -1
- package/dist/clients/openai.d.ts +2 -1
- package/dist/clients/openai.js +15 -3
- package/dist/clients/openaiCompat.d.ts +2 -0
- package/dist/clients/openaiCompat.js +5 -0
- package/dist/clients/openaiResponses.js +1 -1
- package/dist/clients/resolveAttachments.d.ts +8 -4
- package/dist/clients/resolveAttachments.js +101 -50
- package/dist/embed.d.ts +4 -0
- package/dist/files.d.ts +1 -1
- package/dist/files.js +1 -1
- package/dist/image/google.js +2 -2
- package/dist/image/openai.js +3 -3
- package/dist/image.d.ts +1 -1
- package/dist/index.d.ts +10 -2
- package/dist/index.js +7 -1
- package/dist/model.d.ts +15 -4
- package/dist/model.js +48 -7
- package/dist/models.d.ts +143 -19
- package/dist/models.js +137 -30
- package/dist/speech/baseSpeechClient.d.ts +31 -0
- package/dist/speech/baseSpeechClient.js +98 -0
- package/dist/speech/openai.d.ts +6 -0
- package/dist/speech/openai.js +39 -0
- package/dist/speech.d.ts +40 -0
- package/dist/speech.js +57 -0
- package/dist/transcription/baseTranscriptionClient.d.ts +31 -0
- package/dist/transcription/baseTranscriptionClient.js +107 -0
- package/dist/transcription/openai.d.ts +6 -0
- package/dist/transcription/openai.js +59 -0
- package/dist/transcription.d.ts +51 -0
- package/dist/transcription.js +58 -0
- package/dist/types/tokenUsage.d.ts +4 -0
- package/dist/types/tokenUsage.js +4 -0
- package/dist/types.d.ts +3 -0
- package/dist/util/attachments.d.ts +1 -1
- package/dist/util/audioMime.d.ts +9 -0
- package/dist/util/audioMime.js +33 -0
- package/dist/util/{imageRef.d.ts → blobRef.d.ts} +9 -9
- package/dist/util/{imageRef.js → blobRef.js} +6 -13
- package/dist/util/mime.d.ts +21 -0
- package/dist/util/mime.js +52 -0
- package/dist/util/modalities.d.ts +6 -2
- package/dist/util/modalities.js +13 -15
- package/dist/util/provider.d.ts +2 -0
- package/dist/util/provider.js +1 -1
- package/package.json +1 -1
package/dist/models.d.ts
CHANGED
|
@@ -34,6 +34,23 @@ export type BaseModel = {
|
|
|
34
34
|
export type SpeechToTextModel = BaseModel & {
|
|
35
35
|
type: "speech-to-text";
|
|
36
36
|
perMinuteCost?: number;
|
|
37
|
+
/** Canonical MIME types accepted after alias normalization through AUDIO_FORMATS. */
|
|
38
|
+
supportedMimeTypes?: readonly string[];
|
|
39
|
+
/** Provider upload cap in bytes. */
|
|
40
|
+
maxBytes?: number;
|
|
41
|
+
};
|
|
42
|
+
export type TextToSpeechModel = BaseModel & {
|
|
43
|
+
type: "text-to-speech";
|
|
44
|
+
perCharacterCost?: number;
|
|
45
|
+
/** Input cap in Unicode code points. */
|
|
46
|
+
maxInputChars?: number;
|
|
47
|
+
/** Accepted values for the speed option. */
|
|
48
|
+
speedRange?: {
|
|
49
|
+
min: number;
|
|
50
|
+
max: number;
|
|
51
|
+
};
|
|
52
|
+
/** Output formats the provider can render for this model. */
|
|
53
|
+
formats?: readonly string[];
|
|
37
54
|
};
|
|
38
55
|
export type ImageModel = BaseModel & {
|
|
39
56
|
type: "image";
|
|
@@ -91,12 +108,37 @@ export type EmbeddingsModel = {
|
|
|
91
108
|
provider: string;
|
|
92
109
|
tokenCost?: number;
|
|
93
110
|
};
|
|
94
|
-
export type ModelType = SpeechToTextModel | TextModel | EmbeddingsModel | ImageModel;
|
|
111
|
+
export type ModelType = SpeechToTextModel | TextToSpeechModel | TextModel | EmbeddingsModel | ImageModel;
|
|
95
112
|
export declare const speechToTextModels: readonly [{
|
|
96
113
|
readonly type: "speech-to-text";
|
|
97
|
-
readonly modelName: "whisper-
|
|
114
|
+
readonly modelName: "whisper-1";
|
|
98
115
|
readonly perMinuteCost: 0.006;
|
|
99
116
|
readonly provider: "openai";
|
|
117
|
+
readonly supportedMimeTypes: readonly ["audio/flac", "audio/mpeg", "audio/mp4", "audio/m4a", "audio/ogg", "audio/wav", "audio/webm"];
|
|
118
|
+
readonly maxBytes: number;
|
|
119
|
+
}];
|
|
120
|
+
export declare const textToSpeechModels: readonly [{
|
|
121
|
+
readonly type: "text-to-speech";
|
|
122
|
+
readonly modelName: "tts-1";
|
|
123
|
+
readonly perCharacterCost: 0.000015;
|
|
124
|
+
readonly provider: "openai";
|
|
125
|
+
readonly maxInputChars: 4096;
|
|
126
|
+
readonly speedRange: {
|
|
127
|
+
readonly min: 0.25;
|
|
128
|
+
readonly max: 4;
|
|
129
|
+
};
|
|
130
|
+
readonly formats: readonly ["mp3", "opus", "aac", "flac", "wav", "pcm"];
|
|
131
|
+
}, {
|
|
132
|
+
readonly type: "text-to-speech";
|
|
133
|
+
readonly modelName: "tts-1-hd";
|
|
134
|
+
readonly perCharacterCost: 0.00003;
|
|
135
|
+
readonly provider: "openai";
|
|
136
|
+
readonly maxInputChars: 4096;
|
|
137
|
+
readonly speedRange: {
|
|
138
|
+
readonly min: 0.25;
|
|
139
|
+
readonly max: 4;
|
|
140
|
+
};
|
|
141
|
+
readonly formats: readonly ["mp3", "opus", "aac", "flac", "wav", "pcm"];
|
|
100
142
|
}];
|
|
101
143
|
export declare const textModels: readonly [{
|
|
102
144
|
readonly type: "text";
|
|
@@ -810,16 +852,16 @@ export declare const textModels: readonly [{
|
|
|
810
852
|
}, {
|
|
811
853
|
readonly type: "text";
|
|
812
854
|
readonly modelName: "gpt-5.6-terra";
|
|
813
|
-
readonly description: "GPT-5.6 Terra balances capability and cost
|
|
855
|
+
readonly description: "GPT-5.6 Terra balances capability and cost. 1M context window. Standard pricing for ≤272K tokens, 2x input/1.5x output for >272K. Reflects the July 30, 2026 price cut. Knowledge cutoff: February 2026.";
|
|
814
856
|
readonly maxInputTokens: 1050000;
|
|
815
857
|
readonly maxOutputTokens: 128000;
|
|
816
|
-
readonly inputTokenCost: 2
|
|
817
|
-
readonly cachedInputTokenCost: 0.
|
|
818
|
-
readonly outputTokenCost:
|
|
858
|
+
readonly inputTokenCost: 2;
|
|
859
|
+
readonly cachedInputTokenCost: 0.2;
|
|
860
|
+
readonly outputTokenCost: 12;
|
|
819
861
|
readonly longContext: {
|
|
820
|
-
readonly inputTokenCost:
|
|
821
|
-
readonly cachedInputTokenCost: 0.
|
|
822
|
-
readonly outputTokenCost:
|
|
862
|
+
readonly inputTokenCost: 4;
|
|
863
|
+
readonly cachedInputTokenCost: 0.4;
|
|
864
|
+
readonly outputTokenCost: 18;
|
|
823
865
|
readonly thresholdTokens: 200000;
|
|
824
866
|
};
|
|
825
867
|
readonly reasoning: {
|
|
@@ -844,16 +886,16 @@ export declare const textModels: readonly [{
|
|
|
844
886
|
}, {
|
|
845
887
|
readonly type: "text";
|
|
846
888
|
readonly modelName: "gpt-5.6-luna";
|
|
847
|
-
readonly description: "GPT-5.6 Luna is the fast, most affordable member of the GPT-5.6 family. 1M context window. Standard pricing for ≤272K tokens, 2x input/1.5x output for >272K. Knowledge cutoff: February 2026.";
|
|
889
|
+
readonly description: "GPT-5.6 Luna is the fast, most affordable member of the GPT-5.6 family. 1M context window. Standard pricing for ≤272K tokens, 2x input/1.5x output for >272K. Reflects the July 30, 2026 price cut (input/output down ~80%). Knowledge cutoff: February 2026.";
|
|
848
890
|
readonly maxInputTokens: 1050000;
|
|
849
891
|
readonly maxOutputTokens: 128000;
|
|
850
|
-
readonly inputTokenCost:
|
|
851
|
-
readonly cachedInputTokenCost: 0.
|
|
852
|
-
readonly outputTokenCost:
|
|
892
|
+
readonly inputTokenCost: 0.2;
|
|
893
|
+
readonly cachedInputTokenCost: 0.02;
|
|
894
|
+
readonly outputTokenCost: 1.2;
|
|
853
895
|
readonly longContext: {
|
|
854
|
-
readonly inputTokenCost:
|
|
855
|
-
readonly cachedInputTokenCost: 0.
|
|
856
|
-
readonly outputTokenCost:
|
|
896
|
+
readonly inputTokenCost: 0.4;
|
|
897
|
+
readonly cachedInputTokenCost: 0.04;
|
|
898
|
+
readonly outputTokenCost: 1.8;
|
|
857
899
|
readonly thresholdTokens: 200000;
|
|
858
900
|
};
|
|
859
901
|
readonly reasoning: {
|
|
@@ -938,10 +980,39 @@ export declare const textModels: readonly [{
|
|
|
938
980
|
readonly temperatureSupported: true;
|
|
939
981
|
readonly disabled: true;
|
|
940
982
|
readonly provider: "google";
|
|
983
|
+
}, {
|
|
984
|
+
readonly type: "text";
|
|
985
|
+
readonly modelName: "gemini-3.6-flash";
|
|
986
|
+
readonly description: "Latest Gemini 3.6 Flash model (GA July 21, 2026). Supersedes Gemini 3.5 Flash with better token efficiency and agentic planning at a lower output price ($7.50 vs $9.00/1M). 1M context window, 64K output. Context caching: $0.15/1M read.";
|
|
987
|
+
readonly maxInputTokens: 1048576;
|
|
988
|
+
readonly maxOutputTokens: 65536;
|
|
989
|
+
readonly inputTokenCost: 1.5;
|
|
990
|
+
readonly cachedInputTokenCost: 0.15;
|
|
991
|
+
readonly outputTokenCost: 7.5;
|
|
992
|
+
readonly inputAudioTokenCost: 1.5;
|
|
993
|
+
readonly reasoning: {
|
|
994
|
+
readonly levels: readonly ["minimal", "low", "medium", "high"];
|
|
995
|
+
readonly defaultLevel: "high";
|
|
996
|
+
readonly canDisable: false;
|
|
997
|
+
readonly outputsThinking: true;
|
|
998
|
+
readonly outputsSignatures: true;
|
|
999
|
+
};
|
|
1000
|
+
readonly modalities: {
|
|
1001
|
+
readonly input: readonly ["text", "image", "video", "audio", "pdf"];
|
|
1002
|
+
readonly output: readonly ["text"];
|
|
1003
|
+
};
|
|
1004
|
+
readonly knowledge: "2026-03";
|
|
1005
|
+
readonly releaseDate: "2026-07-21";
|
|
1006
|
+
readonly lastUpdated: "2026-07-21";
|
|
1007
|
+
readonly family: "gemini-flash";
|
|
1008
|
+
readonly openWeights: false;
|
|
1009
|
+
readonly structuredOutput: true;
|
|
1010
|
+
readonly temperatureSupported: true;
|
|
1011
|
+
readonly provider: "google";
|
|
941
1012
|
}, {
|
|
942
1013
|
readonly type: "text";
|
|
943
1014
|
readonly modelName: "gemini-3.5-flash";
|
|
944
|
-
readonly description: "
|
|
1015
|
+
readonly description: "Gemini 3.5 Flash (GA May 2026). Superseded by gemini-3.6-flash. Outperforms Gemini 3.1 Pro on coding and agentic suites at 4x the speed. 1M context window, 64K output. Context caching: $0.15/1M read.";
|
|
945
1016
|
readonly maxInputTokens: 1048576;
|
|
946
1017
|
readonly maxOutputTokens: 65536;
|
|
947
1018
|
readonly inputTokenCost: 1.5;
|
|
@@ -997,10 +1068,39 @@ export declare const textModels: readonly [{
|
|
|
997
1068
|
readonly structuredOutput: true;
|
|
998
1069
|
readonly temperatureSupported: true;
|
|
999
1070
|
readonly provider: "google";
|
|
1071
|
+
}, {
|
|
1072
|
+
readonly type: "text";
|
|
1073
|
+
readonly modelName: "gemini-3.5-flash-lite";
|
|
1074
|
+
readonly description: "Most cost-effective Gemini model (GA July 21, 2026). Supersedes gemini-3.1-flash-lite. Built for high-volume, low-reasoning work (search, document processing, translation). Thinking support, 1M context window, 64K output.";
|
|
1075
|
+
readonly maxInputTokens: 1048576;
|
|
1076
|
+
readonly maxOutputTokens: 65536;
|
|
1077
|
+
readonly inputTokenCost: 0.3;
|
|
1078
|
+
readonly cachedInputTokenCost: 0.03;
|
|
1079
|
+
readonly outputTokenCost: 2.5;
|
|
1080
|
+
readonly inputAudioTokenCost: 0.5;
|
|
1081
|
+
readonly reasoning: {
|
|
1082
|
+
readonly levels: readonly ["minimal", "low", "medium", "high"];
|
|
1083
|
+
readonly defaultLevel: "minimal";
|
|
1084
|
+
readonly canDisable: false;
|
|
1085
|
+
readonly outputsThinking: true;
|
|
1086
|
+
readonly outputsSignatures: true;
|
|
1087
|
+
};
|
|
1088
|
+
readonly modalities: {
|
|
1089
|
+
readonly input: readonly ["text", "image", "video", "audio", "pdf"];
|
|
1090
|
+
readonly output: readonly ["text"];
|
|
1091
|
+
};
|
|
1092
|
+
readonly knowledge: "2025-01";
|
|
1093
|
+
readonly releaseDate: "2026-07-21";
|
|
1094
|
+
readonly lastUpdated: "2026-07-21";
|
|
1095
|
+
readonly family: "gemini-flash-lite";
|
|
1096
|
+
readonly openWeights: false;
|
|
1097
|
+
readonly structuredOutput: true;
|
|
1098
|
+
readonly temperatureSupported: true;
|
|
1099
|
+
readonly provider: "google";
|
|
1000
1100
|
}, {
|
|
1001
1101
|
readonly type: "text";
|
|
1002
1102
|
readonly modelName: "gemini-3.1-flash-lite";
|
|
1003
|
-
readonly description: "
|
|
1103
|
+
readonly description: "Cost-effective Gemini 3.1 model (GA). Superseded by gemini-3.5-flash-lite. Thinking support, 1M context window, 64K output. 2.5x faster TTFA and 45% faster output than 2.5 Flash.";
|
|
1004
1104
|
readonly maxInputTokens: 1048576;
|
|
1005
1105
|
readonly maxOutputTokens: 65536;
|
|
1006
1106
|
readonly inputTokenCost: 0.25;
|
|
@@ -1558,6 +1658,21 @@ export declare const textModels: readonly [{
|
|
|
1558
1658
|
readonly structuredOutput: true;
|
|
1559
1659
|
readonly temperatureSupported: false;
|
|
1560
1660
|
readonly provider: "openai-responses";
|
|
1661
|
+
}, {
|
|
1662
|
+
readonly type: "text";
|
|
1663
|
+
readonly modelName: "gpt-audio-1.5";
|
|
1664
|
+
readonly description: "OpenAI GA audio chat model (Chat Completions). Text+audio in, text+audio out.";
|
|
1665
|
+
readonly provider: "openai";
|
|
1666
|
+
readonly modalities: {
|
|
1667
|
+
readonly input: readonly ["text", "audio"];
|
|
1668
|
+
readonly output: readonly ["text", "audio"];
|
|
1669
|
+
};
|
|
1670
|
+
readonly inputTokenCost: 2.5;
|
|
1671
|
+
readonly outputTokenCost: 10;
|
|
1672
|
+
readonly inputAudioTokenCost: 32;
|
|
1673
|
+
readonly outputAudioTokenCost: 64;
|
|
1674
|
+
readonly maxInputTokens: 128000;
|
|
1675
|
+
readonly maxOutputTokens: 16384;
|
|
1561
1676
|
}];
|
|
1562
1677
|
export declare const imageModels: readonly [{
|
|
1563
1678
|
readonly type: "image";
|
|
@@ -1610,6 +1725,7 @@ export declare const embeddingsModels: EmbeddingsModel[];
|
|
|
1610
1725
|
export type TextModelName = (typeof textModels)[number]["modelName"];
|
|
1611
1726
|
export type ImageModelName = (typeof imageModels)[number]["modelName"];
|
|
1612
1727
|
export type SpeechToTextModelName = (typeof speechToTextModels)[number]["modelName"];
|
|
1728
|
+
export type TextToSpeechModelName = (typeof textToSpeechModels)[number]["modelName"];
|
|
1613
1729
|
export type EmbeddingsModelName = (typeof embeddingsModels)[number]["modelName"];
|
|
1614
1730
|
export type ModelName = string;
|
|
1615
1731
|
export declare const hostedTools: HostedTool[];
|
|
@@ -1630,12 +1746,19 @@ export declare function getRegisteredModelData(): ModelDataBlob | null;
|
|
|
1630
1746
|
*/
|
|
1631
1747
|
export declare function getAllModels(requestData?: ModelDataBlob): ModelType[];
|
|
1632
1748
|
export declare function getModel(modelName: ModelName, requestData?: ModelDataBlob): ModelType | undefined;
|
|
1749
|
+
/**
|
|
1750
|
+
* Like `getModel`, but also matches on `provider`. Use this whenever a
|
|
1751
|
+
* modelName may collide across providers (the merge key everywhere else in
|
|
1752
|
+
* this module is `provider:modelName`) — plain `getModel` returns whichever
|
|
1753
|
+
* matching entry comes first and can silently pick the wrong provider.
|
|
1754
|
+
*/
|
|
1755
|
+
export declare function getModelForProvider(provider: string, modelName: ModelName, requestData?: ModelDataBlob): ModelType | undefined;
|
|
1633
1756
|
/**
|
|
1634
1757
|
* Whether a model is known to accept the given input modality ("image", "pdf", …).
|
|
1635
1758
|
* Returns undefined when the model is unknown or carries no `modalities` data —
|
|
1636
1759
|
* callers should treat undefined as "don't gate".
|
|
1637
1760
|
*/
|
|
1638
|
-
export declare function modelSupportsInputModality(modelName: ModelName, modality: string, requestData?: ModelDataBlob): boolean | undefined;
|
|
1761
|
+
export declare function modelSupportsInputModality(modelName: ModelName, modality: string, requestData?: ModelDataBlob, provider?: string): boolean | undefined;
|
|
1639
1762
|
export declare function getHostedTools(opts?: {
|
|
1640
1763
|
provider?: string;
|
|
1641
1764
|
model?: string;
|
|
@@ -1647,5 +1770,6 @@ export declare function hostedToolPricingFor(tool: HostedTool, model?: string):
|
|
|
1647
1770
|
export declare function isImageModel(model: ModelType): model is ImageModel;
|
|
1648
1771
|
export declare function isTextModel(model: ModelType): model is TextModel;
|
|
1649
1772
|
export declare function isSpeechToTextModel(model: ModelType): model is SpeechToTextModel;
|
|
1773
|
+
export declare function isTextToSpeechModel(model: ModelType): model is TextToSpeechModel;
|
|
1650
1774
|
export declare function isEmbeddingsModel(model: ModelType): model is EmbeddingsModel;
|
|
1651
1775
|
export declare const ModelNameSchema: z.ZodString;
|
package/dist/models.js
CHANGED
|
@@ -17,20 +17,35 @@ export const ProviderSchema = z.enum(providers);
|
|
|
17
17
|
export const speechToTextModels = [
|
|
18
18
|
{
|
|
19
19
|
type: "speech-to-text",
|
|
20
|
-
modelName: "whisper-
|
|
20
|
+
modelName: "whisper-1",
|
|
21
21
|
perMinuteCost: 0.006,
|
|
22
22
|
provider: "openai",
|
|
23
|
+
supportedMimeTypes: [
|
|
24
|
+
"audio/flac", "audio/mpeg", "audio/mp4", "audio/m4a", "audio/ogg",
|
|
25
|
+
"audio/wav", "audio/webm",
|
|
26
|
+
],
|
|
27
|
+
maxBytes: 25 * 1024 * 1024,
|
|
28
|
+
},
|
|
29
|
+
];
|
|
30
|
+
export const textToSpeechModels = [
|
|
31
|
+
{
|
|
32
|
+
type: "text-to-speech",
|
|
33
|
+
modelName: "tts-1",
|
|
34
|
+
perCharacterCost: 0.000015,
|
|
35
|
+
provider: "openai",
|
|
36
|
+
maxInputChars: 4096,
|
|
37
|
+
speedRange: { min: 0.25, max: 4 },
|
|
38
|
+
formats: ["mp3", "opus", "aac", "flac", "wav", "pcm"],
|
|
39
|
+
},
|
|
40
|
+
{
|
|
41
|
+
type: "text-to-speech",
|
|
42
|
+
modelName: "tts-1-hd",
|
|
43
|
+
perCharacterCost: 0.00003,
|
|
44
|
+
provider: "openai",
|
|
45
|
+
maxInputChars: 4096,
|
|
46
|
+
speedRange: { min: 0.25, max: 4 },
|
|
47
|
+
formats: ["mp3", "opus", "aac", "flac", "wav", "pcm"],
|
|
23
48
|
},
|
|
24
|
-
// not a speech to text model?
|
|
25
|
-
/* {
|
|
26
|
-
type: "speech-to-text",
|
|
27
|
-
modelName: "gpt-4o-audio-preview",
|
|
28
|
-
description:
|
|
29
|
-
"This is a preview release of the GPT-4o Audio models. These models accept audio inputs and outputs, and can be used in the Chat Completions REST API. Learn more. The knowledge cutoff for GPT-4o Audio models is October, 2023.",
|
|
30
|
-
inputTokenCost: 2.5,
|
|
31
|
-
outputTokenCost: 10,
|
|
32
|
-
provider: "openai",
|
|
33
|
-
}, */
|
|
34
49
|
];
|
|
35
50
|
export const textModels = [
|
|
36
51
|
{
|
|
@@ -771,16 +786,16 @@ export const textModels = [
|
|
|
771
786
|
{
|
|
772
787
|
type: "text",
|
|
773
788
|
modelName: "gpt-5.6-terra",
|
|
774
|
-
description: "GPT-5.6 Terra balances capability and cost
|
|
789
|
+
description: "GPT-5.6 Terra balances capability and cost. 1M context window. Standard pricing for ≤272K tokens, 2x input/1.5x output for >272K. Reflects the July 30, 2026 price cut. Knowledge cutoff: February 2026.",
|
|
775
790
|
maxInputTokens: 1050000,
|
|
776
791
|
maxOutputTokens: 128000,
|
|
777
|
-
inputTokenCost: 2
|
|
778
|
-
cachedInputTokenCost: 0.
|
|
779
|
-
outputTokenCost:
|
|
792
|
+
inputTokenCost: 2,
|
|
793
|
+
cachedInputTokenCost: 0.2,
|
|
794
|
+
outputTokenCost: 12,
|
|
780
795
|
longContext: {
|
|
781
|
-
inputTokenCost:
|
|
782
|
-
cachedInputTokenCost: 0.
|
|
783
|
-
outputTokenCost:
|
|
796
|
+
inputTokenCost: 4,
|
|
797
|
+
cachedInputTokenCost: 0.4,
|
|
798
|
+
outputTokenCost: 18,
|
|
784
799
|
thresholdTokens: 200000,
|
|
785
800
|
},
|
|
786
801
|
reasoning: {
|
|
@@ -806,16 +821,16 @@ export const textModels = [
|
|
|
806
821
|
{
|
|
807
822
|
type: "text",
|
|
808
823
|
modelName: "gpt-5.6-luna",
|
|
809
|
-
description: "GPT-5.6 Luna is the fast, most affordable member of the GPT-5.6 family. 1M context window. Standard pricing for ≤272K tokens, 2x input/1.5x output for >272K. Knowledge cutoff: February 2026.",
|
|
824
|
+
description: "GPT-5.6 Luna is the fast, most affordable member of the GPT-5.6 family. 1M context window. Standard pricing for ≤272K tokens, 2x input/1.5x output for >272K. Reflects the July 30, 2026 price cut (input/output down ~80%). Knowledge cutoff: February 2026.",
|
|
810
825
|
maxInputTokens: 1050000,
|
|
811
826
|
maxOutputTokens: 128000,
|
|
812
|
-
inputTokenCost:
|
|
813
|
-
cachedInputTokenCost: 0.
|
|
814
|
-
outputTokenCost:
|
|
827
|
+
inputTokenCost: 0.2,
|
|
828
|
+
cachedInputTokenCost: 0.02,
|
|
829
|
+
outputTokenCost: 1.2,
|
|
815
830
|
longContext: {
|
|
816
|
-
inputTokenCost:
|
|
817
|
-
cachedInputTokenCost: 0.
|
|
818
|
-
outputTokenCost:
|
|
831
|
+
inputTokenCost: 0.4,
|
|
832
|
+
cachedInputTokenCost: 0.04,
|
|
833
|
+
outputTokenCost: 1.8,
|
|
819
834
|
thresholdTokens: 200000,
|
|
820
835
|
},
|
|
821
836
|
reasoning: {
|
|
@@ -903,10 +918,40 @@ export const textModels = [
|
|
|
903
918
|
disabled: true,
|
|
904
919
|
provider: "google",
|
|
905
920
|
},
|
|
921
|
+
{
|
|
922
|
+
type: "text",
|
|
923
|
+
modelName: "gemini-3.6-flash",
|
|
924
|
+
description: "Latest Gemini 3.6 Flash model (GA July 21, 2026). Supersedes Gemini 3.5 Flash with better token efficiency and agentic planning at a lower output price ($7.50 vs $9.00/1M). 1M context window, 64K output. Context caching: $0.15/1M read.",
|
|
925
|
+
maxInputTokens: 1048576,
|
|
926
|
+
maxOutputTokens: 65536,
|
|
927
|
+
inputTokenCost: 1.5,
|
|
928
|
+
cachedInputTokenCost: 0.15,
|
|
929
|
+
outputTokenCost: 7.5,
|
|
930
|
+
inputAudioTokenCost: 1.5,
|
|
931
|
+
reasoning: {
|
|
932
|
+
levels: ["minimal", "low", "medium", "high"],
|
|
933
|
+
defaultLevel: "high",
|
|
934
|
+
canDisable: false,
|
|
935
|
+
outputsThinking: true,
|
|
936
|
+
outputsSignatures: true,
|
|
937
|
+
},
|
|
938
|
+
modalities: {
|
|
939
|
+
input: ["text", "image", "video", "audio", "pdf"],
|
|
940
|
+
output: ["text"],
|
|
941
|
+
},
|
|
942
|
+
knowledge: "2026-03",
|
|
943
|
+
releaseDate: "2026-07-21",
|
|
944
|
+
lastUpdated: "2026-07-21",
|
|
945
|
+
family: "gemini-flash",
|
|
946
|
+
openWeights: false,
|
|
947
|
+
structuredOutput: true,
|
|
948
|
+
temperatureSupported: true,
|
|
949
|
+
provider: "google",
|
|
950
|
+
},
|
|
906
951
|
{
|
|
907
952
|
type: "text",
|
|
908
953
|
modelName: "gemini-3.5-flash",
|
|
909
|
-
description: "
|
|
954
|
+
description: "Gemini 3.5 Flash (GA May 2026). Superseded by gemini-3.6-flash. Outperforms Gemini 3.1 Pro on coding and agentic suites at 4x the speed. 1M context window, 64K output. Context caching: $0.15/1M read.",
|
|
910
955
|
maxInputTokens: 1048576,
|
|
911
956
|
maxOutputTokens: 65536,
|
|
912
957
|
inputTokenCost: 1.5,
|
|
@@ -964,10 +1009,40 @@ export const textModels = [
|
|
|
964
1009
|
temperatureSupported: true,
|
|
965
1010
|
provider: "google",
|
|
966
1011
|
},
|
|
1012
|
+
{
|
|
1013
|
+
type: "text",
|
|
1014
|
+
modelName: "gemini-3.5-flash-lite",
|
|
1015
|
+
description: "Most cost-effective Gemini model (GA July 21, 2026). Supersedes gemini-3.1-flash-lite. Built for high-volume, low-reasoning work (search, document processing, translation). Thinking support, 1M context window, 64K output.",
|
|
1016
|
+
maxInputTokens: 1048576,
|
|
1017
|
+
maxOutputTokens: 65536,
|
|
1018
|
+
inputTokenCost: 0.3,
|
|
1019
|
+
cachedInputTokenCost: 0.03,
|
|
1020
|
+
outputTokenCost: 2.5,
|
|
1021
|
+
inputAudioTokenCost: 0.5,
|
|
1022
|
+
reasoning: {
|
|
1023
|
+
levels: ["minimal", "low", "medium", "high"],
|
|
1024
|
+
defaultLevel: "minimal",
|
|
1025
|
+
canDisable: false,
|
|
1026
|
+
outputsThinking: true,
|
|
1027
|
+
outputsSignatures: true,
|
|
1028
|
+
},
|
|
1029
|
+
modalities: {
|
|
1030
|
+
input: ["text", "image", "video", "audio", "pdf"],
|
|
1031
|
+
output: ["text"],
|
|
1032
|
+
},
|
|
1033
|
+
knowledge: "2025-01",
|
|
1034
|
+
releaseDate: "2026-07-21",
|
|
1035
|
+
lastUpdated: "2026-07-21",
|
|
1036
|
+
family: "gemini-flash-lite",
|
|
1037
|
+
openWeights: false,
|
|
1038
|
+
structuredOutput: true,
|
|
1039
|
+
temperatureSupported: true,
|
|
1040
|
+
provider: "google",
|
|
1041
|
+
},
|
|
967
1042
|
{
|
|
968
1043
|
type: "text",
|
|
969
1044
|
modelName: "gemini-3.1-flash-lite",
|
|
970
|
-
description: "
|
|
1045
|
+
description: "Cost-effective Gemini 3.1 model (GA). Superseded by gemini-3.5-flash-lite. Thinking support, 1M context window, 64K output. 2.5x faster TTFA and 45% faster output than 2.5 Flash.",
|
|
971
1046
|
maxInputTokens: 1048576,
|
|
972
1047
|
maxOutputTokens: 65536,
|
|
973
1048
|
inputTokenCost: 0.25,
|
|
@@ -1549,6 +1624,19 @@ export const textModels = [
|
|
|
1549
1624
|
temperatureSupported: false,
|
|
1550
1625
|
provider: "openai-responses",
|
|
1551
1626
|
},
|
|
1627
|
+
{
|
|
1628
|
+
type: "text",
|
|
1629
|
+
modelName: "gpt-audio-1.5",
|
|
1630
|
+
description: "OpenAI GA audio chat model (Chat Completions). Text+audio in, text+audio out.",
|
|
1631
|
+
provider: "openai",
|
|
1632
|
+
modalities: { input: ["text", "audio"], output: ["text", "audio"] },
|
|
1633
|
+
inputTokenCost: 2.5,
|
|
1634
|
+
outputTokenCost: 10,
|
|
1635
|
+
inputAudioTokenCost: 32,
|
|
1636
|
+
outputAudioTokenCost: 64,
|
|
1637
|
+
maxInputTokens: 128000,
|
|
1638
|
+
maxOutputTokens: 16384,
|
|
1639
|
+
},
|
|
1552
1640
|
];
|
|
1553
1641
|
export const imageModels = [
|
|
1554
1642
|
{
|
|
@@ -1730,7 +1818,7 @@ export const hostedTools = [
|
|
|
1730
1818
|
category: "maps_grounding",
|
|
1731
1819
|
description: "Grounding with Google Maps (Gemini 3 only).",
|
|
1732
1820
|
providerToolId: "google_maps",
|
|
1733
|
-
models: ["gemini-3-pro-preview", "gemini-3.1-pro-preview", "gemini-3-flash-preview", "gemini-3.5-flash", "gemini-3.1-flash-lite"],
|
|
1821
|
+
models: ["gemini-3-pro-preview", "gemini-3.1-pro-preview", "gemini-3-flash-preview", "gemini-3.5-flash", "gemini-3.6-flash", "gemini-3.1-flash-lite", "gemini-3.5-flash-lite"],
|
|
1734
1822
|
pricing: { unit: "per_call", note: "Gemini 3 family only; see Google pricing." },
|
|
1735
1823
|
},
|
|
1736
1824
|
{
|
|
@@ -1765,6 +1853,7 @@ function baselineModels() {
|
|
|
1765
1853
|
...textModels,
|
|
1766
1854
|
...imageModels,
|
|
1767
1855
|
...speechToTextModels,
|
|
1856
|
+
...textToSpeechModels,
|
|
1768
1857
|
...registeredTextModels,
|
|
1769
1858
|
...embeddingsModels,
|
|
1770
1859
|
];
|
|
@@ -1790,13 +1879,28 @@ export function getAllModels(requestData) {
|
|
|
1790
1879
|
export function getModel(modelName, requestData) {
|
|
1791
1880
|
return getAllModels(requestData).find((model) => model.modelName === modelName);
|
|
1792
1881
|
}
|
|
1882
|
+
/**
|
|
1883
|
+
* Like `getModel`, but also matches on `provider`. Use this whenever a
|
|
1884
|
+
* modelName may collide across providers (the merge key everywhere else in
|
|
1885
|
+
* this module is `provider:modelName`) — plain `getModel` returns whichever
|
|
1886
|
+
* matching entry comes first and can silently pick the wrong provider.
|
|
1887
|
+
*/
|
|
1888
|
+
export function getModelForProvider(provider, modelName, requestData) {
|
|
1889
|
+
return getAllModels(requestData).find((model) => model.modelName === modelName && model.provider === provider);
|
|
1890
|
+
}
|
|
1793
1891
|
/**
|
|
1794
1892
|
* Whether a model is known to accept the given input modality ("image", "pdf", …).
|
|
1795
1893
|
* Returns undefined when the model is unknown or carries no `modalities` data —
|
|
1796
1894
|
* callers should treat undefined as "don't gate".
|
|
1797
1895
|
*/
|
|
1798
|
-
export function modelSupportsInputModality(modelName, modality, requestData) {
|
|
1799
|
-
|
|
1896
|
+
export function modelSupportsInputModality(modelName, modality, requestData, provider) {
|
|
1897
|
+
let model;
|
|
1898
|
+
if (provider !== undefined) {
|
|
1899
|
+
model = getModelForProvider(provider, modelName, requestData);
|
|
1900
|
+
}
|
|
1901
|
+
else {
|
|
1902
|
+
model = getModel(modelName, requestData);
|
|
1903
|
+
}
|
|
1800
1904
|
if (!model || model.type !== "text") {
|
|
1801
1905
|
return undefined;
|
|
1802
1906
|
}
|
|
@@ -1868,6 +1972,9 @@ export function isTextModel(model) {
|
|
|
1868
1972
|
export function isSpeechToTextModel(model) {
|
|
1869
1973
|
return model.type === "speech-to-text";
|
|
1870
1974
|
}
|
|
1975
|
+
export function isTextToSpeechModel(model) {
|
|
1976
|
+
return model.type === "text-to-speech";
|
|
1977
|
+
}
|
|
1871
1978
|
export function isEmbeddingsModel(model) {
|
|
1872
1979
|
return model.type === "embeddings";
|
|
1873
1980
|
}
|
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
import type { ModelDataBlob } from "../modelData.js";
|
|
2
|
+
import { Result } from "../types/result.js";
|
|
3
|
+
import type { SpeechResult } from "../speech.js";
|
|
4
|
+
export type SpeechClientConfig = {
|
|
5
|
+
model: string;
|
|
6
|
+
/** Resolved provider name. */
|
|
7
|
+
provider: string;
|
|
8
|
+
/** Resolved API key; empty string when none was found. */
|
|
9
|
+
apiKey: string;
|
|
10
|
+
voice: string;
|
|
11
|
+
modelData?: ModelDataBlob;
|
|
12
|
+
/** Output format; provider-specific vocabulary (OpenAI: mp3/opus/aac/flac/wav/pcm). */
|
|
13
|
+
format?: string;
|
|
14
|
+
speed?: number;
|
|
15
|
+
metadata?: Record<string, unknown>;
|
|
16
|
+
};
|
|
17
|
+
/**
|
|
18
|
+
* Shared TTS behavior, mirroring BaseClient for text generation: the public
|
|
19
|
+
* speak() template method owns model-data-driven validation (char cap, speed
|
|
20
|
+
* range, format list), cost, and the single redacting/logging exception
|
|
21
|
+
* boundary. Subclasses implement only _speak(): SDK call + response mapping.
|
|
22
|
+
* A model with no registry entry skips validation — the provider is then the
|
|
23
|
+
* authority, matching how cost is silently omitted for unknown models.
|
|
24
|
+
*/
|
|
25
|
+
export declare abstract class BaseSpeechClient {
|
|
26
|
+
protected config: SpeechClientConfig;
|
|
27
|
+
constructor(config: SpeechClientConfig);
|
|
28
|
+
speak(text: string): Promise<Result<SpeechResult>>;
|
|
29
|
+
/** Provider hook: SDK call + response mapping only; validation and cost live in the base. */
|
|
30
|
+
protected abstract _speak(text: string): Promise<Result<SpeechResult>>;
|
|
31
|
+
}
|
|
@@ -0,0 +1,98 @@
|
|
|
1
|
+
import { getModelForProvider, isTextToSpeechModel, } from "../models.js";
|
|
2
|
+
import { calculateSpeechCost } from "../model.js";
|
|
3
|
+
import { failure } from "../types/result.js";
|
|
4
|
+
import { redactSecret } from "../util/redact.js";
|
|
5
|
+
import { getLogger } from "../util/logger.js";
|
|
6
|
+
/** Validate the declarative TTS constraint block once before consuming it. */
|
|
7
|
+
function speechConstraintError(model) {
|
|
8
|
+
const maxInputChars = model.maxInputChars;
|
|
9
|
+
if (maxInputChars !== undefined &&
|
|
10
|
+
(typeof maxInputChars !== "number" ||
|
|
11
|
+
!Number.isInteger(maxInputChars) ||
|
|
12
|
+
maxInputChars <= 0)) {
|
|
13
|
+
return `Model "${model.modelName}" has an invalid maxInputChars value.`;
|
|
14
|
+
}
|
|
15
|
+
const speedRange = model.speedRange;
|
|
16
|
+
if (speedRange !== undefined) {
|
|
17
|
+
if (typeof speedRange !== "object" || speedRange === null) {
|
|
18
|
+
return `Model "${model.modelName}" has an invalid speedRange.`;
|
|
19
|
+
}
|
|
20
|
+
const min = speedRange.min;
|
|
21
|
+
const max = speedRange.max;
|
|
22
|
+
if (typeof min !== "number" ||
|
|
23
|
+
typeof max !== "number" ||
|
|
24
|
+
!Number.isFinite(min) ||
|
|
25
|
+
!Number.isFinite(max) ||
|
|
26
|
+
min > max) {
|
|
27
|
+
return `Model "${model.modelName}" has an invalid speedRange.`;
|
|
28
|
+
}
|
|
29
|
+
}
|
|
30
|
+
const formats = model.formats;
|
|
31
|
+
if (formats !== undefined &&
|
|
32
|
+
(!Array.isArray(formats) ||
|
|
33
|
+
!formats.every((format) => typeof format === "string"))) {
|
|
34
|
+
return `Model "${model.modelName}" has invalid formats.`;
|
|
35
|
+
}
|
|
36
|
+
return null;
|
|
37
|
+
}
|
|
38
|
+
/**
|
|
39
|
+
* Shared TTS behavior, mirroring BaseClient for text generation: the public
|
|
40
|
+
* speak() template method owns model-data-driven validation (char cap, speed
|
|
41
|
+
* range, format list), cost, and the single redacting/logging exception
|
|
42
|
+
* boundary. Subclasses implement only _speak(): SDK call + response mapping.
|
|
43
|
+
* A model with no registry entry skips validation — the provider is then the
|
|
44
|
+
* authority, matching how cost is silently omitted for unknown models.
|
|
45
|
+
*/
|
|
46
|
+
export class BaseSpeechClient {
|
|
47
|
+
config;
|
|
48
|
+
constructor(config) {
|
|
49
|
+
this.config = config;
|
|
50
|
+
}
|
|
51
|
+
async speak(text) {
|
|
52
|
+
try {
|
|
53
|
+
const model = getModelForProvider(this.config.provider, this.config.model, this.config.modelData);
|
|
54
|
+
if (model !== undefined && !isTextToSpeechModel(model)) {
|
|
55
|
+
return failure(`Model "${this.config.model}" is not a text-to-speech model.`);
|
|
56
|
+
}
|
|
57
|
+
if (model !== undefined) {
|
|
58
|
+
const constraintError = speechConstraintError(model);
|
|
59
|
+
if (constraintError !== null) {
|
|
60
|
+
return failure(constraintError);
|
|
61
|
+
}
|
|
62
|
+
if (model.maxInputChars !== undefined && [...text].length > model.maxInputChars) {
|
|
63
|
+
return failure(`Input exceeds the ${model.maxInputChars}-character limit for model "${this.config.model}".`);
|
|
64
|
+
}
|
|
65
|
+
if (this.config.speed !== undefined && model.speedRange !== undefined) {
|
|
66
|
+
const { min, max } = model.speedRange;
|
|
67
|
+
if (!Number.isFinite(this.config.speed) || this.config.speed < min || this.config.speed > max) {
|
|
68
|
+
return failure(`speed must be a finite number in [${min}, ${max}].`);
|
|
69
|
+
}
|
|
70
|
+
}
|
|
71
|
+
if (this.config.format !== undefined &&
|
|
72
|
+
model.formats !== undefined &&
|
|
73
|
+
!model.formats.includes(this.config.format)) {
|
|
74
|
+
return failure(`Format "${this.config.format}" is not supported by model "${this.config.model}". ` +
|
|
75
|
+
`Supported: ${model.formats.join(", ")}.`);
|
|
76
|
+
}
|
|
77
|
+
}
|
|
78
|
+
const result = await this._speak(text);
|
|
79
|
+
if (!result.success) {
|
|
80
|
+
return result;
|
|
81
|
+
}
|
|
82
|
+
const cost = calculateSpeechCost(model, [...text].length);
|
|
83
|
+
if (cost !== undefined) {
|
|
84
|
+
result.value.cost = cost;
|
|
85
|
+
}
|
|
86
|
+
return result;
|
|
87
|
+
}
|
|
88
|
+
catch (err) {
|
|
89
|
+
let msg = "speak() failed";
|
|
90
|
+
if (err instanceof Error) {
|
|
91
|
+
msg = err.message;
|
|
92
|
+
}
|
|
93
|
+
const redacted = redactSecret(msg, this.config.apiKey);
|
|
94
|
+
getLogger().error("speak() provider failed:", redacted);
|
|
95
|
+
return failure(redacted);
|
|
96
|
+
}
|
|
97
|
+
}
|
|
98
|
+
}
|
|
@@ -0,0 +1,6 @@
|
|
|
1
|
+
import { Result } from "../types/result.js";
|
|
2
|
+
import { BaseSpeechClient } from "./baseSpeechClient.js";
|
|
3
|
+
import type { SpeechResult } from "../speech.js";
|
|
4
|
+
export declare class OpenAISpeechClient extends BaseSpeechClient {
|
|
5
|
+
protected _speak(text: string): Promise<Result<SpeechResult>>;
|
|
6
|
+
}
|