smoltalk 0.9.0 → 0.10.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +175 -7
- package/dist/classes/message/AssistantMessage.d.ts +2 -0
- package/dist/classes/message/UserMessage.d.ts +21 -0
- package/dist/classes/message/UserMessage.js +3 -0
- package/dist/classes/message/contentParts.d.ts +71 -2
- package/dist/classes/message/contentParts.js +6 -0
- package/dist/classes/message/index.d.ts +5 -2
- package/dist/classes/message/index.js +7 -0
- package/dist/classes/message/renderers/AnthropicRenderer.d.ts +2 -1
- package/dist/classes/message/renderers/AnthropicRenderer.js +3 -0
- package/dist/classes/message/renderers/GoogleRenderer.d.ts +2 -1
- package/dist/classes/message/renderers/GoogleRenderer.js +3 -0
- package/dist/classes/message/renderers/JSONRenderer.d.ts +2 -1
- package/dist/classes/message/renderers/JSONRenderer.js +4 -0
- package/dist/classes/message/renderers/OpenAIChatRenderer.d.ts +8 -1
- package/dist/classes/message/renderers/OpenAIChatRenderer.js +18 -0
- package/dist/classes/message/renderers/OpenAIResponsesRenderer.d.ts +2 -1
- package/dist/classes/message/renderers/OpenAIResponsesRenderer.js +3 -0
- package/dist/classes/message/renderers/PartRenderer.d.ts +3 -2
- package/dist/classes/message/renderers/PartRenderer.js +3 -0
- package/dist/client.js +1 -0
- package/dist/clients/anthropic.js +1 -1
- package/dist/clients/baseClient.d.ts +13 -1
- package/dist/clients/baseClient.js +36 -7
- package/dist/clients/google.js +1 -1
- package/dist/clients/ollama.js +1 -1
- package/dist/clients/openai.d.ts +2 -1
- package/dist/clients/openai.js +15 -3
- package/dist/clients/openaiCompat.d.ts +2 -0
- package/dist/clients/openaiCompat.js +5 -0
- package/dist/clients/openaiResponses.js +1 -1
- package/dist/clients/resolveAttachments.d.ts +8 -4
- package/dist/clients/resolveAttachments.js +101 -50
- package/dist/embed.d.ts +4 -0
- package/dist/files.d.ts +1 -1
- package/dist/files.js +1 -1
- package/dist/image/google.js +2 -2
- package/dist/image/openai.js +3 -3
- package/dist/image.d.ts +1 -1
- package/dist/index.d.ts +10 -2
- package/dist/index.js +7 -1
- package/dist/model.d.ts +15 -4
- package/dist/model.js +72 -12
- package/dist/models.d.ts +204 -22
- package/dist/models.js +211 -30
- package/dist/speech/baseSpeechClient.d.ts +36 -0
- package/dist/speech/baseSpeechClient.js +117 -0
- package/dist/speech/google.d.ts +6 -0
- package/dist/speech/google.js +54 -0
- package/dist/speech/groq.d.ts +11 -0
- package/dist/speech/groq.js +19 -0
- package/dist/speech/openai.d.ts +14 -0
- package/dist/speech/openai.js +51 -0
- package/dist/speech/openaiCompat.d.ts +13 -0
- package/dist/speech/openaiCompat.js +22 -0
- package/dist/speech.d.ts +45 -0
- package/dist/speech.js +63 -0
- package/dist/transcription/baseTranscriptionClient.d.ts +36 -0
- package/dist/transcription/baseTranscriptionClient.js +133 -0
- package/dist/transcription/google.d.ts +6 -0
- package/dist/transcription/google.js +56 -0
- package/dist/transcription/groq.d.ts +10 -0
- package/dist/transcription/groq.js +17 -0
- package/dist/transcription/openai.d.ts +11 -0
- package/dist/transcription/openai.js +67 -0
- package/dist/transcription/openaiCompat.d.ts +13 -0
- package/dist/transcription/openaiCompat.js +22 -0
- package/dist/transcription.d.ts +54 -0
- package/dist/transcription.js +64 -0
- package/dist/types/tokenUsage.d.ts +4 -0
- package/dist/types/tokenUsage.js +4 -0
- package/dist/types.d.ts +4 -0
- package/dist/util/attachments.d.ts +1 -1
- package/dist/util/audioMime.d.ts +26 -0
- package/dist/util/audioMime.js +74 -0
- package/dist/util/{imageRef.d.ts → blobRef.d.ts} +9 -9
- package/dist/util/{imageRef.js → blobRef.js} +6 -13
- package/dist/util/googleAudioUsage.d.ts +14 -0
- package/dist/util/googleAudioUsage.js +52 -0
- package/dist/util/mime.d.ts +21 -0
- package/dist/util/mime.js +54 -0
- package/dist/util/modalities.d.ts +6 -2
- package/dist/util/modalities.js +13 -15
- package/dist/util/provider.d.ts +3 -0
- package/dist/util/provider.js +3 -1
- package/package.json +1 -1
package/dist/models.js
CHANGED
|
@@ -12,25 +12,101 @@ export const providers = [
|
|
|
12
12
|
"deepinfra",
|
|
13
13
|
"litellm",
|
|
14
14
|
"openai-compat",
|
|
15
|
+
"groq",
|
|
15
16
|
];
|
|
16
17
|
export const ProviderSchema = z.enum(providers);
|
|
17
18
|
export const speechToTextModels = [
|
|
18
19
|
{
|
|
19
20
|
type: "speech-to-text",
|
|
20
|
-
modelName: "whisper-
|
|
21
|
+
modelName: "whisper-1",
|
|
21
22
|
perMinuteCost: 0.006,
|
|
22
23
|
provider: "openai",
|
|
24
|
+
supportedMimeTypes: [
|
|
25
|
+
"audio/flac", "audio/mpeg", "audio/mp4", "audio/m4a", "audio/ogg",
|
|
26
|
+
"audio/wav", "audio/webm",
|
|
27
|
+
],
|
|
28
|
+
maxBytes: 25 * 1024 * 1024,
|
|
29
|
+
},
|
|
30
|
+
{
|
|
31
|
+
type: "speech-to-text",
|
|
32
|
+
modelName: "whisper-large-v3",
|
|
33
|
+
provider: "groq",
|
|
34
|
+
perMinuteCost: 0.00185, // $0.111/hr, verified 2026-08-09
|
|
35
|
+
minimumBillableSeconds: 10, // Groq bills a 10s minimum per request
|
|
36
|
+
supportedMimeTypes: [
|
|
37
|
+
"audio/flac", "audio/mpeg", "audio/mp4", "audio/m4a", "audio/ogg",
|
|
38
|
+
"audio/wav", "audio/webm",
|
|
39
|
+
],
|
|
40
|
+
// Conservative free-tier / direct-attachment cap; Groq's developer tier
|
|
41
|
+
// allows 100 MB, but a single baked-in record cannot vary by account tier.
|
|
42
|
+
maxBytes: 25 * 1024 * 1024,
|
|
43
|
+
},
|
|
44
|
+
{
|
|
45
|
+
type: "speech-to-text",
|
|
46
|
+
modelName: "whisper-large-v3-turbo",
|
|
47
|
+
provider: "groq",
|
|
48
|
+
perMinuteCost: 0.000667, // $0.04/hr, verified 2026-08-09
|
|
49
|
+
minimumBillableSeconds: 10, // Groq bills a 10s minimum per request
|
|
50
|
+
supportedMimeTypes: [
|
|
51
|
+
"audio/flac", "audio/mpeg", "audio/mp4", "audio/m4a", "audio/ogg",
|
|
52
|
+
"audio/wav", "audio/webm",
|
|
53
|
+
],
|
|
54
|
+
maxBytes: 25 * 1024 * 1024,
|
|
55
|
+
},
|
|
56
|
+
];
|
|
57
|
+
export const textToSpeechModels = [
|
|
58
|
+
{
|
|
59
|
+
type: "text-to-speech",
|
|
60
|
+
modelName: "tts-1",
|
|
61
|
+
perCharacterCost: 0.000015,
|
|
62
|
+
provider: "openai",
|
|
63
|
+
maxInputChars: 4096,
|
|
64
|
+
speedRange: { min: 0.25, max: 4 },
|
|
65
|
+
formats: ["mp3", "opus", "aac", "flac", "wav", "pcm"],
|
|
66
|
+
},
|
|
67
|
+
{
|
|
68
|
+
type: "text-to-speech",
|
|
69
|
+
modelName: "tts-1-hd",
|
|
70
|
+
perCharacterCost: 0.00003,
|
|
71
|
+
provider: "openai",
|
|
72
|
+
maxInputChars: 4096,
|
|
73
|
+
speedRange: { min: 0.25, max: 4 },
|
|
74
|
+
formats: ["mp3", "opus", "aac", "flac", "wav", "pcm"],
|
|
75
|
+
},
|
|
76
|
+
{
|
|
77
|
+
type: "text-to-speech",
|
|
78
|
+
modelName: "canopylabs/orpheus-v1-english",
|
|
79
|
+
provider: "groq",
|
|
80
|
+
perCharacterCost: 0.000022, // $22 / 1M chars, verified 2026-08-09
|
|
81
|
+
maxInputChars: 200,
|
|
82
|
+
formats: ["wav"],
|
|
83
|
+
},
|
|
84
|
+
{
|
|
85
|
+
type: "text-to-speech",
|
|
86
|
+
modelName: "canopylabs/orpheus-arabic-saudi",
|
|
87
|
+
provider: "groq",
|
|
88
|
+
perCharacterCost: 0.00004, // $40 / 1M chars, verified 2026-08-09
|
|
89
|
+
maxInputChars: 200,
|
|
90
|
+
formats: ["wav"],
|
|
91
|
+
},
|
|
92
|
+
// Gemini TTS is token-billed (text input + audio output). No maxInputChars:
|
|
93
|
+
// Gemini documents a 32k-token context, and characters are not a sound proxy.
|
|
94
|
+
{
|
|
95
|
+
type: "text-to-speech",
|
|
96
|
+
modelName: "gemini-2.5-flash-preview-tts",
|
|
97
|
+
provider: "google",
|
|
98
|
+
inputTokenCost: 0.5, // $/1M text-input tokens, verified 2026-08-09
|
|
99
|
+
outputAudioTokenCost: 10.0, // $/1M audio-output tokens
|
|
100
|
+
formats: ["pcm", "wav"],
|
|
101
|
+
},
|
|
102
|
+
{
|
|
103
|
+
type: "text-to-speech",
|
|
104
|
+
modelName: "gemini-2.5-pro-preview-tts",
|
|
105
|
+
provider: "google",
|
|
106
|
+
inputTokenCost: 1.0, // $/1M text-input tokens, verified 2026-08-09
|
|
107
|
+
outputAudioTokenCost: 20.0, // $/1M audio-output tokens
|
|
108
|
+
formats: ["pcm", "wav"],
|
|
23
109
|
},
|
|
24
|
-
// not a speech to text model?
|
|
25
|
-
/* {
|
|
26
|
-
type: "speech-to-text",
|
|
27
|
-
modelName: "gpt-4o-audio-preview",
|
|
28
|
-
description:
|
|
29
|
-
"This is a preview release of the GPT-4o Audio models. These models accept audio inputs and outputs, and can be used in the Chat Completions REST API. Learn more. The knowledge cutoff for GPT-4o Audio models is October, 2023.",
|
|
30
|
-
inputTokenCost: 2.5,
|
|
31
|
-
outputTokenCost: 10,
|
|
32
|
-
provider: "openai",
|
|
33
|
-
}, */
|
|
34
110
|
];
|
|
35
111
|
export const textModels = [
|
|
36
112
|
{
|
|
@@ -771,16 +847,16 @@ export const textModels = [
|
|
|
771
847
|
{
|
|
772
848
|
type: "text",
|
|
773
849
|
modelName: "gpt-5.6-terra",
|
|
774
|
-
description: "GPT-5.6 Terra balances capability and cost
|
|
850
|
+
description: "GPT-5.6 Terra balances capability and cost. 1M context window. Standard pricing for ≤272K tokens, 2x input/1.5x output for >272K. Reflects the July 30, 2026 price cut. Knowledge cutoff: February 2026.",
|
|
775
851
|
maxInputTokens: 1050000,
|
|
776
852
|
maxOutputTokens: 128000,
|
|
777
|
-
inputTokenCost: 2
|
|
778
|
-
cachedInputTokenCost: 0.
|
|
779
|
-
outputTokenCost:
|
|
853
|
+
inputTokenCost: 2,
|
|
854
|
+
cachedInputTokenCost: 0.2,
|
|
855
|
+
outputTokenCost: 12,
|
|
780
856
|
longContext: {
|
|
781
|
-
inputTokenCost:
|
|
782
|
-
cachedInputTokenCost: 0.
|
|
783
|
-
outputTokenCost:
|
|
857
|
+
inputTokenCost: 4,
|
|
858
|
+
cachedInputTokenCost: 0.4,
|
|
859
|
+
outputTokenCost: 18,
|
|
784
860
|
thresholdTokens: 200000,
|
|
785
861
|
},
|
|
786
862
|
reasoning: {
|
|
@@ -806,16 +882,16 @@ export const textModels = [
|
|
|
806
882
|
{
|
|
807
883
|
type: "text",
|
|
808
884
|
modelName: "gpt-5.6-luna",
|
|
809
|
-
description: "GPT-5.6 Luna is the fast, most affordable member of the GPT-5.6 family. 1M context window. Standard pricing for ≤272K tokens, 2x input/1.5x output for >272K. Knowledge cutoff: February 2026.",
|
|
885
|
+
description: "GPT-5.6 Luna is the fast, most affordable member of the GPT-5.6 family. 1M context window. Standard pricing for ≤272K tokens, 2x input/1.5x output for >272K. Reflects the July 30, 2026 price cut (input/output down ~80%). Knowledge cutoff: February 2026.",
|
|
810
886
|
maxInputTokens: 1050000,
|
|
811
887
|
maxOutputTokens: 128000,
|
|
812
|
-
inputTokenCost:
|
|
813
|
-
cachedInputTokenCost: 0.
|
|
814
|
-
outputTokenCost:
|
|
888
|
+
inputTokenCost: 0.2,
|
|
889
|
+
cachedInputTokenCost: 0.02,
|
|
890
|
+
outputTokenCost: 1.2,
|
|
815
891
|
longContext: {
|
|
816
|
-
inputTokenCost:
|
|
817
|
-
cachedInputTokenCost: 0.
|
|
818
|
-
outputTokenCost:
|
|
892
|
+
inputTokenCost: 0.4,
|
|
893
|
+
cachedInputTokenCost: 0.04,
|
|
894
|
+
outputTokenCost: 1.8,
|
|
819
895
|
thresholdTokens: 200000,
|
|
820
896
|
},
|
|
821
897
|
reasoning: {
|
|
@@ -903,10 +979,40 @@ export const textModels = [
|
|
|
903
979
|
disabled: true,
|
|
904
980
|
provider: "google",
|
|
905
981
|
},
|
|
982
|
+
{
|
|
983
|
+
type: "text",
|
|
984
|
+
modelName: "gemini-3.6-flash",
|
|
985
|
+
description: "Latest Gemini 3.6 Flash model (GA July 21, 2026). Supersedes Gemini 3.5 Flash with better token efficiency and agentic planning at a lower output price ($7.50 vs $9.00/1M). 1M context window, 64K output. Context caching: $0.15/1M read.",
|
|
986
|
+
maxInputTokens: 1048576,
|
|
987
|
+
maxOutputTokens: 65536,
|
|
988
|
+
inputTokenCost: 1.5,
|
|
989
|
+
cachedInputTokenCost: 0.15,
|
|
990
|
+
outputTokenCost: 7.5,
|
|
991
|
+
inputAudioTokenCost: 1.5,
|
|
992
|
+
reasoning: {
|
|
993
|
+
levels: ["minimal", "low", "medium", "high"],
|
|
994
|
+
defaultLevel: "high",
|
|
995
|
+
canDisable: false,
|
|
996
|
+
outputsThinking: true,
|
|
997
|
+
outputsSignatures: true,
|
|
998
|
+
},
|
|
999
|
+
modalities: {
|
|
1000
|
+
input: ["text", "image", "video", "audio", "pdf"],
|
|
1001
|
+
output: ["text"],
|
|
1002
|
+
},
|
|
1003
|
+
knowledge: "2026-03",
|
|
1004
|
+
releaseDate: "2026-07-21",
|
|
1005
|
+
lastUpdated: "2026-07-21",
|
|
1006
|
+
family: "gemini-flash",
|
|
1007
|
+
openWeights: false,
|
|
1008
|
+
structuredOutput: true,
|
|
1009
|
+
temperatureSupported: true,
|
|
1010
|
+
provider: "google",
|
|
1011
|
+
},
|
|
906
1012
|
{
|
|
907
1013
|
type: "text",
|
|
908
1014
|
modelName: "gemini-3.5-flash",
|
|
909
|
-
description: "
|
|
1015
|
+
description: "Gemini 3.5 Flash (GA May 2026). Superseded by gemini-3.6-flash. Outperforms Gemini 3.1 Pro on coding and agentic suites at 4x the speed. 1M context window, 64K output. Context caching: $0.15/1M read.",
|
|
910
1016
|
maxInputTokens: 1048576,
|
|
911
1017
|
maxOutputTokens: 65536,
|
|
912
1018
|
inputTokenCost: 1.5,
|
|
@@ -964,10 +1070,40 @@ export const textModels = [
|
|
|
964
1070
|
temperatureSupported: true,
|
|
965
1071
|
provider: "google",
|
|
966
1072
|
},
|
|
1073
|
+
{
|
|
1074
|
+
type: "text",
|
|
1075
|
+
modelName: "gemini-3.5-flash-lite",
|
|
1076
|
+
description: "Most cost-effective Gemini model (GA July 21, 2026). Supersedes gemini-3.1-flash-lite. Built for high-volume, low-reasoning work (search, document processing, translation). Thinking support, 1M context window, 64K output.",
|
|
1077
|
+
maxInputTokens: 1048576,
|
|
1078
|
+
maxOutputTokens: 65536,
|
|
1079
|
+
inputTokenCost: 0.3,
|
|
1080
|
+
cachedInputTokenCost: 0.03,
|
|
1081
|
+
outputTokenCost: 2.5,
|
|
1082
|
+
inputAudioTokenCost: 0.5,
|
|
1083
|
+
reasoning: {
|
|
1084
|
+
levels: ["minimal", "low", "medium", "high"],
|
|
1085
|
+
defaultLevel: "minimal",
|
|
1086
|
+
canDisable: false,
|
|
1087
|
+
outputsThinking: true,
|
|
1088
|
+
outputsSignatures: true,
|
|
1089
|
+
},
|
|
1090
|
+
modalities: {
|
|
1091
|
+
input: ["text", "image", "video", "audio", "pdf"],
|
|
1092
|
+
output: ["text"],
|
|
1093
|
+
},
|
|
1094
|
+
knowledge: "2025-01",
|
|
1095
|
+
releaseDate: "2026-07-21",
|
|
1096
|
+
lastUpdated: "2026-07-21",
|
|
1097
|
+
family: "gemini-flash-lite",
|
|
1098
|
+
openWeights: false,
|
|
1099
|
+
structuredOutput: true,
|
|
1100
|
+
temperatureSupported: true,
|
|
1101
|
+
provider: "google",
|
|
1102
|
+
},
|
|
967
1103
|
{
|
|
968
1104
|
type: "text",
|
|
969
1105
|
modelName: "gemini-3.1-flash-lite",
|
|
970
|
-
description: "
|
|
1106
|
+
description: "Cost-effective Gemini 3.1 model (GA). Superseded by gemini-3.5-flash-lite. Thinking support, 1M context window, 64K output. 2.5x faster TTFA and 45% faster output than 2.5 Flash.",
|
|
971
1107
|
maxInputTokens: 1048576,
|
|
972
1108
|
maxOutputTokens: 65536,
|
|
973
1109
|
inputTokenCost: 0.25,
|
|
@@ -1073,6 +1209,11 @@ export const textModels = [
|
|
|
1073
1209
|
input: ["text", "image", "audio", "video", "pdf"],
|
|
1074
1210
|
output: ["text"],
|
|
1075
1211
|
},
|
|
1212
|
+
// Audio-input (transcription) constraints. maxBytes is a conservative raw cap
|
|
1213
|
+
// leaving room for base64 expansion + instructions under Gemini's 20 MB total
|
|
1214
|
+
// inline request limit; the client also checks the encoded request size.
|
|
1215
|
+
supportedMimeTypes: ["audio/wav", "audio/mpeg", "audio/aac", "audio/ogg", "audio/flac", "audio/aiff"],
|
|
1216
|
+
maxBytes: 14_000_000,
|
|
1076
1217
|
knowledge: "2025-01",
|
|
1077
1218
|
releaseDate: "2025-06-17",
|
|
1078
1219
|
lastUpdated: "2025-06-17",
|
|
@@ -1549,6 +1690,19 @@ export const textModels = [
|
|
|
1549
1690
|
temperatureSupported: false,
|
|
1550
1691
|
provider: "openai-responses",
|
|
1551
1692
|
},
|
|
1693
|
+
{
|
|
1694
|
+
type: "text",
|
|
1695
|
+
modelName: "gpt-audio-1.5",
|
|
1696
|
+
description: "OpenAI GA audio chat model (Chat Completions). Text+audio in, text+audio out.",
|
|
1697
|
+
provider: "openai",
|
|
1698
|
+
modalities: { input: ["text", "audio"], output: ["text", "audio"] },
|
|
1699
|
+
inputTokenCost: 2.5,
|
|
1700
|
+
outputTokenCost: 10,
|
|
1701
|
+
inputAudioTokenCost: 32,
|
|
1702
|
+
outputAudioTokenCost: 64,
|
|
1703
|
+
maxInputTokens: 128000,
|
|
1704
|
+
maxOutputTokens: 16384,
|
|
1705
|
+
},
|
|
1552
1706
|
];
|
|
1553
1707
|
export const imageModels = [
|
|
1554
1708
|
{
|
|
@@ -1730,7 +1884,7 @@ export const hostedTools = [
|
|
|
1730
1884
|
category: "maps_grounding",
|
|
1731
1885
|
description: "Grounding with Google Maps (Gemini 3 only).",
|
|
1732
1886
|
providerToolId: "google_maps",
|
|
1733
|
-
models: ["gemini-3-pro-preview", "gemini-3.1-pro-preview", "gemini-3-flash-preview", "gemini-3.5-flash", "gemini-3.1-flash-lite"],
|
|
1887
|
+
models: ["gemini-3-pro-preview", "gemini-3.1-pro-preview", "gemini-3-flash-preview", "gemini-3.5-flash", "gemini-3.6-flash", "gemini-3.1-flash-lite", "gemini-3.5-flash-lite"],
|
|
1734
1888
|
pricing: { unit: "per_call", note: "Gemini 3 family only; see Google pricing." },
|
|
1735
1889
|
},
|
|
1736
1890
|
{
|
|
@@ -1765,6 +1919,7 @@ function baselineModels() {
|
|
|
1765
1919
|
...textModels,
|
|
1766
1920
|
...imageModels,
|
|
1767
1921
|
...speechToTextModels,
|
|
1922
|
+
...textToSpeechModels,
|
|
1768
1923
|
...registeredTextModels,
|
|
1769
1924
|
...embeddingsModels,
|
|
1770
1925
|
];
|
|
@@ -1790,13 +1945,28 @@ export function getAllModels(requestData) {
|
|
|
1790
1945
|
export function getModel(modelName, requestData) {
|
|
1791
1946
|
return getAllModels(requestData).find((model) => model.modelName === modelName);
|
|
1792
1947
|
}
|
|
1948
|
+
/**
|
|
1949
|
+
* Like `getModel`, but also matches on `provider`. Use this whenever a
|
|
1950
|
+
* modelName may collide across providers (the merge key everywhere else in
|
|
1951
|
+
* this module is `provider:modelName`) — plain `getModel` returns whichever
|
|
1952
|
+
* matching entry comes first and can silently pick the wrong provider.
|
|
1953
|
+
*/
|
|
1954
|
+
export function getModelForProvider(provider, modelName, requestData) {
|
|
1955
|
+
return getAllModels(requestData).find((model) => model.modelName === modelName && model.provider === provider);
|
|
1956
|
+
}
|
|
1793
1957
|
/**
|
|
1794
1958
|
* Whether a model is known to accept the given input modality ("image", "pdf", …).
|
|
1795
1959
|
* Returns undefined when the model is unknown or carries no `modalities` data —
|
|
1796
1960
|
* callers should treat undefined as "don't gate".
|
|
1797
1961
|
*/
|
|
1798
|
-
export function modelSupportsInputModality(modelName, modality, requestData) {
|
|
1799
|
-
|
|
1962
|
+
export function modelSupportsInputModality(modelName, modality, requestData, provider) {
|
|
1963
|
+
let model;
|
|
1964
|
+
if (provider !== undefined) {
|
|
1965
|
+
model = getModelForProvider(provider, modelName, requestData);
|
|
1966
|
+
}
|
|
1967
|
+
else {
|
|
1968
|
+
model = getModel(modelName, requestData);
|
|
1969
|
+
}
|
|
1800
1970
|
if (!model || model.type !== "text") {
|
|
1801
1971
|
return undefined;
|
|
1802
1972
|
}
|
|
@@ -1868,6 +2038,17 @@ export function isTextModel(model) {
|
|
|
1868
2038
|
export function isSpeechToTextModel(model) {
|
|
1869
2039
|
return model.type === "speech-to-text";
|
|
1870
2040
|
}
|
|
2041
|
+
export function isTextToSpeechModel(model) {
|
|
2042
|
+
return model.type === "text-to-speech";
|
|
2043
|
+
}
|
|
2044
|
+
/** Audio-input constraints, readable off either a dedicated STT model or a
|
|
2045
|
+
* multimodal text model. Empty for any other model type. */
|
|
2046
|
+
export function audioInputConstraints(model) {
|
|
2047
|
+
if (model.type === "speech-to-text" || model.type === "text") {
|
|
2048
|
+
return { maxBytes: model.maxBytes, supportedMimeTypes: model.supportedMimeTypes };
|
|
2049
|
+
}
|
|
2050
|
+
return {};
|
|
2051
|
+
}
|
|
1871
2052
|
export function isEmbeddingsModel(model) {
|
|
1872
2053
|
return model.type === "embeddings";
|
|
1873
2054
|
}
|
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
import type { ModelDataBlob } from "../modelData.js";
|
|
2
|
+
import type { SmolConfig } from "../types.js";
|
|
3
|
+
import { Result } from "../types/result.js";
|
|
4
|
+
import type { SpeechResult } from "../speech.js";
|
|
5
|
+
export type SpeechClientConfig = {
|
|
6
|
+
model: string;
|
|
7
|
+
/** Resolved provider name. */
|
|
8
|
+
provider: string;
|
|
9
|
+
/** Resolved API key; empty string when none was found. */
|
|
10
|
+
apiKey: string;
|
|
11
|
+
/** Base-URL map (for OpenAI-compatible providers); read via resolveBaseUrl. */
|
|
12
|
+
baseUrl?: SmolConfig["baseUrl"];
|
|
13
|
+
voice: string;
|
|
14
|
+
modelData?: ModelDataBlob;
|
|
15
|
+
/** Output format; provider-specific vocabulary (OpenAI: mp3/opus/aac/flac/wav/pcm). */
|
|
16
|
+
format?: string;
|
|
17
|
+
speed?: number;
|
|
18
|
+
metadata?: Record<string, unknown>;
|
|
19
|
+
/** Abort the in-flight provider request when this signal fires. */
|
|
20
|
+
abortSignal?: AbortSignal;
|
|
21
|
+
};
|
|
22
|
+
/**
|
|
23
|
+
* Shared TTS behavior, mirroring BaseClient for text generation: the public
|
|
24
|
+
* speak() template method owns model-data-driven validation (char cap, speed
|
|
25
|
+
* range, format list), cost, and the single redacting/logging exception
|
|
26
|
+
* boundary. Subclasses implement only _speak(): SDK call + response mapping.
|
|
27
|
+
* A model with no registry entry skips validation — the provider is then the
|
|
28
|
+
* authority, matching how cost is silently omitted for unknown models.
|
|
29
|
+
*/
|
|
30
|
+
export declare abstract class BaseSpeechClient {
|
|
31
|
+
protected config: SpeechClientConfig;
|
|
32
|
+
constructor(config: SpeechClientConfig);
|
|
33
|
+
speak(text: string): Promise<Result<SpeechResult>>;
|
|
34
|
+
/** Provider hook: SDK call + response mapping only; validation and cost live in the base. */
|
|
35
|
+
protected abstract _speak(text: string): Promise<Result<SpeechResult>>;
|
|
36
|
+
}
|
|
@@ -0,0 +1,117 @@
|
|
|
1
|
+
import { getModelForProvider, isTextToSpeechModel, } from "../models.js";
|
|
2
|
+
import { Model, calculateSpeechCost } from "../model.js";
|
|
3
|
+
import { failure } from "../types/result.js";
|
|
4
|
+
import { redactSecret } from "../util/redact.js";
|
|
5
|
+
import { getLogger } from "../util/logger.js";
|
|
6
|
+
/** Validate the declarative TTS constraint block once before consuming it. */
|
|
7
|
+
function speechConstraintError(model) {
|
|
8
|
+
const maxInputChars = model.maxInputChars;
|
|
9
|
+
if (maxInputChars !== undefined &&
|
|
10
|
+
(typeof maxInputChars !== "number" ||
|
|
11
|
+
!Number.isInteger(maxInputChars) ||
|
|
12
|
+
maxInputChars <= 0)) {
|
|
13
|
+
return `Model "${model.modelName}" has an invalid maxInputChars value.`;
|
|
14
|
+
}
|
|
15
|
+
const speedRange = model.speedRange;
|
|
16
|
+
if (speedRange !== undefined) {
|
|
17
|
+
if (typeof speedRange !== "object" || speedRange === null) {
|
|
18
|
+
return `Model "${model.modelName}" has an invalid speedRange.`;
|
|
19
|
+
}
|
|
20
|
+
const min = speedRange.min;
|
|
21
|
+
const max = speedRange.max;
|
|
22
|
+
if (typeof min !== "number" ||
|
|
23
|
+
typeof max !== "number" ||
|
|
24
|
+
!Number.isFinite(min) ||
|
|
25
|
+
!Number.isFinite(max) ||
|
|
26
|
+
min > max) {
|
|
27
|
+
return `Model "${model.modelName}" has an invalid speedRange.`;
|
|
28
|
+
}
|
|
29
|
+
}
|
|
30
|
+
const formats = model.formats;
|
|
31
|
+
if (formats !== undefined &&
|
|
32
|
+
(!Array.isArray(formats) ||
|
|
33
|
+
!formats.every((format) => typeof format === "string"))) {
|
|
34
|
+
return `Model "${model.modelName}" has invalid formats.`;
|
|
35
|
+
}
|
|
36
|
+
return null;
|
|
37
|
+
}
|
|
38
|
+
/**
|
|
39
|
+
* Shared TTS behavior, mirroring BaseClient for text generation: the public
|
|
40
|
+
* speak() template method owns model-data-driven validation (char cap, speed
|
|
41
|
+
* range, format list), cost, and the single redacting/logging exception
|
|
42
|
+
* boundary. Subclasses implement only _speak(): SDK call + response mapping.
|
|
43
|
+
* A model with no registry entry skips validation — the provider is then the
|
|
44
|
+
* authority, matching how cost is silently omitted for unknown models.
|
|
45
|
+
*/
|
|
46
|
+
export class BaseSpeechClient {
|
|
47
|
+
config;
|
|
48
|
+
constructor(config) {
|
|
49
|
+
this.config = config;
|
|
50
|
+
}
|
|
51
|
+
async speak(text) {
|
|
52
|
+
// Already-aborted signal: stop before doing any paid work.
|
|
53
|
+
if (this.config.abortSignal?.aborted) {
|
|
54
|
+
return failure("Request was aborted");
|
|
55
|
+
}
|
|
56
|
+
try {
|
|
57
|
+
const model = getModelForProvider(this.config.provider, this.config.model, this.config.modelData);
|
|
58
|
+
if (model !== undefined && !isTextToSpeechModel(model)) {
|
|
59
|
+
return failure(`Model "${this.config.model}" is not a text-to-speech model.`);
|
|
60
|
+
}
|
|
61
|
+
if (model !== undefined) {
|
|
62
|
+
const constraintError = speechConstraintError(model);
|
|
63
|
+
if (constraintError !== null) {
|
|
64
|
+
return failure(constraintError);
|
|
65
|
+
}
|
|
66
|
+
if (model.maxInputChars !== undefined && [...text].length > model.maxInputChars) {
|
|
67
|
+
return failure(`Input exceeds the ${model.maxInputChars}-character limit for model "${this.config.model}".`);
|
|
68
|
+
}
|
|
69
|
+
if (this.config.speed !== undefined && model.speedRange !== undefined) {
|
|
70
|
+
const { min, max } = model.speedRange;
|
|
71
|
+
if (!Number.isFinite(this.config.speed) || this.config.speed < min || this.config.speed > max) {
|
|
72
|
+
return failure(`speed must be a finite number in [${min}, ${max}].`);
|
|
73
|
+
}
|
|
74
|
+
}
|
|
75
|
+
if (this.config.format !== undefined &&
|
|
76
|
+
model.formats !== undefined &&
|
|
77
|
+
!model.formats.includes(this.config.format)) {
|
|
78
|
+
return failure(`Format "${this.config.format}" is not supported by model "${this.config.model}". ` +
|
|
79
|
+
`Supported: ${model.formats.join(", ")}.`);
|
|
80
|
+
}
|
|
81
|
+
}
|
|
82
|
+
// Re-check after preflight validation: the signal may have fired during
|
|
83
|
+
// it, and we must not dispatch a request once cancelled.
|
|
84
|
+
if (this.config.abortSignal?.aborted) {
|
|
85
|
+
return failure("Request was aborted");
|
|
86
|
+
}
|
|
87
|
+
const result = await this._speak(text);
|
|
88
|
+
if (!result.success) {
|
|
89
|
+
return result;
|
|
90
|
+
}
|
|
91
|
+
let cost = calculateSpeechCost(model, [...text].length);
|
|
92
|
+
if (cost === undefined && result.value.usage !== undefined) {
|
|
93
|
+
// Token-billed providers (Gemini) price through the shared cost engine.
|
|
94
|
+
cost =
|
|
95
|
+
new Model(this.config.model, this.config.provider, this.config.modelData).calculateCost(result.value.usage) ?? undefined;
|
|
96
|
+
}
|
|
97
|
+
if (cost !== undefined) {
|
|
98
|
+
result.value.cost = cost;
|
|
99
|
+
}
|
|
100
|
+
return result;
|
|
101
|
+
}
|
|
102
|
+
catch (err) {
|
|
103
|
+
// Caller-initiated cancellation surfaces as a distinguishable failure
|
|
104
|
+
// (matching the chat path), not a redacted provider error.
|
|
105
|
+
if (this.config.abortSignal?.aborted) {
|
|
106
|
+
return failure("Request was aborted");
|
|
107
|
+
}
|
|
108
|
+
let msg = "speak() failed";
|
|
109
|
+
if (err instanceof Error) {
|
|
110
|
+
msg = err.message;
|
|
111
|
+
}
|
|
112
|
+
const redacted = redactSecret(msg, this.config.apiKey);
|
|
113
|
+
getLogger().error("speak() provider failed:", redacted);
|
|
114
|
+
return failure(redacted);
|
|
115
|
+
}
|
|
116
|
+
}
|
|
117
|
+
}
|
|
@@ -0,0 +1,6 @@
|
|
|
1
|
+
import { Result } from "../types/result.js";
|
|
2
|
+
import { BaseSpeechClient } from "./baseSpeechClient.js";
|
|
3
|
+
import type { SpeechResult } from "../speech.js";
|
|
4
|
+
export declare class GoogleSpeechClient extends BaseSpeechClient {
|
|
5
|
+
protected _speak(text: string): Promise<Result<SpeechResult>>;
|
|
6
|
+
}
|
|
@@ -0,0 +1,54 @@
|
|
|
1
|
+
import { GoogleGenAI } from "@google/genai";
|
|
2
|
+
import { success, failure } from "../types/result.js";
|
|
3
|
+
import { pcmToWav } from "../util/audioMime.js";
|
|
4
|
+
import { normalizeGoogleAudioUsage } from "../util/googleAudioUsage.js";
|
|
5
|
+
import { BaseSpeechClient } from "./baseSpeechClient.js";
|
|
6
|
+
const GEMINI_PCM = { sampleRateHz: 24000, sampleFormat: "s16le", channels: 1 };
|
|
7
|
+
export class GoogleSpeechClient extends BaseSpeechClient {
|
|
8
|
+
// No try/catch: BaseSpeechClient.speak() is the exception boundary.
|
|
9
|
+
async _speak(text) {
|
|
10
|
+
if (!this.config.apiKey) {
|
|
11
|
+
return failure("No Google API key provided. Set apiKey.google or GEMINI_API_KEY.");
|
|
12
|
+
}
|
|
13
|
+
// Gemini controls pacing via prompt style, not a numeric speed parameter.
|
|
14
|
+
if (this.config.speed !== undefined) {
|
|
15
|
+
return failure("Gemini TTS does not support the 'speed' option; control pacing via the prompt text.");
|
|
16
|
+
}
|
|
17
|
+
const format = this.config.format ?? "pcm";
|
|
18
|
+
if (format !== "pcm" && format !== "wav") {
|
|
19
|
+
return failure(`Gemini TTS only produces raw PCM. Supported formats: pcm (default), wav. Got "${format}".`);
|
|
20
|
+
}
|
|
21
|
+
const ai = new GoogleGenAI({ apiKey: this.config.apiKey });
|
|
22
|
+
const res = await ai.models.generateContent({
|
|
23
|
+
model: this.config.model,
|
|
24
|
+
contents: [{ role: "user", parts: [{ text }] }],
|
|
25
|
+
config: {
|
|
26
|
+
responseModalities: ["AUDIO"],
|
|
27
|
+
speechConfig: {
|
|
28
|
+
voiceConfig: { prebuiltVoiceConfig: { voiceName: this.config.voice } },
|
|
29
|
+
},
|
|
30
|
+
// Client-only cancellation: tears down the request, but Gemini still
|
|
31
|
+
// bills server-side work.
|
|
32
|
+
abortSignal: this.config.abortSignal,
|
|
33
|
+
},
|
|
34
|
+
});
|
|
35
|
+
const dataB64 = res.candidates?.[0]?.content?.parts?.find((part) => part.inlineData?.data !== undefined)?.inlineData?.data;
|
|
36
|
+
if (!dataB64) {
|
|
37
|
+
return failure("Gemini returned no audio data.");
|
|
38
|
+
}
|
|
39
|
+
const pcm = new Uint8Array(Buffer.from(dataB64, "base64"));
|
|
40
|
+
let audio = pcm;
|
|
41
|
+
let mimeType = "application/octet-stream";
|
|
42
|
+
if (format === "wav") {
|
|
43
|
+
audio = pcmToWav(pcm, { sampleRateHz: 24000, channels: 1, bitsPerSample: 16 });
|
|
44
|
+
mimeType = "audio/wav";
|
|
45
|
+
}
|
|
46
|
+
const usage = normalizeGoogleAudioUsage(res.usageMetadata, "output");
|
|
47
|
+
const result = { audio, mimeType, raw: res };
|
|
48
|
+
if (format === "pcm")
|
|
49
|
+
result.pcm = GEMINI_PCM;
|
|
50
|
+
if (usage)
|
|
51
|
+
result.usage = usage;
|
|
52
|
+
return success(result);
|
|
53
|
+
}
|
|
54
|
+
}
|
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
import OpenAI from "openai";
|
|
2
|
+
import { OpenAISpeechClient } from "./openai.js";
|
|
3
|
+
import type { SpeakFormat } from "../util/audioMime.js";
|
|
4
|
+
/**
|
|
5
|
+
* Groq exposes OpenAI-compatible Orpheus TTS and supports WAV only.
|
|
6
|
+
*/
|
|
7
|
+
export declare class GroqSpeechClient extends OpenAISpeechClient {
|
|
8
|
+
protected makeClient(): OpenAI;
|
|
9
|
+
protected defaultFormat(): SpeakFormat;
|
|
10
|
+
protected noKeyMessage(): string;
|
|
11
|
+
}
|
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
import OpenAI from "openai";
|
|
2
|
+
import { OpenAISpeechClient } from "./openai.js";
|
|
3
|
+
/**
|
|
4
|
+
* Groq exposes OpenAI-compatible Orpheus TTS and supports WAV only.
|
|
5
|
+
*/
|
|
6
|
+
export class GroqSpeechClient extends OpenAISpeechClient {
|
|
7
|
+
makeClient() {
|
|
8
|
+
return new OpenAI({
|
|
9
|
+
apiKey: this.config.apiKey,
|
|
10
|
+
baseURL: "https://api.groq.com/openai/v1",
|
|
11
|
+
});
|
|
12
|
+
}
|
|
13
|
+
defaultFormat() {
|
|
14
|
+
return "wav";
|
|
15
|
+
}
|
|
16
|
+
noKeyMessage() {
|
|
17
|
+
return "No Groq API key provided. Set apiKey.groq or GROQ_API_KEY.";
|
|
18
|
+
}
|
|
19
|
+
}
|
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
import OpenAI from "openai";
|
|
2
|
+
import { Result } from "../types/result.js";
|
|
3
|
+
import { type SpeakFormat } from "../util/audioMime.js";
|
|
4
|
+
import { BaseSpeechClient } from "./baseSpeechClient.js";
|
|
5
|
+
import type { SpeechResult } from "../speech.js";
|
|
6
|
+
export declare class OpenAISpeechClient extends BaseSpeechClient {
|
|
7
|
+
/** Build the OpenAI SDK client. Subclasses override to point at a compatible base URL. */
|
|
8
|
+
protected makeClient(): OpenAI;
|
|
9
|
+
/** Provider default used when the declarative call omits format. */
|
|
10
|
+
protected defaultFormat(): SpeakFormat;
|
|
11
|
+
/** Provider-specific diagnostic when no API key is resolved. Subclasses override. */
|
|
12
|
+
protected noKeyMessage(): string;
|
|
13
|
+
protected _speak(text: string): Promise<Result<SpeechResult>>;
|
|
14
|
+
}
|
|
@@ -0,0 +1,51 @@
|
|
|
1
|
+
import OpenAI from "openai";
|
|
2
|
+
import { success, failure } from "../types/result.js";
|
|
3
|
+
import { SPEECH_FORMAT_TO_MIME, isSpeakFormat, } from "../util/audioMime.js";
|
|
4
|
+
import { BaseSpeechClient } from "./baseSpeechClient.js";
|
|
5
|
+
export class OpenAISpeechClient extends BaseSpeechClient {
|
|
6
|
+
/** Build the OpenAI SDK client. Subclasses override to point at a compatible base URL. */
|
|
7
|
+
makeClient() {
|
|
8
|
+
return new OpenAI({ apiKey: this.config.apiKey });
|
|
9
|
+
}
|
|
10
|
+
/** Provider default used when the declarative call omits format. */
|
|
11
|
+
defaultFormat() {
|
|
12
|
+
return "mp3";
|
|
13
|
+
}
|
|
14
|
+
/** Provider-specific diagnostic when no API key is resolved. Subclasses override. */
|
|
15
|
+
noKeyMessage() {
|
|
16
|
+
return "No OpenAI API key provided. Set apiKey.openAi or OPENAI_API_KEY.";
|
|
17
|
+
}
|
|
18
|
+
// No try/catch here: BaseSpeechClient.speak() is the single
|
|
19
|
+
// redacting/logging exception boundary.
|
|
20
|
+
async _speak(text) {
|
|
21
|
+
if (!this.config.apiKey) {
|
|
22
|
+
return failure(this.noKeyMessage());
|
|
23
|
+
}
|
|
24
|
+
// The shared contract carries format as a plain string; narrow to OpenAI's
|
|
25
|
+
// closed union at runtime before indexing the MIME table.
|
|
26
|
+
const requestedFormat = this.config.format ?? this.defaultFormat();
|
|
27
|
+
if (!isSpeakFormat(requestedFormat)) {
|
|
28
|
+
return failure(`Format "${requestedFormat}" is not a supported OpenAI speech format. ` +
|
|
29
|
+
`Supported: ${Object.keys(SPEECH_FORMAT_TO_MIME).join(", ")}.`);
|
|
30
|
+
}
|
|
31
|
+
const format = requestedFormat;
|
|
32
|
+
const mimeType = SPEECH_FORMAT_TO_MIME[format];
|
|
33
|
+
const client = this.makeClient();
|
|
34
|
+
const params = {
|
|
35
|
+
model: this.config.model,
|
|
36
|
+
voice: this.config.voice,
|
|
37
|
+
input: text,
|
|
38
|
+
response_format: format,
|
|
39
|
+
};
|
|
40
|
+
if (this.config.speed !== undefined) {
|
|
41
|
+
params.speed = this.config.speed;
|
|
42
|
+
}
|
|
43
|
+
const res = await client.audio.speech.create(params, { signal: this.config.abortSignal });
|
|
44
|
+
const audio = new Uint8Array(await res.arrayBuffer());
|
|
45
|
+
const result = { audio, mimeType };
|
|
46
|
+
if (format === "pcm") {
|
|
47
|
+
result.pcm = { sampleRateHz: 24000, sampleFormat: "s16le", channels: 1 };
|
|
48
|
+
}
|
|
49
|
+
return success(result);
|
|
50
|
+
}
|
|
51
|
+
}
|
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
import OpenAI from "openai";
|
|
2
|
+
import { OpenAISpeechClient } from "./openai.js";
|
|
3
|
+
/**
|
|
4
|
+
* Generic OpenAI-compatible speech (TTS) client. Point it at any provider
|
|
5
|
+
* exposing an OpenAI-shaped /audio/speech endpoint via
|
|
6
|
+
* `config.baseUrl.openAiCompat` (or OPENAI_COMPAT_BASE_URL) and
|
|
7
|
+
* `config.apiKey.openAiCompat` (or OPENAI_COMPAT_API_KEY). Mirrors the chat
|
|
8
|
+
* `SmolOpenAiCompat` client. Inherits OpenAI's `mp3` default format.
|
|
9
|
+
*/
|
|
10
|
+
export declare class OpenAiCompatSpeechClient extends OpenAISpeechClient {
|
|
11
|
+
protected makeClient(): OpenAI;
|
|
12
|
+
protected noKeyMessage(): string;
|
|
13
|
+
}
|