smoltalk 0.9.0 → 0.10.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (86) hide show
  1. package/README.md +175 -7
  2. package/dist/classes/message/AssistantMessage.d.ts +2 -0
  3. package/dist/classes/message/UserMessage.d.ts +21 -0
  4. package/dist/classes/message/UserMessage.js +3 -0
  5. package/dist/classes/message/contentParts.d.ts +71 -2
  6. package/dist/classes/message/contentParts.js +6 -0
  7. package/dist/classes/message/index.d.ts +5 -2
  8. package/dist/classes/message/index.js +7 -0
  9. package/dist/classes/message/renderers/AnthropicRenderer.d.ts +2 -1
  10. package/dist/classes/message/renderers/AnthropicRenderer.js +3 -0
  11. package/dist/classes/message/renderers/GoogleRenderer.d.ts +2 -1
  12. package/dist/classes/message/renderers/GoogleRenderer.js +3 -0
  13. package/dist/classes/message/renderers/JSONRenderer.d.ts +2 -1
  14. package/dist/classes/message/renderers/JSONRenderer.js +4 -0
  15. package/dist/classes/message/renderers/OpenAIChatRenderer.d.ts +8 -1
  16. package/dist/classes/message/renderers/OpenAIChatRenderer.js +18 -0
  17. package/dist/classes/message/renderers/OpenAIResponsesRenderer.d.ts +2 -1
  18. package/dist/classes/message/renderers/OpenAIResponsesRenderer.js +3 -0
  19. package/dist/classes/message/renderers/PartRenderer.d.ts +3 -2
  20. package/dist/classes/message/renderers/PartRenderer.js +3 -0
  21. package/dist/client.js +1 -0
  22. package/dist/clients/anthropic.js +1 -1
  23. package/dist/clients/baseClient.d.ts +13 -1
  24. package/dist/clients/baseClient.js +36 -7
  25. package/dist/clients/google.js +1 -1
  26. package/dist/clients/ollama.js +1 -1
  27. package/dist/clients/openai.d.ts +2 -1
  28. package/dist/clients/openai.js +15 -3
  29. package/dist/clients/openaiCompat.d.ts +2 -0
  30. package/dist/clients/openaiCompat.js +5 -0
  31. package/dist/clients/openaiResponses.js +1 -1
  32. package/dist/clients/resolveAttachments.d.ts +8 -4
  33. package/dist/clients/resolveAttachments.js +101 -50
  34. package/dist/embed.d.ts +4 -0
  35. package/dist/files.d.ts +1 -1
  36. package/dist/files.js +1 -1
  37. package/dist/image/google.js +2 -2
  38. package/dist/image/openai.js +3 -3
  39. package/dist/image.d.ts +1 -1
  40. package/dist/index.d.ts +10 -2
  41. package/dist/index.js +7 -1
  42. package/dist/model.d.ts +15 -4
  43. package/dist/model.js +72 -12
  44. package/dist/models.d.ts +204 -22
  45. package/dist/models.js +211 -30
  46. package/dist/speech/baseSpeechClient.d.ts +36 -0
  47. package/dist/speech/baseSpeechClient.js +117 -0
  48. package/dist/speech/google.d.ts +6 -0
  49. package/dist/speech/google.js +54 -0
  50. package/dist/speech/groq.d.ts +11 -0
  51. package/dist/speech/groq.js +19 -0
  52. package/dist/speech/openai.d.ts +14 -0
  53. package/dist/speech/openai.js +51 -0
  54. package/dist/speech/openaiCompat.d.ts +13 -0
  55. package/dist/speech/openaiCompat.js +22 -0
  56. package/dist/speech.d.ts +45 -0
  57. package/dist/speech.js +63 -0
  58. package/dist/transcription/baseTranscriptionClient.d.ts +36 -0
  59. package/dist/transcription/baseTranscriptionClient.js +133 -0
  60. package/dist/transcription/google.d.ts +6 -0
  61. package/dist/transcription/google.js +56 -0
  62. package/dist/transcription/groq.d.ts +10 -0
  63. package/dist/transcription/groq.js +17 -0
  64. package/dist/transcription/openai.d.ts +11 -0
  65. package/dist/transcription/openai.js +67 -0
  66. package/dist/transcription/openaiCompat.d.ts +13 -0
  67. package/dist/transcription/openaiCompat.js +22 -0
  68. package/dist/transcription.d.ts +54 -0
  69. package/dist/transcription.js +64 -0
  70. package/dist/types/tokenUsage.d.ts +4 -0
  71. package/dist/types/tokenUsage.js +4 -0
  72. package/dist/types.d.ts +4 -0
  73. package/dist/util/attachments.d.ts +1 -1
  74. package/dist/util/audioMime.d.ts +26 -0
  75. package/dist/util/audioMime.js +74 -0
  76. package/dist/util/{imageRef.d.ts → blobRef.d.ts} +9 -9
  77. package/dist/util/{imageRef.js → blobRef.js} +6 -13
  78. package/dist/util/googleAudioUsage.d.ts +14 -0
  79. package/dist/util/googleAudioUsage.js +52 -0
  80. package/dist/util/mime.d.ts +21 -0
  81. package/dist/util/mime.js +54 -0
  82. package/dist/util/modalities.d.ts +6 -2
  83. package/dist/util/modalities.js +13 -15
  84. package/dist/util/provider.d.ts +3 -0
  85. package/dist/util/provider.js +3 -1
  86. package/package.json +1 -1
package/dist/models.js CHANGED
@@ -12,25 +12,101 @@ export const providers = [
12
12
  "deepinfra",
13
13
  "litellm",
14
14
  "openai-compat",
15
+ "groq",
15
16
  ];
16
17
  export const ProviderSchema = z.enum(providers);
17
18
  export const speechToTextModels = [
18
19
  {
19
20
  type: "speech-to-text",
20
- modelName: "whisper-web",
21
+ modelName: "whisper-1",
21
22
  perMinuteCost: 0.006,
22
23
  provider: "openai",
24
+ supportedMimeTypes: [
25
+ "audio/flac", "audio/mpeg", "audio/mp4", "audio/m4a", "audio/ogg",
26
+ "audio/wav", "audio/webm",
27
+ ],
28
+ maxBytes: 25 * 1024 * 1024,
29
+ },
30
+ {
31
+ type: "speech-to-text",
32
+ modelName: "whisper-large-v3",
33
+ provider: "groq",
34
+ perMinuteCost: 0.00185, // $0.111/hr, verified 2026-08-09
35
+ minimumBillableSeconds: 10, // Groq bills a 10s minimum per request
36
+ supportedMimeTypes: [
37
+ "audio/flac", "audio/mpeg", "audio/mp4", "audio/m4a", "audio/ogg",
38
+ "audio/wav", "audio/webm",
39
+ ],
40
+ // Conservative free-tier / direct-attachment cap; Groq's developer tier
41
+ // allows 100 MB, but a single baked-in record cannot vary by account tier.
42
+ maxBytes: 25 * 1024 * 1024,
43
+ },
44
+ {
45
+ type: "speech-to-text",
46
+ modelName: "whisper-large-v3-turbo",
47
+ provider: "groq",
48
+ perMinuteCost: 0.000667, // $0.04/hr, verified 2026-08-09
49
+ minimumBillableSeconds: 10, // Groq bills a 10s minimum per request
50
+ supportedMimeTypes: [
51
+ "audio/flac", "audio/mpeg", "audio/mp4", "audio/m4a", "audio/ogg",
52
+ "audio/wav", "audio/webm",
53
+ ],
54
+ maxBytes: 25 * 1024 * 1024,
55
+ },
56
+ ];
57
+ export const textToSpeechModels = [
58
+ {
59
+ type: "text-to-speech",
60
+ modelName: "tts-1",
61
+ perCharacterCost: 0.000015,
62
+ provider: "openai",
63
+ maxInputChars: 4096,
64
+ speedRange: { min: 0.25, max: 4 },
65
+ formats: ["mp3", "opus", "aac", "flac", "wav", "pcm"],
66
+ },
67
+ {
68
+ type: "text-to-speech",
69
+ modelName: "tts-1-hd",
70
+ perCharacterCost: 0.00003,
71
+ provider: "openai",
72
+ maxInputChars: 4096,
73
+ speedRange: { min: 0.25, max: 4 },
74
+ formats: ["mp3", "opus", "aac", "flac", "wav", "pcm"],
75
+ },
76
+ {
77
+ type: "text-to-speech",
78
+ modelName: "canopylabs/orpheus-v1-english",
79
+ provider: "groq",
80
+ perCharacterCost: 0.000022, // $22 / 1M chars, verified 2026-08-09
81
+ maxInputChars: 200,
82
+ formats: ["wav"],
83
+ },
84
+ {
85
+ type: "text-to-speech",
86
+ modelName: "canopylabs/orpheus-arabic-saudi",
87
+ provider: "groq",
88
+ perCharacterCost: 0.00004, // $40 / 1M chars, verified 2026-08-09
89
+ maxInputChars: 200,
90
+ formats: ["wav"],
91
+ },
92
+ // Gemini TTS is token-billed (text input + audio output). No maxInputChars:
93
+ // Gemini documents a 32k-token context, and characters are not a sound proxy.
94
+ {
95
+ type: "text-to-speech",
96
+ modelName: "gemini-2.5-flash-preview-tts",
97
+ provider: "google",
98
+ inputTokenCost: 0.5, // $/1M text-input tokens, verified 2026-08-09
99
+ outputAudioTokenCost: 10.0, // $/1M audio-output tokens
100
+ formats: ["pcm", "wav"],
101
+ },
102
+ {
103
+ type: "text-to-speech",
104
+ modelName: "gemini-2.5-pro-preview-tts",
105
+ provider: "google",
106
+ inputTokenCost: 1.0, // $/1M text-input tokens, verified 2026-08-09
107
+ outputAudioTokenCost: 20.0, // $/1M audio-output tokens
108
+ formats: ["pcm", "wav"],
23
109
  },
24
- // not a speech to text model?
25
- /* {
26
- type: "speech-to-text",
27
- modelName: "gpt-4o-audio-preview",
28
- description:
29
- "This is a preview release of the GPT-4o Audio models. These models accept audio inputs and outputs, and can be used in the Chat Completions REST API. Learn more. The knowledge cutoff for GPT-4o Audio models is October, 2023.",
30
- inputTokenCost: 2.5,
31
- outputTokenCost: 10,
32
- provider: "openai",
33
- }, */
34
110
  ];
35
111
  export const textModels = [
36
112
  {
@@ -771,16 +847,16 @@ export const textModels = [
771
847
  {
772
848
  type: "text",
773
849
  modelName: "gpt-5.6-terra",
774
- description: "GPT-5.6 Terra balances capability and cost — competitive with GPT-5.5 at roughly half the price. 1M context window. Standard pricing for ≤272K tokens, 2x input/1.5x output for >272K. Knowledge cutoff: February 2026.",
850
+ description: "GPT-5.6 Terra balances capability and cost. 1M context window. Standard pricing for ≤272K tokens, 2x input/1.5x output for >272K. Reflects the July 30, 2026 price cut. Knowledge cutoff: February 2026.",
775
851
  maxInputTokens: 1050000,
776
852
  maxOutputTokens: 128000,
777
- inputTokenCost: 2.5,
778
- cachedInputTokenCost: 0.25,
779
- outputTokenCost: 15,
853
+ inputTokenCost: 2,
854
+ cachedInputTokenCost: 0.2,
855
+ outputTokenCost: 12,
780
856
  longContext: {
781
- inputTokenCost: 5,
782
- cachedInputTokenCost: 0.5,
783
- outputTokenCost: 22.5,
857
+ inputTokenCost: 4,
858
+ cachedInputTokenCost: 0.4,
859
+ outputTokenCost: 18,
784
860
  thresholdTokens: 200000,
785
861
  },
786
862
  reasoning: {
@@ -806,16 +882,16 @@ export const textModels = [
806
882
  {
807
883
  type: "text",
808
884
  modelName: "gpt-5.6-luna",
809
- description: "GPT-5.6 Luna is the fast, most affordable member of the GPT-5.6 family. 1M context window. Standard pricing for ≤272K tokens, 2x input/1.5x output for >272K. Knowledge cutoff: February 2026.",
885
+ description: "GPT-5.6 Luna is the fast, most affordable member of the GPT-5.6 family. 1M context window. Standard pricing for ≤272K tokens, 2x input/1.5x output for >272K. Reflects the July 30, 2026 price cut (input/output down ~80%). Knowledge cutoff: February 2026.",
810
886
  maxInputTokens: 1050000,
811
887
  maxOutputTokens: 128000,
812
- inputTokenCost: 1,
813
- cachedInputTokenCost: 0.1,
814
- outputTokenCost: 6,
888
+ inputTokenCost: 0.2,
889
+ cachedInputTokenCost: 0.02,
890
+ outputTokenCost: 1.2,
815
891
  longContext: {
816
- inputTokenCost: 2,
817
- cachedInputTokenCost: 0.2,
818
- outputTokenCost: 9,
892
+ inputTokenCost: 0.4,
893
+ cachedInputTokenCost: 0.04,
894
+ outputTokenCost: 1.8,
819
895
  thresholdTokens: 200000,
820
896
  },
821
897
  reasoning: {
@@ -903,10 +979,40 @@ export const textModels = [
903
979
  disabled: true,
904
980
  provider: "google",
905
981
  },
982
+ {
983
+ type: "text",
984
+ modelName: "gemini-3.6-flash",
985
+ description: "Latest Gemini 3.6 Flash model (GA July 21, 2026). Supersedes Gemini 3.5 Flash with better token efficiency and agentic planning at a lower output price ($7.50 vs $9.00/1M). 1M context window, 64K output. Context caching: $0.15/1M read.",
986
+ maxInputTokens: 1048576,
987
+ maxOutputTokens: 65536,
988
+ inputTokenCost: 1.5,
989
+ cachedInputTokenCost: 0.15,
990
+ outputTokenCost: 7.5,
991
+ inputAudioTokenCost: 1.5,
992
+ reasoning: {
993
+ levels: ["minimal", "low", "medium", "high"],
994
+ defaultLevel: "high",
995
+ canDisable: false,
996
+ outputsThinking: true,
997
+ outputsSignatures: true,
998
+ },
999
+ modalities: {
1000
+ input: ["text", "image", "video", "audio", "pdf"],
1001
+ output: ["text"],
1002
+ },
1003
+ knowledge: "2026-03",
1004
+ releaseDate: "2026-07-21",
1005
+ lastUpdated: "2026-07-21",
1006
+ family: "gemini-flash",
1007
+ openWeights: false,
1008
+ structuredOutput: true,
1009
+ temperatureSupported: true,
1010
+ provider: "google",
1011
+ },
906
1012
  {
907
1013
  type: "text",
908
1014
  modelName: "gemini-3.5-flash",
909
- description: "Latest Gemini 3.5 Flash model (GA May 2026). Outperforms Gemini 3.1 Pro on coding and agentic suites at 4x the speed. 1M context window, 64K output. Context caching: $0.15/1M read.",
1015
+ description: "Gemini 3.5 Flash (GA May 2026). Superseded by gemini-3.6-flash. Outperforms Gemini 3.1 Pro on coding and agentic suites at 4x the speed. 1M context window, 64K output. Context caching: $0.15/1M read.",
910
1016
  maxInputTokens: 1048576,
911
1017
  maxOutputTokens: 65536,
912
1018
  inputTokenCost: 1.5,
@@ -964,10 +1070,40 @@ export const textModels = [
964
1070
  temperatureSupported: true,
965
1071
  provider: "google",
966
1072
  },
1073
+ {
1074
+ type: "text",
1075
+ modelName: "gemini-3.5-flash-lite",
1076
+ description: "Most cost-effective Gemini model (GA July 21, 2026). Supersedes gemini-3.1-flash-lite. Built for high-volume, low-reasoning work (search, document processing, translation). Thinking support, 1M context window, 64K output.",
1077
+ maxInputTokens: 1048576,
1078
+ maxOutputTokens: 65536,
1079
+ inputTokenCost: 0.3,
1080
+ cachedInputTokenCost: 0.03,
1081
+ outputTokenCost: 2.5,
1082
+ inputAudioTokenCost: 0.5,
1083
+ reasoning: {
1084
+ levels: ["minimal", "low", "medium", "high"],
1085
+ defaultLevel: "minimal",
1086
+ canDisable: false,
1087
+ outputsThinking: true,
1088
+ outputsSignatures: true,
1089
+ },
1090
+ modalities: {
1091
+ input: ["text", "image", "video", "audio", "pdf"],
1092
+ output: ["text"],
1093
+ },
1094
+ knowledge: "2025-01",
1095
+ releaseDate: "2026-07-21",
1096
+ lastUpdated: "2026-07-21",
1097
+ family: "gemini-flash-lite",
1098
+ openWeights: false,
1099
+ structuredOutput: true,
1100
+ temperatureSupported: true,
1101
+ provider: "google",
1102
+ },
967
1103
  {
968
1104
  type: "text",
969
1105
  modelName: "gemini-3.1-flash-lite",
970
- description: "Most cost-effective Gemini 3.1 model (GA). Thinking support, 1M context window, 64K output. 2.5x faster TTFA and 45% faster output than 2.5 Flash.",
1106
+ description: "Cost-effective Gemini 3.1 model (GA). Superseded by gemini-3.5-flash-lite. Thinking support, 1M context window, 64K output. 2.5x faster TTFA and 45% faster output than 2.5 Flash.",
971
1107
  maxInputTokens: 1048576,
972
1108
  maxOutputTokens: 65536,
973
1109
  inputTokenCost: 0.25,
@@ -1073,6 +1209,11 @@ export const textModels = [
1073
1209
  input: ["text", "image", "audio", "video", "pdf"],
1074
1210
  output: ["text"],
1075
1211
  },
1212
+ // Audio-input (transcription) constraints. maxBytes is a conservative raw cap
1213
+ // leaving room for base64 expansion + instructions under Gemini's 20 MB total
1214
+ // inline request limit; the client also checks the encoded request size.
1215
+ supportedMimeTypes: ["audio/wav", "audio/mpeg", "audio/aac", "audio/ogg", "audio/flac", "audio/aiff"],
1216
+ maxBytes: 14_000_000,
1076
1217
  knowledge: "2025-01",
1077
1218
  releaseDate: "2025-06-17",
1078
1219
  lastUpdated: "2025-06-17",
@@ -1549,6 +1690,19 @@ export const textModels = [
1549
1690
  temperatureSupported: false,
1550
1691
  provider: "openai-responses",
1551
1692
  },
1693
+ {
1694
+ type: "text",
1695
+ modelName: "gpt-audio-1.5",
1696
+ description: "OpenAI GA audio chat model (Chat Completions). Text+audio in, text+audio out.",
1697
+ provider: "openai",
1698
+ modalities: { input: ["text", "audio"], output: ["text", "audio"] },
1699
+ inputTokenCost: 2.5,
1700
+ outputTokenCost: 10,
1701
+ inputAudioTokenCost: 32,
1702
+ outputAudioTokenCost: 64,
1703
+ maxInputTokens: 128000,
1704
+ maxOutputTokens: 16384,
1705
+ },
1552
1706
  ];
1553
1707
  export const imageModels = [
1554
1708
  {
@@ -1730,7 +1884,7 @@ export const hostedTools = [
1730
1884
  category: "maps_grounding",
1731
1885
  description: "Grounding with Google Maps (Gemini 3 only).",
1732
1886
  providerToolId: "google_maps",
1733
- models: ["gemini-3-pro-preview", "gemini-3.1-pro-preview", "gemini-3-flash-preview", "gemini-3.5-flash", "gemini-3.1-flash-lite"],
1887
+ models: ["gemini-3-pro-preview", "gemini-3.1-pro-preview", "gemini-3-flash-preview", "gemini-3.5-flash", "gemini-3.6-flash", "gemini-3.1-flash-lite", "gemini-3.5-flash-lite"],
1734
1888
  pricing: { unit: "per_call", note: "Gemini 3 family only; see Google pricing." },
1735
1889
  },
1736
1890
  {
@@ -1765,6 +1919,7 @@ function baselineModels() {
1765
1919
  ...textModels,
1766
1920
  ...imageModels,
1767
1921
  ...speechToTextModels,
1922
+ ...textToSpeechModels,
1768
1923
  ...registeredTextModels,
1769
1924
  ...embeddingsModels,
1770
1925
  ];
@@ -1790,13 +1945,28 @@ export function getAllModels(requestData) {
1790
1945
  export function getModel(modelName, requestData) {
1791
1946
  return getAllModels(requestData).find((model) => model.modelName === modelName);
1792
1947
  }
1948
+ /**
1949
+ * Like `getModel`, but also matches on `provider`. Use this whenever a
1950
+ * modelName may collide across providers (the merge key everywhere else in
1951
+ * this module is `provider:modelName`) — plain `getModel` returns whichever
1952
+ * matching entry comes first and can silently pick the wrong provider.
1953
+ */
1954
+ export function getModelForProvider(provider, modelName, requestData) {
1955
+ return getAllModels(requestData).find((model) => model.modelName === modelName && model.provider === provider);
1956
+ }
1793
1957
  /**
1794
1958
  * Whether a model is known to accept the given input modality ("image", "pdf", …).
1795
1959
  * Returns undefined when the model is unknown or carries no `modalities` data —
1796
1960
  * callers should treat undefined as "don't gate".
1797
1961
  */
1798
- export function modelSupportsInputModality(modelName, modality, requestData) {
1799
- const model = getModel(modelName, requestData);
1962
+ export function modelSupportsInputModality(modelName, modality, requestData, provider) {
1963
+ let model;
1964
+ if (provider !== undefined) {
1965
+ model = getModelForProvider(provider, modelName, requestData);
1966
+ }
1967
+ else {
1968
+ model = getModel(modelName, requestData);
1969
+ }
1800
1970
  if (!model || model.type !== "text") {
1801
1971
  return undefined;
1802
1972
  }
@@ -1868,6 +2038,17 @@ export function isTextModel(model) {
1868
2038
  export function isSpeechToTextModel(model) {
1869
2039
  return model.type === "speech-to-text";
1870
2040
  }
2041
+ export function isTextToSpeechModel(model) {
2042
+ return model.type === "text-to-speech";
2043
+ }
2044
+ /** Audio-input constraints, readable off either a dedicated STT model or a
2045
+ * multimodal text model. Empty for any other model type. */
2046
+ export function audioInputConstraints(model) {
2047
+ if (model.type === "speech-to-text" || model.type === "text") {
2048
+ return { maxBytes: model.maxBytes, supportedMimeTypes: model.supportedMimeTypes };
2049
+ }
2050
+ return {};
2051
+ }
1871
2052
  export function isEmbeddingsModel(model) {
1872
2053
  return model.type === "embeddings";
1873
2054
  }
@@ -0,0 +1,36 @@
1
+ import type { ModelDataBlob } from "../modelData.js";
2
+ import type { SmolConfig } from "../types.js";
3
+ import { Result } from "../types/result.js";
4
+ import type { SpeechResult } from "../speech.js";
5
+ export type SpeechClientConfig = {
6
+ model: string;
7
+ /** Resolved provider name. */
8
+ provider: string;
9
+ /** Resolved API key; empty string when none was found. */
10
+ apiKey: string;
11
+ /** Base-URL map (for OpenAI-compatible providers); read via resolveBaseUrl. */
12
+ baseUrl?: SmolConfig["baseUrl"];
13
+ voice: string;
14
+ modelData?: ModelDataBlob;
15
+ /** Output format; provider-specific vocabulary (OpenAI: mp3/opus/aac/flac/wav/pcm). */
16
+ format?: string;
17
+ speed?: number;
18
+ metadata?: Record<string, unknown>;
19
+ /** Abort the in-flight provider request when this signal fires. */
20
+ abortSignal?: AbortSignal;
21
+ };
22
+ /**
23
+ * Shared TTS behavior, mirroring BaseClient for text generation: the public
24
+ * speak() template method owns model-data-driven validation (char cap, speed
25
+ * range, format list), cost, and the single redacting/logging exception
26
+ * boundary. Subclasses implement only _speak(): SDK call + response mapping.
27
+ * A model with no registry entry skips validation — the provider is then the
28
+ * authority, matching how cost is silently omitted for unknown models.
29
+ */
30
+ export declare abstract class BaseSpeechClient {
31
+ protected config: SpeechClientConfig;
32
+ constructor(config: SpeechClientConfig);
33
+ speak(text: string): Promise<Result<SpeechResult>>;
34
+ /** Provider hook: SDK call + response mapping only; validation and cost live in the base. */
35
+ protected abstract _speak(text: string): Promise<Result<SpeechResult>>;
36
+ }
@@ -0,0 +1,117 @@
1
+ import { getModelForProvider, isTextToSpeechModel, } from "../models.js";
2
+ import { Model, calculateSpeechCost } from "../model.js";
3
+ import { failure } from "../types/result.js";
4
+ import { redactSecret } from "../util/redact.js";
5
+ import { getLogger } from "../util/logger.js";
6
+ /** Validate the declarative TTS constraint block once before consuming it. */
7
+ function speechConstraintError(model) {
8
+ const maxInputChars = model.maxInputChars;
9
+ if (maxInputChars !== undefined &&
10
+ (typeof maxInputChars !== "number" ||
11
+ !Number.isInteger(maxInputChars) ||
12
+ maxInputChars <= 0)) {
13
+ return `Model "${model.modelName}" has an invalid maxInputChars value.`;
14
+ }
15
+ const speedRange = model.speedRange;
16
+ if (speedRange !== undefined) {
17
+ if (typeof speedRange !== "object" || speedRange === null) {
18
+ return `Model "${model.modelName}" has an invalid speedRange.`;
19
+ }
20
+ const min = speedRange.min;
21
+ const max = speedRange.max;
22
+ if (typeof min !== "number" ||
23
+ typeof max !== "number" ||
24
+ !Number.isFinite(min) ||
25
+ !Number.isFinite(max) ||
26
+ min > max) {
27
+ return `Model "${model.modelName}" has an invalid speedRange.`;
28
+ }
29
+ }
30
+ const formats = model.formats;
31
+ if (formats !== undefined &&
32
+ (!Array.isArray(formats) ||
33
+ !formats.every((format) => typeof format === "string"))) {
34
+ return `Model "${model.modelName}" has invalid formats.`;
35
+ }
36
+ return null;
37
+ }
38
+ /**
39
+ * Shared TTS behavior, mirroring BaseClient for text generation: the public
40
+ * speak() template method owns model-data-driven validation (char cap, speed
41
+ * range, format list), cost, and the single redacting/logging exception
42
+ * boundary. Subclasses implement only _speak(): SDK call + response mapping.
43
+ * A model with no registry entry skips validation — the provider is then the
44
+ * authority, matching how cost is silently omitted for unknown models.
45
+ */
46
+ export class BaseSpeechClient {
47
+ config;
48
+ constructor(config) {
49
+ this.config = config;
50
+ }
51
+ async speak(text) {
52
+ // Already-aborted signal: stop before doing any paid work.
53
+ if (this.config.abortSignal?.aborted) {
54
+ return failure("Request was aborted");
55
+ }
56
+ try {
57
+ const model = getModelForProvider(this.config.provider, this.config.model, this.config.modelData);
58
+ if (model !== undefined && !isTextToSpeechModel(model)) {
59
+ return failure(`Model "${this.config.model}" is not a text-to-speech model.`);
60
+ }
61
+ if (model !== undefined) {
62
+ const constraintError = speechConstraintError(model);
63
+ if (constraintError !== null) {
64
+ return failure(constraintError);
65
+ }
66
+ if (model.maxInputChars !== undefined && [...text].length > model.maxInputChars) {
67
+ return failure(`Input exceeds the ${model.maxInputChars}-character limit for model "${this.config.model}".`);
68
+ }
69
+ if (this.config.speed !== undefined && model.speedRange !== undefined) {
70
+ const { min, max } = model.speedRange;
71
+ if (!Number.isFinite(this.config.speed) || this.config.speed < min || this.config.speed > max) {
72
+ return failure(`speed must be a finite number in [${min}, ${max}].`);
73
+ }
74
+ }
75
+ if (this.config.format !== undefined &&
76
+ model.formats !== undefined &&
77
+ !model.formats.includes(this.config.format)) {
78
+ return failure(`Format "${this.config.format}" is not supported by model "${this.config.model}". ` +
79
+ `Supported: ${model.formats.join(", ")}.`);
80
+ }
81
+ }
82
+ // Re-check after preflight validation: the signal may have fired during
83
+ // it, and we must not dispatch a request once cancelled.
84
+ if (this.config.abortSignal?.aborted) {
85
+ return failure("Request was aborted");
86
+ }
87
+ const result = await this._speak(text);
88
+ if (!result.success) {
89
+ return result;
90
+ }
91
+ let cost = calculateSpeechCost(model, [...text].length);
92
+ if (cost === undefined && result.value.usage !== undefined) {
93
+ // Token-billed providers (Gemini) price through the shared cost engine.
94
+ cost =
95
+ new Model(this.config.model, this.config.provider, this.config.modelData).calculateCost(result.value.usage) ?? undefined;
96
+ }
97
+ if (cost !== undefined) {
98
+ result.value.cost = cost;
99
+ }
100
+ return result;
101
+ }
102
+ catch (err) {
103
+ // Caller-initiated cancellation surfaces as a distinguishable failure
104
+ // (matching the chat path), not a redacted provider error.
105
+ if (this.config.abortSignal?.aborted) {
106
+ return failure("Request was aborted");
107
+ }
108
+ let msg = "speak() failed";
109
+ if (err instanceof Error) {
110
+ msg = err.message;
111
+ }
112
+ const redacted = redactSecret(msg, this.config.apiKey);
113
+ getLogger().error("speak() provider failed:", redacted);
114
+ return failure(redacted);
115
+ }
116
+ }
117
+ }
@@ -0,0 +1,6 @@
1
+ import { Result } from "../types/result.js";
2
+ import { BaseSpeechClient } from "./baseSpeechClient.js";
3
+ import type { SpeechResult } from "../speech.js";
4
+ export declare class GoogleSpeechClient extends BaseSpeechClient {
5
+ protected _speak(text: string): Promise<Result<SpeechResult>>;
6
+ }
@@ -0,0 +1,54 @@
1
+ import { GoogleGenAI } from "@google/genai";
2
+ import { success, failure } from "../types/result.js";
3
+ import { pcmToWav } from "../util/audioMime.js";
4
+ import { normalizeGoogleAudioUsage } from "../util/googleAudioUsage.js";
5
+ import { BaseSpeechClient } from "./baseSpeechClient.js";
6
+ const GEMINI_PCM = { sampleRateHz: 24000, sampleFormat: "s16le", channels: 1 };
7
+ export class GoogleSpeechClient extends BaseSpeechClient {
8
+ // No try/catch: BaseSpeechClient.speak() is the exception boundary.
9
+ async _speak(text) {
10
+ if (!this.config.apiKey) {
11
+ return failure("No Google API key provided. Set apiKey.google or GEMINI_API_KEY.");
12
+ }
13
+ // Gemini controls pacing via prompt style, not a numeric speed parameter.
14
+ if (this.config.speed !== undefined) {
15
+ return failure("Gemini TTS does not support the 'speed' option; control pacing via the prompt text.");
16
+ }
17
+ const format = this.config.format ?? "pcm";
18
+ if (format !== "pcm" && format !== "wav") {
19
+ return failure(`Gemini TTS only produces raw PCM. Supported formats: pcm (default), wav. Got "${format}".`);
20
+ }
21
+ const ai = new GoogleGenAI({ apiKey: this.config.apiKey });
22
+ const res = await ai.models.generateContent({
23
+ model: this.config.model,
24
+ contents: [{ role: "user", parts: [{ text }] }],
25
+ config: {
26
+ responseModalities: ["AUDIO"],
27
+ speechConfig: {
28
+ voiceConfig: { prebuiltVoiceConfig: { voiceName: this.config.voice } },
29
+ },
30
+ // Client-only cancellation: tears down the request, but Gemini still
31
+ // bills server-side work.
32
+ abortSignal: this.config.abortSignal,
33
+ },
34
+ });
35
+ const dataB64 = res.candidates?.[0]?.content?.parts?.find((part) => part.inlineData?.data !== undefined)?.inlineData?.data;
36
+ if (!dataB64) {
37
+ return failure("Gemini returned no audio data.");
38
+ }
39
+ const pcm = new Uint8Array(Buffer.from(dataB64, "base64"));
40
+ let audio = pcm;
41
+ let mimeType = "application/octet-stream";
42
+ if (format === "wav") {
43
+ audio = pcmToWav(pcm, { sampleRateHz: 24000, channels: 1, bitsPerSample: 16 });
44
+ mimeType = "audio/wav";
45
+ }
46
+ const usage = normalizeGoogleAudioUsage(res.usageMetadata, "output");
47
+ const result = { audio, mimeType, raw: res };
48
+ if (format === "pcm")
49
+ result.pcm = GEMINI_PCM;
50
+ if (usage)
51
+ result.usage = usage;
52
+ return success(result);
53
+ }
54
+ }
@@ -0,0 +1,11 @@
1
+ import OpenAI from "openai";
2
+ import { OpenAISpeechClient } from "./openai.js";
3
+ import type { SpeakFormat } from "../util/audioMime.js";
4
+ /**
5
+ * Groq exposes OpenAI-compatible Orpheus TTS and supports WAV only.
6
+ */
7
+ export declare class GroqSpeechClient extends OpenAISpeechClient {
8
+ protected makeClient(): OpenAI;
9
+ protected defaultFormat(): SpeakFormat;
10
+ protected noKeyMessage(): string;
11
+ }
@@ -0,0 +1,19 @@
1
+ import OpenAI from "openai";
2
+ import { OpenAISpeechClient } from "./openai.js";
3
+ /**
4
+ * Groq exposes OpenAI-compatible Orpheus TTS and supports WAV only.
5
+ */
6
+ export class GroqSpeechClient extends OpenAISpeechClient {
7
+ makeClient() {
8
+ return new OpenAI({
9
+ apiKey: this.config.apiKey,
10
+ baseURL: "https://api.groq.com/openai/v1",
11
+ });
12
+ }
13
+ defaultFormat() {
14
+ return "wav";
15
+ }
16
+ noKeyMessage() {
17
+ return "No Groq API key provided. Set apiKey.groq or GROQ_API_KEY.";
18
+ }
19
+ }
@@ -0,0 +1,14 @@
1
+ import OpenAI from "openai";
2
+ import { Result } from "../types/result.js";
3
+ import { type SpeakFormat } from "../util/audioMime.js";
4
+ import { BaseSpeechClient } from "./baseSpeechClient.js";
5
+ import type { SpeechResult } from "../speech.js";
6
+ export declare class OpenAISpeechClient extends BaseSpeechClient {
7
+ /** Build the OpenAI SDK client. Subclasses override to point at a compatible base URL. */
8
+ protected makeClient(): OpenAI;
9
+ /** Provider default used when the declarative call omits format. */
10
+ protected defaultFormat(): SpeakFormat;
11
+ /** Provider-specific diagnostic when no API key is resolved. Subclasses override. */
12
+ protected noKeyMessage(): string;
13
+ protected _speak(text: string): Promise<Result<SpeechResult>>;
14
+ }
@@ -0,0 +1,51 @@
1
+ import OpenAI from "openai";
2
+ import { success, failure } from "../types/result.js";
3
+ import { SPEECH_FORMAT_TO_MIME, isSpeakFormat, } from "../util/audioMime.js";
4
+ import { BaseSpeechClient } from "./baseSpeechClient.js";
5
+ export class OpenAISpeechClient extends BaseSpeechClient {
6
+ /** Build the OpenAI SDK client. Subclasses override to point at a compatible base URL. */
7
+ makeClient() {
8
+ return new OpenAI({ apiKey: this.config.apiKey });
9
+ }
10
+ /** Provider default used when the declarative call omits format. */
11
+ defaultFormat() {
12
+ return "mp3";
13
+ }
14
+ /** Provider-specific diagnostic when no API key is resolved. Subclasses override. */
15
+ noKeyMessage() {
16
+ return "No OpenAI API key provided. Set apiKey.openAi or OPENAI_API_KEY.";
17
+ }
18
+ // No try/catch here: BaseSpeechClient.speak() is the single
19
+ // redacting/logging exception boundary.
20
+ async _speak(text) {
21
+ if (!this.config.apiKey) {
22
+ return failure(this.noKeyMessage());
23
+ }
24
+ // The shared contract carries format as a plain string; narrow to OpenAI's
25
+ // closed union at runtime before indexing the MIME table.
26
+ const requestedFormat = this.config.format ?? this.defaultFormat();
27
+ if (!isSpeakFormat(requestedFormat)) {
28
+ return failure(`Format "${requestedFormat}" is not a supported OpenAI speech format. ` +
29
+ `Supported: ${Object.keys(SPEECH_FORMAT_TO_MIME).join(", ")}.`);
30
+ }
31
+ const format = requestedFormat;
32
+ const mimeType = SPEECH_FORMAT_TO_MIME[format];
33
+ const client = this.makeClient();
34
+ const params = {
35
+ model: this.config.model,
36
+ voice: this.config.voice,
37
+ input: text,
38
+ response_format: format,
39
+ };
40
+ if (this.config.speed !== undefined) {
41
+ params.speed = this.config.speed;
42
+ }
43
+ const res = await client.audio.speech.create(params, { signal: this.config.abortSignal });
44
+ const audio = new Uint8Array(await res.arrayBuffer());
45
+ const result = { audio, mimeType };
46
+ if (format === "pcm") {
47
+ result.pcm = { sampleRateHz: 24000, sampleFormat: "s16le", channels: 1 };
48
+ }
49
+ return success(result);
50
+ }
51
+ }
@@ -0,0 +1,13 @@
1
+ import OpenAI from "openai";
2
+ import { OpenAISpeechClient } from "./openai.js";
3
+ /**
4
+ * Generic OpenAI-compatible speech (TTS) client. Point it at any provider
5
+ * exposing an OpenAI-shaped /audio/speech endpoint via
6
+ * `config.baseUrl.openAiCompat` (or OPENAI_COMPAT_BASE_URL) and
7
+ * `config.apiKey.openAiCompat` (or OPENAI_COMPAT_API_KEY). Mirrors the chat
8
+ * `SmolOpenAiCompat` client. Inherits OpenAI's `mp3` default format.
9
+ */
10
+ export declare class OpenAiCompatSpeechClient extends OpenAISpeechClient {
11
+ protected makeClient(): OpenAI;
12
+ protected noKeyMessage(): string;
13
+ }