smoltalk 0.8.4 → 0.10.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (75) hide show
  1. package/README.md +145 -18
  2. package/dist/classes/ToolCall.js +18 -10
  3. package/dist/classes/message/AssistantMessage.d.ts +2 -0
  4. package/dist/classes/message/ToolMessage.js +13 -10
  5. package/dist/classes/message/UserMessage.d.ts +21 -0
  6. package/dist/classes/message/UserMessage.js +3 -0
  7. package/dist/classes/message/contentParts.d.ts +71 -2
  8. package/dist/classes/message/contentParts.js +6 -0
  9. package/dist/classes/message/index.d.ts +5 -2
  10. package/dist/classes/message/index.js +7 -0
  11. package/dist/classes/message/renderers/AnthropicRenderer.d.ts +2 -1
  12. package/dist/classes/message/renderers/AnthropicRenderer.js +3 -0
  13. package/dist/classes/message/renderers/GoogleRenderer.d.ts +2 -1
  14. package/dist/classes/message/renderers/GoogleRenderer.js +3 -0
  15. package/dist/classes/message/renderers/JSONRenderer.d.ts +2 -1
  16. package/dist/classes/message/renderers/JSONRenderer.js +4 -0
  17. package/dist/classes/message/renderers/OpenAIChatRenderer.d.ts +8 -1
  18. package/dist/classes/message/renderers/OpenAIChatRenderer.js +18 -0
  19. package/dist/classes/message/renderers/OpenAIResponsesRenderer.d.ts +2 -1
  20. package/dist/classes/message/renderers/OpenAIResponsesRenderer.js +3 -0
  21. package/dist/classes/message/renderers/PartRenderer.d.ts +3 -2
  22. package/dist/classes/message/renderers/PartRenderer.js +3 -0
  23. package/dist/client.js +1 -0
  24. package/dist/clients/anthropic.js +1 -1
  25. package/dist/clients/baseClient.d.ts +13 -1
  26. package/dist/clients/baseClient.js +36 -7
  27. package/dist/clients/google.d.ts +2 -0
  28. package/dist/clients/google.js +125 -3
  29. package/dist/clients/ollama.js +1 -1
  30. package/dist/clients/openai.d.ts +2 -1
  31. package/dist/clients/openai.js +15 -3
  32. package/dist/clients/openaiCompat.d.ts +2 -0
  33. package/dist/clients/openaiCompat.js +5 -0
  34. package/dist/clients/openaiResponses.js +1 -1
  35. package/dist/clients/resolveAttachments.d.ts +8 -4
  36. package/dist/clients/resolveAttachments.js +101 -50
  37. package/dist/embed.d.ts +4 -0
  38. package/dist/files.d.ts +1 -1
  39. package/dist/files.js +1 -1
  40. package/dist/image/google.js +2 -2
  41. package/dist/image/openai.js +3 -3
  42. package/dist/image.d.ts +1 -1
  43. package/dist/index.d.ts +10 -2
  44. package/dist/index.js +7 -1
  45. package/dist/model.d.ts +15 -4
  46. package/dist/model.js +48 -7
  47. package/dist/models.d.ts +143 -19
  48. package/dist/models.js +137 -30
  49. package/dist/speech/baseSpeechClient.d.ts +31 -0
  50. package/dist/speech/baseSpeechClient.js +98 -0
  51. package/dist/speech/openai.d.ts +6 -0
  52. package/dist/speech/openai.js +39 -0
  53. package/dist/speech.d.ts +40 -0
  54. package/dist/speech.js +57 -0
  55. package/dist/transcription/baseTranscriptionClient.d.ts +31 -0
  56. package/dist/transcription/baseTranscriptionClient.js +107 -0
  57. package/dist/transcription/openai.d.ts +6 -0
  58. package/dist/transcription/openai.js +59 -0
  59. package/dist/transcription.d.ts +51 -0
  60. package/dist/transcription.js +58 -0
  61. package/dist/types/tokenUsage.d.ts +4 -0
  62. package/dist/types/tokenUsage.js +4 -0
  63. package/dist/types.d.ts +3 -0
  64. package/dist/util/attachments.d.ts +1 -1
  65. package/dist/util/audioMime.d.ts +9 -0
  66. package/dist/util/audioMime.js +33 -0
  67. package/dist/util/{imageRef.d.ts → blobRef.d.ts} +9 -9
  68. package/dist/util/{imageRef.js → blobRef.js} +6 -13
  69. package/dist/util/mime.d.ts +21 -0
  70. package/dist/util/mime.js +52 -0
  71. package/dist/util/modalities.d.ts +6 -2
  72. package/dist/util/modalities.js +13 -15
  73. package/dist/util/provider.d.ts +2 -0
  74. package/dist/util/provider.js +1 -1
  75. package/package.json +1 -1
package/dist/models.js CHANGED
@@ -17,20 +17,35 @@ export const ProviderSchema = z.enum(providers);
17
17
  export const speechToTextModels = [
18
18
  {
19
19
  type: "speech-to-text",
20
- modelName: "whisper-web",
20
+ modelName: "whisper-1",
21
21
  perMinuteCost: 0.006,
22
22
  provider: "openai",
23
+ supportedMimeTypes: [
24
+ "audio/flac", "audio/mpeg", "audio/mp4", "audio/m4a", "audio/ogg",
25
+ "audio/wav", "audio/webm",
26
+ ],
27
+ maxBytes: 25 * 1024 * 1024,
28
+ },
29
+ ];
30
+ export const textToSpeechModels = [
31
+ {
32
+ type: "text-to-speech",
33
+ modelName: "tts-1",
34
+ perCharacterCost: 0.000015,
35
+ provider: "openai",
36
+ maxInputChars: 4096,
37
+ speedRange: { min: 0.25, max: 4 },
38
+ formats: ["mp3", "opus", "aac", "flac", "wav", "pcm"],
39
+ },
40
+ {
41
+ type: "text-to-speech",
42
+ modelName: "tts-1-hd",
43
+ perCharacterCost: 0.00003,
44
+ provider: "openai",
45
+ maxInputChars: 4096,
46
+ speedRange: { min: 0.25, max: 4 },
47
+ formats: ["mp3", "opus", "aac", "flac", "wav", "pcm"],
23
48
  },
24
- // not a speech to text model?
25
- /* {
26
- type: "speech-to-text",
27
- modelName: "gpt-4o-audio-preview",
28
- description:
29
- "This is a preview release of the GPT-4o Audio models. These models accept audio inputs and outputs, and can be used in the Chat Completions REST API. Learn more. The knowledge cutoff for GPT-4o Audio models is October, 2023.",
30
- inputTokenCost: 2.5,
31
- outputTokenCost: 10,
32
- provider: "openai",
33
- }, */
34
49
  ];
35
50
  export const textModels = [
36
51
  {
@@ -771,16 +786,16 @@ export const textModels = [
771
786
  {
772
787
  type: "text",
773
788
  modelName: "gpt-5.6-terra",
774
- description: "GPT-5.6 Terra balances capability and cost — competitive with GPT-5.5 at roughly half the price. 1M context window. Standard pricing for ≤272K tokens, 2x input/1.5x output for >272K. Knowledge cutoff: February 2026.",
789
+ description: "GPT-5.6 Terra balances capability and cost. 1M context window. Standard pricing for ≤272K tokens, 2x input/1.5x output for >272K. Reflects the July 30, 2026 price cut. Knowledge cutoff: February 2026.",
775
790
  maxInputTokens: 1050000,
776
791
  maxOutputTokens: 128000,
777
- inputTokenCost: 2.5,
778
- cachedInputTokenCost: 0.25,
779
- outputTokenCost: 15,
792
+ inputTokenCost: 2,
793
+ cachedInputTokenCost: 0.2,
794
+ outputTokenCost: 12,
780
795
  longContext: {
781
- inputTokenCost: 5,
782
- cachedInputTokenCost: 0.5,
783
- outputTokenCost: 22.5,
796
+ inputTokenCost: 4,
797
+ cachedInputTokenCost: 0.4,
798
+ outputTokenCost: 18,
784
799
  thresholdTokens: 200000,
785
800
  },
786
801
  reasoning: {
@@ -806,16 +821,16 @@ export const textModels = [
806
821
  {
807
822
  type: "text",
808
823
  modelName: "gpt-5.6-luna",
809
- description: "GPT-5.6 Luna is the fast, most affordable member of the GPT-5.6 family. 1M context window. Standard pricing for ≤272K tokens, 2x input/1.5x output for >272K. Knowledge cutoff: February 2026.",
824
+ description: "GPT-5.6 Luna is the fast, most affordable member of the GPT-5.6 family. 1M context window. Standard pricing for ≤272K tokens, 2x input/1.5x output for >272K. Reflects the July 30, 2026 price cut (input/output down ~80%). Knowledge cutoff: February 2026.",
810
825
  maxInputTokens: 1050000,
811
826
  maxOutputTokens: 128000,
812
- inputTokenCost: 1,
813
- cachedInputTokenCost: 0.1,
814
- outputTokenCost: 6,
827
+ inputTokenCost: 0.2,
828
+ cachedInputTokenCost: 0.02,
829
+ outputTokenCost: 1.2,
815
830
  longContext: {
816
- inputTokenCost: 2,
817
- cachedInputTokenCost: 0.2,
818
- outputTokenCost: 9,
831
+ inputTokenCost: 0.4,
832
+ cachedInputTokenCost: 0.04,
833
+ outputTokenCost: 1.8,
819
834
  thresholdTokens: 200000,
820
835
  },
821
836
  reasoning: {
@@ -903,10 +918,40 @@ export const textModels = [
903
918
  disabled: true,
904
919
  provider: "google",
905
920
  },
921
+ {
922
+ type: "text",
923
+ modelName: "gemini-3.6-flash",
924
+ description: "Latest Gemini 3.6 Flash model (GA July 21, 2026). Supersedes Gemini 3.5 Flash with better token efficiency and agentic planning at a lower output price ($7.50 vs $9.00/1M). 1M context window, 64K output. Context caching: $0.15/1M read.",
925
+ maxInputTokens: 1048576,
926
+ maxOutputTokens: 65536,
927
+ inputTokenCost: 1.5,
928
+ cachedInputTokenCost: 0.15,
929
+ outputTokenCost: 7.5,
930
+ inputAudioTokenCost: 1.5,
931
+ reasoning: {
932
+ levels: ["minimal", "low", "medium", "high"],
933
+ defaultLevel: "high",
934
+ canDisable: false,
935
+ outputsThinking: true,
936
+ outputsSignatures: true,
937
+ },
938
+ modalities: {
939
+ input: ["text", "image", "video", "audio", "pdf"],
940
+ output: ["text"],
941
+ },
942
+ knowledge: "2026-03",
943
+ releaseDate: "2026-07-21",
944
+ lastUpdated: "2026-07-21",
945
+ family: "gemini-flash",
946
+ openWeights: false,
947
+ structuredOutput: true,
948
+ temperatureSupported: true,
949
+ provider: "google",
950
+ },
906
951
  {
907
952
  type: "text",
908
953
  modelName: "gemini-3.5-flash",
909
- description: "Latest Gemini 3.5 Flash model (GA May 2026). Outperforms Gemini 3.1 Pro on coding and agentic suites at 4x the speed. 1M context window, 64K output. Context caching: $0.15/1M read.",
954
+ description: "Gemini 3.5 Flash (GA May 2026). Superseded by gemini-3.6-flash. Outperforms Gemini 3.1 Pro on coding and agentic suites at 4x the speed. 1M context window, 64K output. Context caching: $0.15/1M read.",
910
955
  maxInputTokens: 1048576,
911
956
  maxOutputTokens: 65536,
912
957
  inputTokenCost: 1.5,
@@ -964,10 +1009,40 @@ export const textModels = [
964
1009
  temperatureSupported: true,
965
1010
  provider: "google",
966
1011
  },
1012
+ {
1013
+ type: "text",
1014
+ modelName: "gemini-3.5-flash-lite",
1015
+ description: "Most cost-effective Gemini model (GA July 21, 2026). Supersedes gemini-3.1-flash-lite. Built for high-volume, low-reasoning work (search, document processing, translation). Thinking support, 1M context window, 64K output.",
1016
+ maxInputTokens: 1048576,
1017
+ maxOutputTokens: 65536,
1018
+ inputTokenCost: 0.3,
1019
+ cachedInputTokenCost: 0.03,
1020
+ outputTokenCost: 2.5,
1021
+ inputAudioTokenCost: 0.5,
1022
+ reasoning: {
1023
+ levels: ["minimal", "low", "medium", "high"],
1024
+ defaultLevel: "minimal",
1025
+ canDisable: false,
1026
+ outputsThinking: true,
1027
+ outputsSignatures: true,
1028
+ },
1029
+ modalities: {
1030
+ input: ["text", "image", "video", "audio", "pdf"],
1031
+ output: ["text"],
1032
+ },
1033
+ knowledge: "2025-01",
1034
+ releaseDate: "2026-07-21",
1035
+ lastUpdated: "2026-07-21",
1036
+ family: "gemini-flash-lite",
1037
+ openWeights: false,
1038
+ structuredOutput: true,
1039
+ temperatureSupported: true,
1040
+ provider: "google",
1041
+ },
967
1042
  {
968
1043
  type: "text",
969
1044
  modelName: "gemini-3.1-flash-lite",
970
- description: "Most cost-effective Gemini 3.1 model (GA). Thinking support, 1M context window, 64K output. 2.5x faster TTFA and 45% faster output than 2.5 Flash.",
1045
+ description: "Cost-effective Gemini 3.1 model (GA). Superseded by gemini-3.5-flash-lite. Thinking support, 1M context window, 64K output. 2.5x faster TTFA and 45% faster output than 2.5 Flash.",
971
1046
  maxInputTokens: 1048576,
972
1047
  maxOutputTokens: 65536,
973
1048
  inputTokenCost: 0.25,
@@ -1549,6 +1624,19 @@ export const textModels = [
1549
1624
  temperatureSupported: false,
1550
1625
  provider: "openai-responses",
1551
1626
  },
1627
+ {
1628
+ type: "text",
1629
+ modelName: "gpt-audio-1.5",
1630
+ description: "OpenAI GA audio chat model (Chat Completions). Text+audio in, text+audio out.",
1631
+ provider: "openai",
1632
+ modalities: { input: ["text", "audio"], output: ["text", "audio"] },
1633
+ inputTokenCost: 2.5,
1634
+ outputTokenCost: 10,
1635
+ inputAudioTokenCost: 32,
1636
+ outputAudioTokenCost: 64,
1637
+ maxInputTokens: 128000,
1638
+ maxOutputTokens: 16384,
1639
+ },
1552
1640
  ];
1553
1641
  export const imageModels = [
1554
1642
  {
@@ -1730,7 +1818,7 @@ export const hostedTools = [
1730
1818
  category: "maps_grounding",
1731
1819
  description: "Grounding with Google Maps (Gemini 3 only).",
1732
1820
  providerToolId: "google_maps",
1733
- models: ["gemini-3-pro-preview", "gemini-3.1-pro-preview", "gemini-3-flash-preview", "gemini-3.5-flash", "gemini-3.1-flash-lite"],
1821
+ models: ["gemini-3-pro-preview", "gemini-3.1-pro-preview", "gemini-3-flash-preview", "gemini-3.5-flash", "gemini-3.6-flash", "gemini-3.1-flash-lite", "gemini-3.5-flash-lite"],
1734
1822
  pricing: { unit: "per_call", note: "Gemini 3 family only; see Google pricing." },
1735
1823
  },
1736
1824
  {
@@ -1765,6 +1853,7 @@ function baselineModels() {
1765
1853
  ...textModels,
1766
1854
  ...imageModels,
1767
1855
  ...speechToTextModels,
1856
+ ...textToSpeechModels,
1768
1857
  ...registeredTextModels,
1769
1858
  ...embeddingsModels,
1770
1859
  ];
@@ -1790,13 +1879,28 @@ export function getAllModels(requestData) {
1790
1879
  export function getModel(modelName, requestData) {
1791
1880
  return getAllModels(requestData).find((model) => model.modelName === modelName);
1792
1881
  }
1882
+ /**
1883
+ * Like `getModel`, but also matches on `provider`. Use this whenever a
1884
+ * modelName may collide across providers (the merge key everywhere else in
1885
+ * this module is `provider:modelName`) — plain `getModel` returns whichever
1886
+ * matching entry comes first and can silently pick the wrong provider.
1887
+ */
1888
+ export function getModelForProvider(provider, modelName, requestData) {
1889
+ return getAllModels(requestData).find((model) => model.modelName === modelName && model.provider === provider);
1890
+ }
1793
1891
  /**
1794
1892
  * Whether a model is known to accept the given input modality ("image", "pdf", …).
1795
1893
  * Returns undefined when the model is unknown or carries no `modalities` data —
1796
1894
  * callers should treat undefined as "don't gate".
1797
1895
  */
1798
- export function modelSupportsInputModality(modelName, modality, requestData) {
1799
- const model = getModel(modelName, requestData);
1896
+ export function modelSupportsInputModality(modelName, modality, requestData, provider) {
1897
+ let model;
1898
+ if (provider !== undefined) {
1899
+ model = getModelForProvider(provider, modelName, requestData);
1900
+ }
1901
+ else {
1902
+ model = getModel(modelName, requestData);
1903
+ }
1800
1904
  if (!model || model.type !== "text") {
1801
1905
  return undefined;
1802
1906
  }
@@ -1868,6 +1972,9 @@ export function isTextModel(model) {
1868
1972
  export function isSpeechToTextModel(model) {
1869
1973
  return model.type === "speech-to-text";
1870
1974
  }
1975
+ export function isTextToSpeechModel(model) {
1976
+ return model.type === "text-to-speech";
1977
+ }
1871
1978
  export function isEmbeddingsModel(model) {
1872
1979
  return model.type === "embeddings";
1873
1980
  }
@@ -0,0 +1,31 @@
1
+ import type { ModelDataBlob } from "../modelData.js";
2
+ import { Result } from "../types/result.js";
3
+ import type { SpeechResult } from "../speech.js";
4
+ export type SpeechClientConfig = {
5
+ model: string;
6
+ /** Resolved provider name. */
7
+ provider: string;
8
+ /** Resolved API key; empty string when none was found. */
9
+ apiKey: string;
10
+ voice: string;
11
+ modelData?: ModelDataBlob;
12
+ /** Output format; provider-specific vocabulary (OpenAI: mp3/opus/aac/flac/wav/pcm). */
13
+ format?: string;
14
+ speed?: number;
15
+ metadata?: Record<string, unknown>;
16
+ };
17
+ /**
18
+ * Shared TTS behavior, mirroring BaseClient for text generation: the public
19
+ * speak() template method owns model-data-driven validation (char cap, speed
20
+ * range, format list), cost, and the single redacting/logging exception
21
+ * boundary. Subclasses implement only _speak(): SDK call + response mapping.
22
+ * A model with no registry entry skips validation — the provider is then the
23
+ * authority, matching how cost is silently omitted for unknown models.
24
+ */
25
+ export declare abstract class BaseSpeechClient {
26
+ protected config: SpeechClientConfig;
27
+ constructor(config: SpeechClientConfig);
28
+ speak(text: string): Promise<Result<SpeechResult>>;
29
+ /** Provider hook: SDK call + response mapping only; validation and cost live in the base. */
30
+ protected abstract _speak(text: string): Promise<Result<SpeechResult>>;
31
+ }
@@ -0,0 +1,98 @@
1
+ import { getModelForProvider, isTextToSpeechModel, } from "../models.js";
2
+ import { calculateSpeechCost } from "../model.js";
3
+ import { failure } from "../types/result.js";
4
+ import { redactSecret } from "../util/redact.js";
5
+ import { getLogger } from "../util/logger.js";
6
+ /** Validate the declarative TTS constraint block once before consuming it. */
7
+ function speechConstraintError(model) {
8
+ const maxInputChars = model.maxInputChars;
9
+ if (maxInputChars !== undefined &&
10
+ (typeof maxInputChars !== "number" ||
11
+ !Number.isInteger(maxInputChars) ||
12
+ maxInputChars <= 0)) {
13
+ return `Model "${model.modelName}" has an invalid maxInputChars value.`;
14
+ }
15
+ const speedRange = model.speedRange;
16
+ if (speedRange !== undefined) {
17
+ if (typeof speedRange !== "object" || speedRange === null) {
18
+ return `Model "${model.modelName}" has an invalid speedRange.`;
19
+ }
20
+ const min = speedRange.min;
21
+ const max = speedRange.max;
22
+ if (typeof min !== "number" ||
23
+ typeof max !== "number" ||
24
+ !Number.isFinite(min) ||
25
+ !Number.isFinite(max) ||
26
+ min > max) {
27
+ return `Model "${model.modelName}" has an invalid speedRange.`;
28
+ }
29
+ }
30
+ const formats = model.formats;
31
+ if (formats !== undefined &&
32
+ (!Array.isArray(formats) ||
33
+ !formats.every((format) => typeof format === "string"))) {
34
+ return `Model "${model.modelName}" has invalid formats.`;
35
+ }
36
+ return null;
37
+ }
38
+ /**
39
+ * Shared TTS behavior, mirroring BaseClient for text generation: the public
40
+ * speak() template method owns model-data-driven validation (char cap, speed
41
+ * range, format list), cost, and the single redacting/logging exception
42
+ * boundary. Subclasses implement only _speak(): SDK call + response mapping.
43
+ * A model with no registry entry skips validation — the provider is then the
44
+ * authority, matching how cost is silently omitted for unknown models.
45
+ */
46
+ export class BaseSpeechClient {
47
+ config;
48
+ constructor(config) {
49
+ this.config = config;
50
+ }
51
+ async speak(text) {
52
+ try {
53
+ const model = getModelForProvider(this.config.provider, this.config.model, this.config.modelData);
54
+ if (model !== undefined && !isTextToSpeechModel(model)) {
55
+ return failure(`Model "${this.config.model}" is not a text-to-speech model.`);
56
+ }
57
+ if (model !== undefined) {
58
+ const constraintError = speechConstraintError(model);
59
+ if (constraintError !== null) {
60
+ return failure(constraintError);
61
+ }
62
+ if (model.maxInputChars !== undefined && [...text].length > model.maxInputChars) {
63
+ return failure(`Input exceeds the ${model.maxInputChars}-character limit for model "${this.config.model}".`);
64
+ }
65
+ if (this.config.speed !== undefined && model.speedRange !== undefined) {
66
+ const { min, max } = model.speedRange;
67
+ if (!Number.isFinite(this.config.speed) || this.config.speed < min || this.config.speed > max) {
68
+ return failure(`speed must be a finite number in [${min}, ${max}].`);
69
+ }
70
+ }
71
+ if (this.config.format !== undefined &&
72
+ model.formats !== undefined &&
73
+ !model.formats.includes(this.config.format)) {
74
+ return failure(`Format "${this.config.format}" is not supported by model "${this.config.model}". ` +
75
+ `Supported: ${model.formats.join(", ")}.`);
76
+ }
77
+ }
78
+ const result = await this._speak(text);
79
+ if (!result.success) {
80
+ return result;
81
+ }
82
+ const cost = calculateSpeechCost(model, [...text].length);
83
+ if (cost !== undefined) {
84
+ result.value.cost = cost;
85
+ }
86
+ return result;
87
+ }
88
+ catch (err) {
89
+ let msg = "speak() failed";
90
+ if (err instanceof Error) {
91
+ msg = err.message;
92
+ }
93
+ const redacted = redactSecret(msg, this.config.apiKey);
94
+ getLogger().error("speak() provider failed:", redacted);
95
+ return failure(redacted);
96
+ }
97
+ }
98
+ }
@@ -0,0 +1,6 @@
1
+ import { Result } from "../types/result.js";
2
+ import { BaseSpeechClient } from "./baseSpeechClient.js";
3
+ import type { SpeechResult } from "../speech.js";
4
+ export declare class OpenAISpeechClient extends BaseSpeechClient {
5
+ protected _speak(text: string): Promise<Result<SpeechResult>>;
6
+ }
@@ -0,0 +1,39 @@
1
+ import OpenAI from "openai";
2
+ import { success, failure } from "../types/result.js";
3
+ import { SPEECH_FORMAT_TO_MIME, isSpeakFormat, } from "../util/audioMime.js";
4
+ import { BaseSpeechClient } from "./baseSpeechClient.js";
5
+ export class OpenAISpeechClient extends BaseSpeechClient {
6
+ // No try/catch here: BaseSpeechClient.speak() is the single
7
+ // redacting/logging exception boundary.
8
+ async _speak(text) {
9
+ if (!this.config.apiKey) {
10
+ return failure("No OpenAI API key provided. Set apiKey.openAi or OPENAI_API_KEY.");
11
+ }
12
+ // The shared contract carries format as a plain string; narrow to OpenAI's
13
+ // closed union at runtime before indexing the MIME table.
14
+ const requestedFormat = this.config.format ?? "mp3";
15
+ if (!isSpeakFormat(requestedFormat)) {
16
+ return failure(`Format "${requestedFormat}" is not a supported OpenAI speech format. ` +
17
+ `Supported: ${Object.keys(SPEECH_FORMAT_TO_MIME).join(", ")}.`);
18
+ }
19
+ const format = requestedFormat;
20
+ const mimeType = SPEECH_FORMAT_TO_MIME[format];
21
+ const client = new OpenAI({ apiKey: this.config.apiKey });
22
+ const params = {
23
+ model: this.config.model,
24
+ voice: this.config.voice,
25
+ input: text,
26
+ response_format: format,
27
+ };
28
+ if (this.config.speed !== undefined) {
29
+ params.speed = this.config.speed;
30
+ }
31
+ const res = await client.audio.speech.create(params);
32
+ const audio = new Uint8Array(await res.arrayBuffer());
33
+ const result = { audio, mimeType };
34
+ if (format === "pcm") {
35
+ result.pcm = { sampleRateHz: 24000, sampleFormat: "s16le", channels: 1 };
36
+ }
37
+ return success(result);
38
+ }
39
+ }
@@ -0,0 +1,40 @@
1
+ import type { ModelDataBlob } from "./modelData.js";
2
+ import type { SmolConfig } from "./types.js";
3
+ import { Result } from "./types/result.js";
4
+ import { CostEstimate } from "./types/costEstimate.js";
5
+ import { BaseSpeechClient, SpeechClientConfig } from "./speech/baseSpeechClient.js";
6
+ export type SpeakOptions = {
7
+ model: string;
8
+ voice: string;
9
+ provider?: string;
10
+ modelData?: ModelDataBlob;
11
+ apiKey?: SmolConfig["apiKey"];
12
+ format?: string;
13
+ speed?: number;
14
+ metadata?: Record<string, unknown>;
15
+ };
16
+ export type PcmAudioMetadata = {
17
+ sampleRateHz: number;
18
+ /** Sample representation, e.g. "s16le"; provider-specific. */
19
+ sampleFormat: string;
20
+ channels: number;
21
+ };
22
+ export type SpeechResult = {
23
+ audio: Uint8Array;
24
+ mimeType: string;
25
+ pcm?: PcmAudioMetadata;
26
+ cost?: CostEstimate;
27
+ raw?: unknown;
28
+ };
29
+ export type SpeechClientClass = new (config: SpeechClientConfig) => BaseSpeechClient;
30
+ export declare function registerSpeechProvider(name: string, cls: SpeechClientClass): void;
31
+ /** Test-only: clear all registered custom providers so registrations don't leak across tests. */
32
+ export declare function _resetForTests(): void;
33
+ /**
34
+ * Resolve provider + API key and instantiate the matching speech client for
35
+ * the declarative speak() operation. Never throws: a custom client class's
36
+ * constructor can throw, and this internal factory's catch redacts the
37
+ * resolved key so a constructor error cannot leak through the public wrapper.
38
+ */
39
+ export declare function getSpeechClient(opts: SpeakOptions): Result<BaseSpeechClient>;
40
+ export declare function speak(text: string, opts: SpeakOptions): Promise<Result<SpeechResult>>;
package/dist/speech.js ADDED
@@ -0,0 +1,57 @@
1
+ import { success, failure } from "./types/result.js";
2
+ import { redactSecret } from "./util/redact.js";
3
+ import { getLogger } from "./util/logger.js";
4
+ import { resolveProvider, resolveApiKey } from "./util/provider.js";
5
+ import { OpenAISpeechClient } from "./speech/openai.js";
6
+ // Checked before the user registry so a registered "openai" can't hijack the built-in.
7
+ const builtinClients = Object.create(null);
8
+ builtinClients["openai"] = OpenAISpeechClient;
9
+ // Null-prototype so provider names like "toString"/"__proto__" can't collide
10
+ // with Object.prototype or pollute the registry.
11
+ const registered = Object.create(null);
12
+ export function registerSpeechProvider(name, cls) {
13
+ registered[name] = cls;
14
+ }
15
+ /** Test-only: clear all registered custom providers so registrations don't leak across tests. */
16
+ export function _resetForTests() {
17
+ for (const key of Object.keys(registered)) {
18
+ delete registered[key];
19
+ }
20
+ }
21
+ /**
22
+ * Resolve provider + API key and instantiate the matching speech client for
23
+ * the declarative speak() operation. Never throws: a custom client class's
24
+ * constructor can throw, and this internal factory's catch redacts the
25
+ * resolved key so a constructor error cannot leak through the public wrapper.
26
+ */
27
+ export function getSpeechClient(opts) {
28
+ let apiKeyForRedaction = "";
29
+ try {
30
+ const provider = resolveProvider(opts.model, opts.provider, opts.modelData);
31
+ const ClientClass = builtinClients[provider] ?? registered[provider];
32
+ if (ClientClass === undefined) {
33
+ return failure(`Provider "${provider}" has no speech API. Register one with registerSpeechProvider(name, ClientClass).`);
34
+ }
35
+ const apiKey = resolveApiKey(provider, opts) ?? "";
36
+ apiKeyForRedaction = apiKey;
37
+ const { apiKey: _callerKeys, ...clientOpts } = opts;
38
+ const config = { ...clientOpts, provider, apiKey };
39
+ return success(new ClientClass(config));
40
+ }
41
+ catch (err) {
42
+ let msg = "getSpeechClient() failed";
43
+ if (err instanceof Error) {
44
+ msg = err.message;
45
+ }
46
+ const redacted = redactSecret(msg, apiKeyForRedaction);
47
+ getLogger().error("getSpeechClient() failed:", redacted);
48
+ return failure(redacted);
49
+ }
50
+ }
51
+ export async function speak(text, opts) {
52
+ const client = getSpeechClient(opts);
53
+ if (!client.success) {
54
+ return client;
55
+ }
56
+ return client.value.speak(text);
57
+ }
@@ -0,0 +1,31 @@
1
+ import type { ModelDataBlob } from "../modelData.js";
2
+ import { Result } from "../types/result.js";
3
+ import { BlobRef } from "../util/blobRef.js";
4
+ import type { TranscriptionResult } from "../transcription.js";
5
+ export declare const DEFAULT_TRANSCRIBE_BYTES: number;
6
+ export type TranscriptionClientConfig = {
7
+ model: string;
8
+ /** Resolved provider name. */
9
+ provider: string;
10
+ /** Resolved API key; empty string when none was found. */
11
+ apiKey: string;
12
+ modelData?: ModelDataBlob;
13
+ language?: string;
14
+ prompt?: string;
15
+ timestampGranularity?: "segment" | "word";
16
+ maxBytes?: number;
17
+ metadata?: Record<string, unknown>;
18
+ };
19
+ /**
20
+ * Shared transcription behavior, mirroring BaseClient for text generation:
21
+ * the public transcribe() template method owns blob loading, model-data-driven
22
+ * validation, cost, and the single redacting/logging exception boundary.
23
+ * Subclasses implement only _transcribe(): SDK call + response mapping.
24
+ */
25
+ export declare abstract class BaseTranscriptionClient {
26
+ protected config: TranscriptionClientConfig;
27
+ constructor(config: TranscriptionClientConfig);
28
+ transcribe(source: BlobRef): Promise<Result<TranscriptionResult>>;
29
+ /** Provider hook: SDK call + response mapping only; validation and cost live in the base. */
30
+ protected abstract _transcribe(data: Uint8Array, mimeType: string): Promise<Result<TranscriptionResult>>;
31
+ }
@@ -0,0 +1,107 @@
1
+ import { getModelForProvider, isSpeechToTextModel, } from "../models.js";
2
+ import { calculateTranscriptionCost } from "../model.js";
3
+ import { success, failure } from "../types/result.js";
4
+ import { loadBlob } from "../util/blobRef.js";
5
+ import { audioFormatForMime, canonicalizeMime } from "../util/mime.js";
6
+ import { redactSecret } from "../util/redact.js";
7
+ import { getLogger } from "../util/logger.js";
8
+ export const DEFAULT_TRANSCRIBE_BYTES = 25 * 1024 * 1024;
9
+ /** Validate the declarative STT constraint block once before consuming it. */
10
+ function transcriptionConstraintError(model) {
11
+ const modelMaxBytes = model.maxBytes;
12
+ if (modelMaxBytes !== undefined &&
13
+ (typeof modelMaxBytes !== "number" || !Number.isFinite(modelMaxBytes) || modelMaxBytes <= 0)) {
14
+ return `Model "${model.modelName}" has an invalid maxBytes value.`;
15
+ }
16
+ const supportedMimeTypes = model.supportedMimeTypes;
17
+ if (supportedMimeTypes !== undefined &&
18
+ (!Array.isArray(supportedMimeTypes) ||
19
+ !supportedMimeTypes.every((mime) => typeof mime === "string"))) {
20
+ return `Model "${model.modelName}" has invalid supportedMimeTypes.`;
21
+ }
22
+ return null;
23
+ }
24
+ // The caller's maxBytes is a safety limit; the model's maxBytes is the
25
+ // provider's hard cap. Take the smaller of whichever are present so a caller
26
+ // can tighten the limit but never bypass the provider cap.
27
+ function resolveTranscriptionMaxBytes(callerMaxBytes, model) {
28
+ if (callerMaxBytes !== undefined &&
29
+ (!Number.isFinite(callerMaxBytes) || callerMaxBytes <= 0)) {
30
+ return failure(`maxBytes must be a positive finite number (got ${callerMaxBytes}).`);
31
+ }
32
+ const limits = [];
33
+ if (callerMaxBytes !== undefined) {
34
+ limits.push(callerMaxBytes);
35
+ }
36
+ if (model?.maxBytes !== undefined) {
37
+ limits.push(model.maxBytes);
38
+ }
39
+ if (limits.length === 0) {
40
+ return success(DEFAULT_TRANSCRIBE_BYTES);
41
+ }
42
+ return success(Math.min(...limits));
43
+ }
44
+ /**
45
+ * Shared transcription behavior, mirroring BaseClient for text generation:
46
+ * the public transcribe() template method owns blob loading, model-data-driven
47
+ * validation, cost, and the single redacting/logging exception boundary.
48
+ * Subclasses implement only _transcribe(): SDK call + response mapping.
49
+ */
50
+ export class BaseTranscriptionClient {
51
+ config;
52
+ constructor(config) {
53
+ this.config = config;
54
+ }
55
+ async transcribe(source) {
56
+ try {
57
+ const model = getModelForProvider(this.config.provider, this.config.model, this.config.modelData);
58
+ if (model !== undefined && !isSpeechToTextModel(model)) {
59
+ return failure(`Model "${this.config.model}" is not a speech-to-text model.`);
60
+ }
61
+ if (model !== undefined) {
62
+ const constraintError = transcriptionConstraintError(model);
63
+ if (constraintError !== null) {
64
+ return failure(constraintError);
65
+ }
66
+ }
67
+ const effectiveLimit = resolveTranscriptionMaxBytes(this.config.maxBytes, model);
68
+ if (!effectiveLimit.success) {
69
+ return effectiveLimit;
70
+ }
71
+ let loaded;
72
+ try {
73
+ loaded = await loadBlob(source, { maxBytes: effectiveLimit.value });
74
+ }
75
+ catch (err) {
76
+ return failure(`Failed to load audio for transcription: ${err.message}`);
77
+ }
78
+ const mimeType = loaded.mimeType ?? "application/octet-stream";
79
+ if (model !== undefined && model.supportedMimeTypes !== undefined) {
80
+ const audioFormat = audioFormatForMime(mimeType);
81
+ const normalizedMime = audioFormat?.mimeType ?? canonicalizeMime(mimeType);
82
+ if (!model.supportedMimeTypes.includes(normalizedMime)) {
83
+ return failure(`Unsupported audio type "${mimeType}" for model "${this.config.model}". ` +
84
+ `Supported: ${model.supportedMimeTypes.join(", ")}.`);
85
+ }
86
+ }
87
+ const result = await this._transcribe(loaded.data, mimeType);
88
+ if (!result.success) {
89
+ return result;
90
+ }
91
+ const cost = calculateTranscriptionCost(model, result.value.durationSeconds);
92
+ if (cost !== undefined) {
93
+ result.value.cost = cost;
94
+ }
95
+ return result;
96
+ }
97
+ catch (err) {
98
+ let msg = "transcribe() failed";
99
+ if (err instanceof Error) {
100
+ msg = err.message;
101
+ }
102
+ const redacted = redactSecret(msg, this.config.apiKey);
103
+ getLogger().error("transcribe() provider failed:", redacted);
104
+ return failure(redacted);
105
+ }
106
+ }
107
+ }