smoltalk 0.9.0 → 0.10.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (72) hide show
  1. package/README.md +141 -7
  2. package/dist/classes/message/AssistantMessage.d.ts +2 -0
  3. package/dist/classes/message/UserMessage.d.ts +21 -0
  4. package/dist/classes/message/UserMessage.js +3 -0
  5. package/dist/classes/message/contentParts.d.ts +71 -2
  6. package/dist/classes/message/contentParts.js +6 -0
  7. package/dist/classes/message/index.d.ts +5 -2
  8. package/dist/classes/message/index.js +7 -0
  9. package/dist/classes/message/renderers/AnthropicRenderer.d.ts +2 -1
  10. package/dist/classes/message/renderers/AnthropicRenderer.js +3 -0
  11. package/dist/classes/message/renderers/GoogleRenderer.d.ts +2 -1
  12. package/dist/classes/message/renderers/GoogleRenderer.js +3 -0
  13. package/dist/classes/message/renderers/JSONRenderer.d.ts +2 -1
  14. package/dist/classes/message/renderers/JSONRenderer.js +4 -0
  15. package/dist/classes/message/renderers/OpenAIChatRenderer.d.ts +8 -1
  16. package/dist/classes/message/renderers/OpenAIChatRenderer.js +18 -0
  17. package/dist/classes/message/renderers/OpenAIResponsesRenderer.d.ts +2 -1
  18. package/dist/classes/message/renderers/OpenAIResponsesRenderer.js +3 -0
  19. package/dist/classes/message/renderers/PartRenderer.d.ts +3 -2
  20. package/dist/classes/message/renderers/PartRenderer.js +3 -0
  21. package/dist/client.js +1 -0
  22. package/dist/clients/anthropic.js +1 -1
  23. package/dist/clients/baseClient.d.ts +13 -1
  24. package/dist/clients/baseClient.js +36 -7
  25. package/dist/clients/google.js +1 -1
  26. package/dist/clients/ollama.js +1 -1
  27. package/dist/clients/openai.d.ts +2 -1
  28. package/dist/clients/openai.js +15 -3
  29. package/dist/clients/openaiCompat.d.ts +2 -0
  30. package/dist/clients/openaiCompat.js +5 -0
  31. package/dist/clients/openaiResponses.js +1 -1
  32. package/dist/clients/resolveAttachments.d.ts +8 -4
  33. package/dist/clients/resolveAttachments.js +101 -50
  34. package/dist/embed.d.ts +4 -0
  35. package/dist/files.d.ts +1 -1
  36. package/dist/files.js +1 -1
  37. package/dist/image/google.js +2 -2
  38. package/dist/image/openai.js +3 -3
  39. package/dist/image.d.ts +1 -1
  40. package/dist/index.d.ts +10 -2
  41. package/dist/index.js +7 -1
  42. package/dist/model.d.ts +15 -4
  43. package/dist/model.js +48 -7
  44. package/dist/models.d.ts +143 -19
  45. package/dist/models.js +137 -30
  46. package/dist/speech/baseSpeechClient.d.ts +31 -0
  47. package/dist/speech/baseSpeechClient.js +98 -0
  48. package/dist/speech/openai.d.ts +6 -0
  49. package/dist/speech/openai.js +39 -0
  50. package/dist/speech.d.ts +40 -0
  51. package/dist/speech.js +57 -0
  52. package/dist/transcription/baseTranscriptionClient.d.ts +31 -0
  53. package/dist/transcription/baseTranscriptionClient.js +107 -0
  54. package/dist/transcription/openai.d.ts +6 -0
  55. package/dist/transcription/openai.js +59 -0
  56. package/dist/transcription.d.ts +51 -0
  57. package/dist/transcription.js +58 -0
  58. package/dist/types/tokenUsage.d.ts +4 -0
  59. package/dist/types/tokenUsage.js +4 -0
  60. package/dist/types.d.ts +3 -0
  61. package/dist/util/attachments.d.ts +1 -1
  62. package/dist/util/audioMime.d.ts +9 -0
  63. package/dist/util/audioMime.js +33 -0
  64. package/dist/util/{imageRef.d.ts → blobRef.d.ts} +9 -9
  65. package/dist/util/{imageRef.js → blobRef.js} +6 -13
  66. package/dist/util/mime.d.ts +21 -0
  67. package/dist/util/mime.js +52 -0
  68. package/dist/util/modalities.d.ts +6 -2
  69. package/dist/util/modalities.js +13 -15
  70. package/dist/util/provider.d.ts +2 -0
  71. package/dist/util/provider.js +1 -1
  72. package/package.json +1 -1
@@ -0,0 +1,39 @@
1
+ import OpenAI from "openai";
2
+ import { success, failure } from "../types/result.js";
3
+ import { SPEECH_FORMAT_TO_MIME, isSpeakFormat, } from "../util/audioMime.js";
4
+ import { BaseSpeechClient } from "./baseSpeechClient.js";
5
+ export class OpenAISpeechClient extends BaseSpeechClient {
6
+ // No try/catch here: BaseSpeechClient.speak() is the single
7
+ // redacting/logging exception boundary.
8
+ async _speak(text) {
9
+ if (!this.config.apiKey) {
10
+ return failure("No OpenAI API key provided. Set apiKey.openAi or OPENAI_API_KEY.");
11
+ }
12
+ // The shared contract carries format as a plain string; narrow to OpenAI's
13
+ // closed union at runtime before indexing the MIME table.
14
+ const requestedFormat = this.config.format ?? "mp3";
15
+ if (!isSpeakFormat(requestedFormat)) {
16
+ return failure(`Format "${requestedFormat}" is not a supported OpenAI speech format. ` +
17
+ `Supported: ${Object.keys(SPEECH_FORMAT_TO_MIME).join(", ")}.`);
18
+ }
19
+ const format = requestedFormat;
20
+ const mimeType = SPEECH_FORMAT_TO_MIME[format];
21
+ const client = new OpenAI({ apiKey: this.config.apiKey });
22
+ const params = {
23
+ model: this.config.model,
24
+ voice: this.config.voice,
25
+ input: text,
26
+ response_format: format,
27
+ };
28
+ if (this.config.speed !== undefined) {
29
+ params.speed = this.config.speed;
30
+ }
31
+ const res = await client.audio.speech.create(params);
32
+ const audio = new Uint8Array(await res.arrayBuffer());
33
+ const result = { audio, mimeType };
34
+ if (format === "pcm") {
35
+ result.pcm = { sampleRateHz: 24000, sampleFormat: "s16le", channels: 1 };
36
+ }
37
+ return success(result);
38
+ }
39
+ }
@@ -0,0 +1,40 @@
1
+ import type { ModelDataBlob } from "./modelData.js";
2
+ import type { SmolConfig } from "./types.js";
3
+ import { Result } from "./types/result.js";
4
+ import { CostEstimate } from "./types/costEstimate.js";
5
+ import { BaseSpeechClient, SpeechClientConfig } from "./speech/baseSpeechClient.js";
6
+ export type SpeakOptions = {
7
+ model: string;
8
+ voice: string;
9
+ provider?: string;
10
+ modelData?: ModelDataBlob;
11
+ apiKey?: SmolConfig["apiKey"];
12
+ format?: string;
13
+ speed?: number;
14
+ metadata?: Record<string, unknown>;
15
+ };
16
+ export type PcmAudioMetadata = {
17
+ sampleRateHz: number;
18
+ /** Sample representation, e.g. "s16le"; provider-specific. */
19
+ sampleFormat: string;
20
+ channels: number;
21
+ };
22
+ export type SpeechResult = {
23
+ audio: Uint8Array;
24
+ mimeType: string;
25
+ pcm?: PcmAudioMetadata;
26
+ cost?: CostEstimate;
27
+ raw?: unknown;
28
+ };
29
+ export type SpeechClientClass = new (config: SpeechClientConfig) => BaseSpeechClient;
30
+ export declare function registerSpeechProvider(name: string, cls: SpeechClientClass): void;
31
+ /** Test-only: clear all registered custom providers so registrations don't leak across tests. */
32
+ export declare function _resetForTests(): void;
33
+ /**
34
+ * Resolve provider + API key and instantiate the matching speech client for
35
+ * the declarative speak() operation. Never throws: a custom client class's
36
+ * constructor can throw, and this internal factory's catch redacts the
37
+ * resolved key so a constructor error cannot leak through the public wrapper.
38
+ */
39
+ export declare function getSpeechClient(opts: SpeakOptions): Result<BaseSpeechClient>;
40
+ export declare function speak(text: string, opts: SpeakOptions): Promise<Result<SpeechResult>>;
package/dist/speech.js ADDED
@@ -0,0 +1,57 @@
1
+ import { success, failure } from "./types/result.js";
2
+ import { redactSecret } from "./util/redact.js";
3
+ import { getLogger } from "./util/logger.js";
4
+ import { resolveProvider, resolveApiKey } from "./util/provider.js";
5
+ import { OpenAISpeechClient } from "./speech/openai.js";
6
+ // Checked before the user registry so a registered "openai" can't hijack the built-in.
7
+ const builtinClients = Object.create(null);
8
+ builtinClients["openai"] = OpenAISpeechClient;
9
+ // Null-prototype so provider names like "toString"/"__proto__" can't collide
10
+ // with Object.prototype or pollute the registry.
11
+ const registered = Object.create(null);
12
+ export function registerSpeechProvider(name, cls) {
13
+ registered[name] = cls;
14
+ }
15
+ /** Test-only: clear all registered custom providers so registrations don't leak across tests. */
16
+ export function _resetForTests() {
17
+ for (const key of Object.keys(registered)) {
18
+ delete registered[key];
19
+ }
20
+ }
21
+ /**
22
+ * Resolve provider + API key and instantiate the matching speech client for
23
+ * the declarative speak() operation. Never throws: a custom client class's
24
+ * constructor can throw, and this internal factory's catch redacts the
25
+ * resolved key so a constructor error cannot leak through the public wrapper.
26
+ */
27
+ export function getSpeechClient(opts) {
28
+ let apiKeyForRedaction = "";
29
+ try {
30
+ const provider = resolveProvider(opts.model, opts.provider, opts.modelData);
31
+ const ClientClass = builtinClients[provider] ?? registered[provider];
32
+ if (ClientClass === undefined) {
33
+ return failure(`Provider "${provider}" has no speech API. Register one with registerSpeechProvider(name, ClientClass).`);
34
+ }
35
+ const apiKey = resolveApiKey(provider, opts) ?? "";
36
+ apiKeyForRedaction = apiKey;
37
+ const { apiKey: _callerKeys, ...clientOpts } = opts;
38
+ const config = { ...clientOpts, provider, apiKey };
39
+ return success(new ClientClass(config));
40
+ }
41
+ catch (err) {
42
+ let msg = "getSpeechClient() failed";
43
+ if (err instanceof Error) {
44
+ msg = err.message;
45
+ }
46
+ const redacted = redactSecret(msg, apiKeyForRedaction);
47
+ getLogger().error("getSpeechClient() failed:", redacted);
48
+ return failure(redacted);
49
+ }
50
+ }
51
+ export async function speak(text, opts) {
52
+ const client = getSpeechClient(opts);
53
+ if (!client.success) {
54
+ return client;
55
+ }
56
+ return client.value.speak(text);
57
+ }
@@ -0,0 +1,31 @@
1
+ import type { ModelDataBlob } from "../modelData.js";
2
+ import { Result } from "../types/result.js";
3
+ import { BlobRef } from "../util/blobRef.js";
4
+ import type { TranscriptionResult } from "../transcription.js";
5
+ export declare const DEFAULT_TRANSCRIBE_BYTES: number;
6
+ export type TranscriptionClientConfig = {
7
+ model: string;
8
+ /** Resolved provider name. */
9
+ provider: string;
10
+ /** Resolved API key; empty string when none was found. */
11
+ apiKey: string;
12
+ modelData?: ModelDataBlob;
13
+ language?: string;
14
+ prompt?: string;
15
+ timestampGranularity?: "segment" | "word";
16
+ maxBytes?: number;
17
+ metadata?: Record<string, unknown>;
18
+ };
19
+ /**
20
+ * Shared transcription behavior, mirroring BaseClient for text generation:
21
+ * the public transcribe() template method owns blob loading, model-data-driven
22
+ * validation, cost, and the single redacting/logging exception boundary.
23
+ * Subclasses implement only _transcribe(): SDK call + response mapping.
24
+ */
25
+ export declare abstract class BaseTranscriptionClient {
26
+ protected config: TranscriptionClientConfig;
27
+ constructor(config: TranscriptionClientConfig);
28
+ transcribe(source: BlobRef): Promise<Result<TranscriptionResult>>;
29
+ /** Provider hook: SDK call + response mapping only; validation and cost live in the base. */
30
+ protected abstract _transcribe(data: Uint8Array, mimeType: string): Promise<Result<TranscriptionResult>>;
31
+ }
@@ -0,0 +1,107 @@
1
+ import { getModelForProvider, isSpeechToTextModel, } from "../models.js";
2
+ import { calculateTranscriptionCost } from "../model.js";
3
+ import { success, failure } from "../types/result.js";
4
+ import { loadBlob } from "../util/blobRef.js";
5
+ import { audioFormatForMime, canonicalizeMime } from "../util/mime.js";
6
+ import { redactSecret } from "../util/redact.js";
7
+ import { getLogger } from "../util/logger.js";
8
+ export const DEFAULT_TRANSCRIBE_BYTES = 25 * 1024 * 1024;
9
+ /** Validate the declarative STT constraint block once before consuming it. */
10
+ function transcriptionConstraintError(model) {
11
+ const modelMaxBytes = model.maxBytes;
12
+ if (modelMaxBytes !== undefined &&
13
+ (typeof modelMaxBytes !== "number" || !Number.isFinite(modelMaxBytes) || modelMaxBytes <= 0)) {
14
+ return `Model "${model.modelName}" has an invalid maxBytes value.`;
15
+ }
16
+ const supportedMimeTypes = model.supportedMimeTypes;
17
+ if (supportedMimeTypes !== undefined &&
18
+ (!Array.isArray(supportedMimeTypes) ||
19
+ !supportedMimeTypes.every((mime) => typeof mime === "string"))) {
20
+ return `Model "${model.modelName}" has invalid supportedMimeTypes.`;
21
+ }
22
+ return null;
23
+ }
24
+ // The caller's maxBytes is a safety limit; the model's maxBytes is the
25
+ // provider's hard cap. Take the smaller of whichever are present so a caller
26
+ // can tighten the limit but never bypass the provider cap.
27
+ function resolveTranscriptionMaxBytes(callerMaxBytes, model) {
28
+ if (callerMaxBytes !== undefined &&
29
+ (!Number.isFinite(callerMaxBytes) || callerMaxBytes <= 0)) {
30
+ return failure(`maxBytes must be a positive finite number (got ${callerMaxBytes}).`);
31
+ }
32
+ const limits = [];
33
+ if (callerMaxBytes !== undefined) {
34
+ limits.push(callerMaxBytes);
35
+ }
36
+ if (model?.maxBytes !== undefined) {
37
+ limits.push(model.maxBytes);
38
+ }
39
+ if (limits.length === 0) {
40
+ return success(DEFAULT_TRANSCRIBE_BYTES);
41
+ }
42
+ return success(Math.min(...limits));
43
+ }
44
+ /**
45
+ * Shared transcription behavior, mirroring BaseClient for text generation:
46
+ * the public transcribe() template method owns blob loading, model-data-driven
47
+ * validation, cost, and the single redacting/logging exception boundary.
48
+ * Subclasses implement only _transcribe(): SDK call + response mapping.
49
+ */
50
+ export class BaseTranscriptionClient {
51
+ config;
52
+ constructor(config) {
53
+ this.config = config;
54
+ }
55
+ async transcribe(source) {
56
+ try {
57
+ const model = getModelForProvider(this.config.provider, this.config.model, this.config.modelData);
58
+ if (model !== undefined && !isSpeechToTextModel(model)) {
59
+ return failure(`Model "${this.config.model}" is not a speech-to-text model.`);
60
+ }
61
+ if (model !== undefined) {
62
+ const constraintError = transcriptionConstraintError(model);
63
+ if (constraintError !== null) {
64
+ return failure(constraintError);
65
+ }
66
+ }
67
+ const effectiveLimit = resolveTranscriptionMaxBytes(this.config.maxBytes, model);
68
+ if (!effectiveLimit.success) {
69
+ return effectiveLimit;
70
+ }
71
+ let loaded;
72
+ try {
73
+ loaded = await loadBlob(source, { maxBytes: effectiveLimit.value });
74
+ }
75
+ catch (err) {
76
+ return failure(`Failed to load audio for transcription: ${err.message}`);
77
+ }
78
+ const mimeType = loaded.mimeType ?? "application/octet-stream";
79
+ if (model !== undefined && model.supportedMimeTypes !== undefined) {
80
+ const audioFormat = audioFormatForMime(mimeType);
81
+ const normalizedMime = audioFormat?.mimeType ?? canonicalizeMime(mimeType);
82
+ if (!model.supportedMimeTypes.includes(normalizedMime)) {
83
+ return failure(`Unsupported audio type "${mimeType}" for model "${this.config.model}". ` +
84
+ `Supported: ${model.supportedMimeTypes.join(", ")}.`);
85
+ }
86
+ }
87
+ const result = await this._transcribe(loaded.data, mimeType);
88
+ if (!result.success) {
89
+ return result;
90
+ }
91
+ const cost = calculateTranscriptionCost(model, result.value.durationSeconds);
92
+ if (cost !== undefined) {
93
+ result.value.cost = cost;
94
+ }
95
+ return result;
96
+ }
97
+ catch (err) {
98
+ let msg = "transcribe() failed";
99
+ if (err instanceof Error) {
100
+ msg = err.message;
101
+ }
102
+ const redacted = redactSecret(msg, this.config.apiKey);
103
+ getLogger().error("transcribe() provider failed:", redacted);
104
+ return failure(redacted);
105
+ }
106
+ }
107
+ }
@@ -0,0 +1,6 @@
1
+ import { Result } from "../types/result.js";
2
+ import { BaseTranscriptionClient } from "./baseTranscriptionClient.js";
3
+ import type { TranscriptionResult } from "../transcription.js";
4
+ export declare class OpenAITranscriptionClient extends BaseTranscriptionClient {
5
+ protected _transcribe(data: Uint8Array, mimeType: string): Promise<Result<TranscriptionResult>>;
6
+ }
@@ -0,0 +1,59 @@
1
+ import OpenAI, { toFile } from "openai";
2
+ import { success, failure } from "../types/result.js";
3
+ import { transcriptionAudioType } from "../util/audioMime.js";
4
+ import { BaseTranscriptionClient } from "./baseTranscriptionClient.js";
5
+ export class OpenAITranscriptionClient extends BaseTranscriptionClient {
6
+ // No try/catch here: BaseTranscriptionClient.transcribe() is the single
7
+ // redacting/logging exception boundary.
8
+ async _transcribe(data, mimeType) {
9
+ if (!this.config.apiKey) {
10
+ return failure("No OpenAI API key provided. Set apiKey.openAi or OPENAI_API_KEY.");
11
+ }
12
+ // Filename is an OpenAI upload detail, not part of the provider-neutral
13
+ // operation contract. Derive the synthetic name from the normalized MIME.
14
+ const filename = transcriptionAudioType(mimeType)?.filename ?? "audio.bin";
15
+ const client = new OpenAI({ apiKey: this.config.apiKey });
16
+ const file = await toFile(data, filename, { type: mimeType });
17
+ const granularities = [];
18
+ if (this.config.timestampGranularity) {
19
+ granularities.push(this.config.timestampGranularity);
20
+ }
21
+ const requestBody = {
22
+ file,
23
+ model: this.config.model,
24
+ response_format: "verbose_json",
25
+ };
26
+ if (this.config.language) {
27
+ requestBody.language = this.config.language;
28
+ }
29
+ if (this.config.prompt) {
30
+ requestBody.prompt = this.config.prompt;
31
+ }
32
+ if (granularities.length > 0) {
33
+ requestBody.timestamp_granularities = granularities;
34
+ }
35
+ const res = (await client.audio.transcriptions.create(requestBody));
36
+ const result = { text: res.text, raw: res };
37
+ if (res.language) {
38
+ result.language = res.language;
39
+ }
40
+ if (typeof res.duration === "number") {
41
+ result.durationSeconds = res.duration;
42
+ }
43
+ if (Array.isArray(res.segments)) {
44
+ result.segments = res.segments.map((segment) => ({
45
+ start: segment.start,
46
+ end: segment.end,
47
+ text: segment.text,
48
+ }));
49
+ }
50
+ if (Array.isArray(res.words)) {
51
+ result.words = res.words.map((word) => ({
52
+ start: word.start,
53
+ end: word.end,
54
+ word: word.word,
55
+ }));
56
+ }
57
+ return success(result);
58
+ }
59
+ }
@@ -0,0 +1,51 @@
1
+ import type { ModelDataBlob } from "./modelData.js";
2
+ import type { SmolConfig } from "./types.js";
3
+ import { Result } from "./types/result.js";
4
+ import { TokenUsage } from "./types/tokenUsage.js";
5
+ import { CostEstimate } from "./types/costEstimate.js";
6
+ import { BlobRef } from "./util/blobRef.js";
7
+ import { BaseTranscriptionClient, TranscriptionClientConfig } from "./transcription/baseTranscriptionClient.js";
8
+ export { DEFAULT_TRANSCRIBE_BYTES } from "./transcription/baseTranscriptionClient.js";
9
+ export type TranscribeOptions = {
10
+ model: string;
11
+ provider?: string;
12
+ modelData?: ModelDataBlob;
13
+ apiKey?: SmolConfig["apiKey"];
14
+ language?: string;
15
+ prompt?: string;
16
+ timestampGranularity?: "segment" | "word";
17
+ maxBytes?: number;
18
+ metadata?: Record<string, unknown>;
19
+ };
20
+ export type TranscriptionSegment = {
21
+ start: number;
22
+ end: number;
23
+ text: string;
24
+ };
25
+ export type TranscriptionWord = {
26
+ start: number;
27
+ end: number;
28
+ word: string;
29
+ };
30
+ export type TranscriptionResult = {
31
+ text: string;
32
+ language?: string;
33
+ durationSeconds?: number;
34
+ segments?: TranscriptionSegment[];
35
+ words?: TranscriptionWord[];
36
+ usage?: TokenUsage;
37
+ cost?: CostEstimate;
38
+ raw?: unknown;
39
+ };
40
+ export type TranscriptionClientClass = new (config: TranscriptionClientConfig) => BaseTranscriptionClient;
41
+ export declare function registerTranscriptionProvider(name: string, cls: TranscriptionClientClass): void;
42
+ /** Test-only: clear all registered custom providers so registrations don't leak across tests. */
43
+ export declare function _resetForTests(): void;
44
+ /**
45
+ * Resolve provider + API key and instantiate the matching transcription client
46
+ * for the declarative transcribe() operation. Never throws: a custom client
47
+ * class's constructor can throw, and this internal factory's catch redacts the
48
+ * resolved key so a constructor error cannot leak through the public wrapper.
49
+ */
50
+ export declare function getTranscriptionClient(opts: TranscribeOptions): Result<BaseTranscriptionClient>;
51
+ export declare function transcribe(source: BlobRef, opts: TranscribeOptions): Promise<Result<TranscriptionResult>>;
@@ -0,0 +1,58 @@
1
+ import { success, failure } from "./types/result.js";
2
+ import { redactSecret } from "./util/redact.js";
3
+ import { getLogger } from "./util/logger.js";
4
+ import { resolveProvider, resolveApiKey } from "./util/provider.js";
5
+ import { OpenAITranscriptionClient } from "./transcription/openai.js";
6
+ export { DEFAULT_TRANSCRIBE_BYTES } from "./transcription/baseTranscriptionClient.js";
7
+ // Checked before the user registry so a registered "openai" can't hijack the built-in.
8
+ const builtinClients = Object.create(null);
9
+ builtinClients["openai"] = OpenAITranscriptionClient;
10
+ // Null-prototype so provider names like "toString"/"__proto__" can't collide
11
+ // with Object.prototype or pollute the registry.
12
+ const registered = Object.create(null);
13
+ export function registerTranscriptionProvider(name, cls) {
14
+ registered[name] = cls;
15
+ }
16
+ /** Test-only: clear all registered custom providers so registrations don't leak across tests. */
17
+ export function _resetForTests() {
18
+ for (const key of Object.keys(registered)) {
19
+ delete registered[key];
20
+ }
21
+ }
22
+ /**
23
+ * Resolve provider + API key and instantiate the matching transcription client
24
+ * for the declarative transcribe() operation. Never throws: a custom client
25
+ * class's constructor can throw, and this internal factory's catch redacts the
26
+ * resolved key so a constructor error cannot leak through the public wrapper.
27
+ */
28
+ export function getTranscriptionClient(opts) {
29
+ let apiKeyForRedaction = "";
30
+ try {
31
+ const provider = resolveProvider(opts.model, opts.provider, opts.modelData);
32
+ const ClientClass = builtinClients[provider] ?? registered[provider];
33
+ if (ClientClass === undefined) {
34
+ return failure(`Provider "${provider}" has no transcription API. Register one with registerTranscriptionProvider(name, ClientClass).`);
35
+ }
36
+ const apiKey = resolveApiKey(provider, opts) ?? "";
37
+ apiKeyForRedaction = apiKey;
38
+ const { apiKey: _callerKeys, ...clientOpts } = opts;
39
+ const config = { ...clientOpts, provider, apiKey };
40
+ return success(new ClientClass(config));
41
+ }
42
+ catch (err) {
43
+ let msg = "getTranscriptionClient() failed";
44
+ if (err instanceof Error) {
45
+ msg = err.message;
46
+ }
47
+ const redacted = redactSecret(msg, apiKeyForRedaction);
48
+ getLogger().error("getTranscriptionClient() failed:", redacted);
49
+ return failure(redacted);
50
+ }
51
+ }
52
+ export async function transcribe(source, opts) {
53
+ const client = getTranscriptionClient(opts);
54
+ if (!client.success) {
55
+ return client;
56
+ }
57
+ return client.value.transcribe(source);
58
+ }
@@ -4,6 +4,8 @@ export type TokenUsage = {
4
4
  outputTokens: number;
5
5
  cachedInputTokens?: number;
6
6
  cacheCreationInputTokens?: number;
7
+ inputAudioTokens?: number;
8
+ outputAudioTokens?: number;
7
9
  totalTokens?: number;
8
10
  };
9
11
  export declare const TokenUsageSchema: z.ZodObject<{
@@ -11,6 +13,8 @@ export declare const TokenUsageSchema: z.ZodObject<{
11
13
  outputTokens: z.ZodNumber;
12
14
  cachedInputTokens: z.ZodOptional<z.ZodNumber>;
13
15
  cacheCreationInputTokens: z.ZodOptional<z.ZodNumber>;
16
+ inputAudioTokens: z.ZodOptional<z.ZodNumber>;
17
+ outputAudioTokens: z.ZodOptional<z.ZodNumber>;
14
18
  totalTokens: z.ZodOptional<z.ZodNumber>;
15
19
  }, z.core.$strip>;
16
20
  export declare function addTokenUsage(_a?: TokenUsage, _b?: TokenUsage): TokenUsage;
@@ -4,6 +4,8 @@ export const TokenUsageSchema = z.object({
4
4
  outputTokens: z.number(),
5
5
  cachedInputTokens: z.number().optional(),
6
6
  cacheCreationInputTokens: z.number().optional(),
7
+ inputAudioTokens: z.number().optional(),
8
+ outputAudioTokens: z.number().optional(),
7
9
  totalTokens: z.number().optional(),
8
10
  });
9
11
  export function addTokenUsage(_a, _b) {
@@ -22,6 +24,8 @@ export function addTokenUsage(_a, _b) {
22
24
  outputTokens: a.outputTokens + b.outputTokens,
23
25
  cachedInputTokens: (a.cachedInputTokens || 0) + (b.cachedInputTokens || 0),
24
26
  cacheCreationInputTokens: (a.cacheCreationInputTokens || 0) + (b.cacheCreationInputTokens || 0),
27
+ inputAudioTokens: (a.inputAudioTokens || 0) + (b.inputAudioTokens || 0),
28
+ outputAudioTokens: (a.outputAudioTokens || 0) + (b.outputAudioTokens || 0),
25
29
  totalTokens: (a.totalTokens || 0) + (b.totalTokens || 0),
26
30
  };
27
31
  }
package/dist/types.d.ts CHANGED
@@ -30,6 +30,9 @@ export type SmolConfig = {
30
30
  deepInfra?: string;
31
31
  liteLlm?: string;
32
32
  openAiCompat?: string;
33
+ /** Arbitrary provider names, for keys targeting a custom-registered provider
34
+ * (e.g. registerProvider("acme", ...)), keyed by the exact registered name. */
35
+ [provider: string]: string | undefined;
33
36
  };
34
37
  /** Custom base URLs, nested by provider. Defaults are baked in where applicable
35
38
  * (e.g. openrouter, deepinfra); litellm and openai-compat require an explicit URL. */
@@ -1,4 +1,4 @@
1
- import type { ImageRef } from "./imageRef.js";
1
+ import type { ImageRef } from "./blobRef.js";
2
2
  /** Filename sent to providers that require one when a file part omits its own. */
3
3
  export declare const DEFAULT_ATTACHMENT_FILENAME = "attachment.pdf";
4
4
  /** A file part's filename, falling back to the shared default when unset. */
@@ -0,0 +1,9 @@
1
+ export type SpeakFormat = "mp3" | "opus" | "aac" | "flac" | "wav" | "pcm";
2
+ export type TranscriptionAudioType = {
3
+ extension: string;
4
+ filename: string;
5
+ };
6
+ export declare function transcriptionAudioType(mime: string): TranscriptionAudioType | null;
7
+ export declare function chatAudioFormat(mime: string): "mp3" | "wav" | null;
8
+ export declare const SPEECH_FORMAT_TO_MIME: Record<SpeakFormat, string>;
9
+ export declare function isSpeakFormat(value: string): value is SpeakFormat;
@@ -0,0 +1,33 @@
1
+ import { audioFormatForMime } from "./mime.js";
2
+ export function transcriptionAudioType(mime) {
3
+ const format = audioFormatForMime(mime);
4
+ if (format === null) {
5
+ return null;
6
+ }
7
+ return { extension: format.extension, filename: `audio.${format.extension}` };
8
+ }
9
+ export function chatAudioFormat(mime) {
10
+ const format = audioFormatForMime(mime);
11
+ if (format === null) {
12
+ return null;
13
+ }
14
+ if (format.extension === "mp3" || format.extension === "wav") {
15
+ return format.extension;
16
+ }
17
+ return null;
18
+ }
19
+ // PCM from OpenAI is headerless s16le / 24kHz / mono, which audio/L16 (big-endian
20
+ // per RFC) would misdescribe — use octet-stream + structured metadata instead.
21
+ export const SPEECH_FORMAT_TO_MIME = {
22
+ mp3: "audio/mpeg",
23
+ opus: "audio/ogg",
24
+ aac: "audio/aac",
25
+ flac: "audio/flac",
26
+ wav: "audio/wav",
27
+ pcm: "application/octet-stream",
28
+ };
29
+ // Object.hasOwn (not `in`) so prototype keys like "toString"/"__proto__"
30
+ // never pass the guard.
31
+ export function isSpeakFormat(value) {
32
+ return Object.hasOwn(SPEECH_FORMAT_TO_MIME, value);
33
+ }
@@ -1,4 +1,4 @@
1
- export type ImageRef = {
1
+ export type BlobRef = {
2
2
  kind: "bytes";
3
3
  data: Uint8Array;
4
4
  mimeType: string;
@@ -20,23 +20,23 @@ export type ImageRef = {
20
20
  */
21
21
  timeoutMs?: number;
22
22
  };
23
- /** Neutral alias of the source union for non-image uses (uploads via {@link loadBlob}). */
24
- export type BlobRef = ImageRef;
25
- export type NormalizedImage = {
23
+ /** Alias of {@link BlobRef} for image call sites, where it documents intent. */
24
+ export type ImageRef = BlobRef;
25
+ export type NormalizedBlob = {
26
26
  data: Uint8Array;
27
27
  mimeType: string;
28
28
  };
29
- /** Default timeout for fetching image URLs during normalization (60 seconds). */
29
+ /** Default timeout for fetching attachment URLs during normalization (60 seconds). */
30
30
  export declare const DEFAULT_FETCH_TIMEOUT_MS = 60000;
31
- export declare function normalizeImageRef(ref: ImageRef, options?: {
31
+ export declare function normalizeBlob(ref: BlobRef, options?: {
32
32
  allowedMimePrefixes?: string[];
33
33
  maxBytes?: number;
34
- }): Promise<NormalizedImage>;
34
+ }): Promise<NormalizedBlob>;
35
35
  /**
36
- * Load a source to bytes WITHOUT the image MIME gate (used by file uploads,
36
+ * Load a source to bytes WITHOUT the MIME gate (used by file uploads,
37
37
  * which accept any type). Enforces `maxBytes` across all kinds.
38
38
  */
39
- export declare function loadBlob(ref: ImageRef, options?: {
39
+ export declare function loadBlob(ref: BlobRef, options?: {
40
40
  maxBytes?: number;
41
41
  }): Promise<{
42
42
  data: Uint8Array;
@@ -1,15 +1,8 @@
1
1
  import { readFile, stat } from "node:fs/promises";
2
2
  import { extname } from "node:path";
3
- /** Default timeout for fetching image URLs during normalization (60 seconds). */
3
+ import { EXT_TO_MIME } from "./mime.js";
4
+ /** Default timeout for fetching attachment URLs during normalization (60 seconds). */
4
5
  export const DEFAULT_FETCH_TIMEOUT_MS = 60_000;
5
- const EXT_TO_MIME = {
6
- ".png": "image/png",
7
- ".jpg": "image/jpeg",
8
- ".jpeg": "image/jpeg",
9
- ".webp": "image/webp",
10
- ".gif": "image/gif",
11
- ".pdf": "application/pdf",
12
- };
13
6
  function isAllowedMime(mimeType, allowedPrefixes) {
14
7
  for (const prefix of allowedPrefixes) {
15
8
  if (mimeType.startsWith(prefix)) {
@@ -18,7 +11,7 @@ function isAllowedMime(mimeType, allowedPrefixes) {
18
11
  }
19
12
  return false;
20
13
  }
21
- export async function normalizeImageRef(ref, options = {}) {
14
+ export async function normalizeBlob(ref, options = {}) {
22
15
  const allowed = options.allowedMimePrefixes ?? ["image/"];
23
16
  const result = await loadRef(ref, allowed, options.maxBytes);
24
17
  if (options.maxBytes !== undefined && result.data.length > options.maxBytes) {
@@ -31,7 +24,7 @@ export async function normalizeImageRef(ref, options = {}) {
31
24
  return { data: result.data, mimeType: result.mimeType };
32
25
  }
33
26
  /**
34
- * Load a source to bytes WITHOUT the image MIME gate (used by file uploads,
27
+ * Load a source to bytes WITHOUT the MIME gate (used by file uploads,
35
28
  * which accept any type). Enforces `maxBytes` across all kinds.
36
29
  */
37
30
  export async function loadBlob(ref, options = {}) {
@@ -94,7 +87,7 @@ async function loadRef(ref, allowed, maxBytes) {
94
87
  const mimeType = ref.mimeType ?? inferred;
95
88
  if (allowed !== null && (!mimeType || !isAllowedMime(mimeType, allowed))) {
96
89
  throw new Error(`Could not determine an allowed MIME type for path "${ref.path}". ` +
97
- `Allowed: ${allowed.join(", ")}. Pass an explicit mimeType on the ImageRef.`);
90
+ `Allowed: ${allowed.join(", ")}. Pass an explicit mimeType on the BlobRef.`);
98
91
  }
99
92
  return { data: new Uint8Array(buf), mimeType };
100
93
  }
@@ -128,7 +121,7 @@ async function loadRef(ref, allowed, maxBytes) {
128
121
  if (allowed !== null && (!mimeType || !isAllowedMime(mimeType, allowed))) {
129
122
  throw new Error(`Could not determine an allowed MIME type for URL "${ref.url}". ` +
130
123
  `Response Content-Type was "${contentType ?? "missing"}". ` +
131
- `Allowed: ${allowed.join(", ")}. Pass an explicit mimeType on the ImageRef.`);
124
+ `Allowed: ${allowed.join(", ")}. Pass an explicit mimeType on the BlobRef.`);
132
125
  }
133
126
  return { data: buf, mimeType };
134
127
  }