smoltalk 0.9.0 → 0.10.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +175 -7
- package/dist/classes/message/AssistantMessage.d.ts +2 -0
- package/dist/classes/message/UserMessage.d.ts +21 -0
- package/dist/classes/message/UserMessage.js +3 -0
- package/dist/classes/message/contentParts.d.ts +71 -2
- package/dist/classes/message/contentParts.js +6 -0
- package/dist/classes/message/index.d.ts +5 -2
- package/dist/classes/message/index.js +7 -0
- package/dist/classes/message/renderers/AnthropicRenderer.d.ts +2 -1
- package/dist/classes/message/renderers/AnthropicRenderer.js +3 -0
- package/dist/classes/message/renderers/GoogleRenderer.d.ts +2 -1
- package/dist/classes/message/renderers/GoogleRenderer.js +3 -0
- package/dist/classes/message/renderers/JSONRenderer.d.ts +2 -1
- package/dist/classes/message/renderers/JSONRenderer.js +4 -0
- package/dist/classes/message/renderers/OpenAIChatRenderer.d.ts +8 -1
- package/dist/classes/message/renderers/OpenAIChatRenderer.js +18 -0
- package/dist/classes/message/renderers/OpenAIResponsesRenderer.d.ts +2 -1
- package/dist/classes/message/renderers/OpenAIResponsesRenderer.js +3 -0
- package/dist/classes/message/renderers/PartRenderer.d.ts +3 -2
- package/dist/classes/message/renderers/PartRenderer.js +3 -0
- package/dist/client.js +1 -0
- package/dist/clients/anthropic.js +1 -1
- package/dist/clients/baseClient.d.ts +13 -1
- package/dist/clients/baseClient.js +36 -7
- package/dist/clients/google.js +1 -1
- package/dist/clients/ollama.js +1 -1
- package/dist/clients/openai.d.ts +2 -1
- package/dist/clients/openai.js +15 -3
- package/dist/clients/openaiCompat.d.ts +2 -0
- package/dist/clients/openaiCompat.js +5 -0
- package/dist/clients/openaiResponses.js +1 -1
- package/dist/clients/resolveAttachments.d.ts +8 -4
- package/dist/clients/resolveAttachments.js +101 -50
- package/dist/embed.d.ts +4 -0
- package/dist/files.d.ts +1 -1
- package/dist/files.js +1 -1
- package/dist/image/google.js +2 -2
- package/dist/image/openai.js +3 -3
- package/dist/image.d.ts +1 -1
- package/dist/index.d.ts +10 -2
- package/dist/index.js +7 -1
- package/dist/model.d.ts +15 -4
- package/dist/model.js +72 -12
- package/dist/models.d.ts +204 -22
- package/dist/models.js +211 -30
- package/dist/speech/baseSpeechClient.d.ts +36 -0
- package/dist/speech/baseSpeechClient.js +117 -0
- package/dist/speech/google.d.ts +6 -0
- package/dist/speech/google.js +54 -0
- package/dist/speech/groq.d.ts +11 -0
- package/dist/speech/groq.js +19 -0
- package/dist/speech/openai.d.ts +14 -0
- package/dist/speech/openai.js +51 -0
- package/dist/speech/openaiCompat.d.ts +13 -0
- package/dist/speech/openaiCompat.js +22 -0
- package/dist/speech.d.ts +45 -0
- package/dist/speech.js +63 -0
- package/dist/transcription/baseTranscriptionClient.d.ts +36 -0
- package/dist/transcription/baseTranscriptionClient.js +133 -0
- package/dist/transcription/google.d.ts +6 -0
- package/dist/transcription/google.js +56 -0
- package/dist/transcription/groq.d.ts +10 -0
- package/dist/transcription/groq.js +17 -0
- package/dist/transcription/openai.d.ts +11 -0
- package/dist/transcription/openai.js +67 -0
- package/dist/transcription/openaiCompat.d.ts +13 -0
- package/dist/transcription/openaiCompat.js +22 -0
- package/dist/transcription.d.ts +54 -0
- package/dist/transcription.js +64 -0
- package/dist/types/tokenUsage.d.ts +4 -0
- package/dist/types/tokenUsage.js +4 -0
- package/dist/types.d.ts +4 -0
- package/dist/util/attachments.d.ts +1 -1
- package/dist/util/audioMime.d.ts +26 -0
- package/dist/util/audioMime.js +74 -0
- package/dist/util/{imageRef.d.ts → blobRef.d.ts} +9 -9
- package/dist/util/{imageRef.js → blobRef.js} +6 -13
- package/dist/util/googleAudioUsage.d.ts +14 -0
- package/dist/util/googleAudioUsage.js +52 -0
- package/dist/util/mime.d.ts +21 -0
- package/dist/util/mime.js +54 -0
- package/dist/util/modalities.d.ts +6 -2
- package/dist/util/modalities.js +13 -15
- package/dist/util/provider.d.ts +3 -0
- package/dist/util/provider.js +3 -1
- package/package.json +1 -1
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
import OpenAI from "openai";
|
|
2
|
+
import { OpenAISpeechClient } from "./openai.js";
|
|
3
|
+
import { resolveBaseUrl } from "../util/provider.js";
|
|
4
|
+
/**
|
|
5
|
+
* Generic OpenAI-compatible speech (TTS) client. Point it at any provider
|
|
6
|
+
* exposing an OpenAI-shaped /audio/speech endpoint via
|
|
7
|
+
* `config.baseUrl.openAiCompat` (or OPENAI_COMPAT_BASE_URL) and
|
|
8
|
+
* `config.apiKey.openAiCompat` (or OPENAI_COMPAT_API_KEY). Mirrors the chat
|
|
9
|
+
* `SmolOpenAiCompat` client. Inherits OpenAI's `mp3` default format.
|
|
10
|
+
*/
|
|
11
|
+
export class OpenAiCompatSpeechClient extends OpenAISpeechClient {
|
|
12
|
+
makeClient() {
|
|
13
|
+
const baseURL = resolveBaseUrl("openai-compat", { baseUrl: this.config.baseUrl });
|
|
14
|
+
if (!baseURL) {
|
|
15
|
+
throw new Error("openai-compat: base URL required (config.baseUrl.openAiCompat or OPENAI_COMPAT_BASE_URL).");
|
|
16
|
+
}
|
|
17
|
+
return new OpenAI({ apiKey: this.config.apiKey, baseURL });
|
|
18
|
+
}
|
|
19
|
+
noKeyMessage() {
|
|
20
|
+
return "No API key provided. Set apiKey.openAiCompat or OPENAI_COMPAT_API_KEY.";
|
|
21
|
+
}
|
|
22
|
+
}
|
package/dist/speech.d.ts
ADDED
|
@@ -0,0 +1,45 @@
|
|
|
1
|
+
import type { ModelDataBlob } from "./modelData.js";
|
|
2
|
+
import type { SmolConfig } from "./types.js";
|
|
3
|
+
import { Result } from "./types/result.js";
|
|
4
|
+
import { CostEstimate } from "./types/costEstimate.js";
|
|
5
|
+
import { TokenUsage } from "./types/tokenUsage.js";
|
|
6
|
+
import { BaseSpeechClient, SpeechClientConfig } from "./speech/baseSpeechClient.js";
|
|
7
|
+
export type SpeakOptions = {
|
|
8
|
+
model: string;
|
|
9
|
+
voice: string;
|
|
10
|
+
provider?: string;
|
|
11
|
+
modelData?: ModelDataBlob;
|
|
12
|
+
apiKey?: SmolConfig["apiKey"];
|
|
13
|
+
baseUrl?: SmolConfig["baseUrl"];
|
|
14
|
+
format?: string;
|
|
15
|
+
speed?: number;
|
|
16
|
+
metadata?: Record<string, unknown>;
|
|
17
|
+
/** Abort the in-flight provider request when this signal fires. */
|
|
18
|
+
abortSignal?: AbortSignal;
|
|
19
|
+
};
|
|
20
|
+
export type PcmAudioMetadata = {
|
|
21
|
+
sampleRateHz: number;
|
|
22
|
+
/** Sample representation, e.g. "s16le"; provider-specific. */
|
|
23
|
+
sampleFormat: string;
|
|
24
|
+
channels: number;
|
|
25
|
+
};
|
|
26
|
+
export type SpeechResult = {
|
|
27
|
+
audio: Uint8Array;
|
|
28
|
+
mimeType: string;
|
|
29
|
+
pcm?: PcmAudioMetadata;
|
|
30
|
+
usage?: TokenUsage;
|
|
31
|
+
cost?: CostEstimate;
|
|
32
|
+
raw?: unknown;
|
|
33
|
+
};
|
|
34
|
+
export type SpeechClientClass = new (config: SpeechClientConfig) => BaseSpeechClient;
|
|
35
|
+
export declare function registerSpeechProvider(name: string, cls: SpeechClientClass): void;
|
|
36
|
+
/** Test-only: clear all registered custom providers so registrations don't leak across tests. */
|
|
37
|
+
export declare function _resetForTests(): void;
|
|
38
|
+
/**
|
|
39
|
+
* Resolve provider + API key and instantiate the matching speech client for
|
|
40
|
+
* the declarative speak() operation. Never throws: a custom client class's
|
|
41
|
+
* constructor can throw, and this internal factory's catch redacts the
|
|
42
|
+
* resolved key so a constructor error cannot leak through the public wrapper.
|
|
43
|
+
*/
|
|
44
|
+
export declare function getSpeechClient(opts: SpeakOptions): Result<BaseSpeechClient>;
|
|
45
|
+
export declare function speak(text: string, opts: SpeakOptions): Promise<Result<SpeechResult>>;
|
package/dist/speech.js
ADDED
|
@@ -0,0 +1,63 @@
|
|
|
1
|
+
import { success, failure } from "./types/result.js";
|
|
2
|
+
import { redactSecret } from "./util/redact.js";
|
|
3
|
+
import { getLogger } from "./util/logger.js";
|
|
4
|
+
import { resolveProvider, resolveApiKey } from "./util/provider.js";
|
|
5
|
+
import { OpenAISpeechClient } from "./speech/openai.js";
|
|
6
|
+
import { GroqSpeechClient } from "./speech/groq.js";
|
|
7
|
+
import { GoogleSpeechClient } from "./speech/google.js";
|
|
8
|
+
import { OpenAiCompatSpeechClient } from "./speech/openaiCompat.js";
|
|
9
|
+
// Checked before the user registry so a registered "openai" can't hijack the built-in.
|
|
10
|
+
const builtinClients = Object.create(null);
|
|
11
|
+
builtinClients["openai"] = OpenAISpeechClient;
|
|
12
|
+
builtinClients["groq"] = GroqSpeechClient;
|
|
13
|
+
builtinClients["google"] = GoogleSpeechClient;
|
|
14
|
+
builtinClients["openai-compat"] = OpenAiCompatSpeechClient;
|
|
15
|
+
// Null-prototype so provider names like "toString"/"__proto__" can't collide
|
|
16
|
+
// with Object.prototype or pollute the registry.
|
|
17
|
+
const registered = Object.create(null);
|
|
18
|
+
export function registerSpeechProvider(name, cls) {
|
|
19
|
+
registered[name] = cls;
|
|
20
|
+
}
|
|
21
|
+
/** Test-only: clear all registered custom providers so registrations don't leak across tests. */
|
|
22
|
+
export function _resetForTests() {
|
|
23
|
+
for (const key of Object.keys(registered)) {
|
|
24
|
+
delete registered[key];
|
|
25
|
+
}
|
|
26
|
+
}
|
|
27
|
+
/**
|
|
28
|
+
* Resolve provider + API key and instantiate the matching speech client for
|
|
29
|
+
* the declarative speak() operation. Never throws: a custom client class's
|
|
30
|
+
* constructor can throw, and this internal factory's catch redacts the
|
|
31
|
+
* resolved key so a constructor error cannot leak through the public wrapper.
|
|
32
|
+
*/
|
|
33
|
+
export function getSpeechClient(opts) {
|
|
34
|
+
let apiKeyForRedaction = "";
|
|
35
|
+
try {
|
|
36
|
+
const provider = resolveProvider(opts.model, opts.provider, opts.modelData);
|
|
37
|
+
const ClientClass = builtinClients[provider] ?? registered[provider];
|
|
38
|
+
if (ClientClass === undefined) {
|
|
39
|
+
return failure(`Provider "${provider}" has no speech API. Register one with registerSpeechProvider(name, ClientClass).`);
|
|
40
|
+
}
|
|
41
|
+
const apiKey = resolveApiKey(provider, opts) ?? "";
|
|
42
|
+
apiKeyForRedaction = apiKey;
|
|
43
|
+
const { apiKey: _callerKeys, ...clientOpts } = opts;
|
|
44
|
+
const config = { ...clientOpts, provider, apiKey };
|
|
45
|
+
return success(new ClientClass(config));
|
|
46
|
+
}
|
|
47
|
+
catch (err) {
|
|
48
|
+
let msg = "getSpeechClient() failed";
|
|
49
|
+
if (err instanceof Error) {
|
|
50
|
+
msg = err.message;
|
|
51
|
+
}
|
|
52
|
+
const redacted = redactSecret(msg, apiKeyForRedaction);
|
|
53
|
+
getLogger().error("getSpeechClient() failed:", redacted);
|
|
54
|
+
return failure(redacted);
|
|
55
|
+
}
|
|
56
|
+
}
|
|
57
|
+
export async function speak(text, opts) {
|
|
58
|
+
const client = getSpeechClient(opts);
|
|
59
|
+
if (!client.success) {
|
|
60
|
+
return client;
|
|
61
|
+
}
|
|
62
|
+
return client.value.speak(text);
|
|
63
|
+
}
|
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
import type { ModelDataBlob } from "../modelData.js";
|
|
2
|
+
import type { SmolConfig } from "../types.js";
|
|
3
|
+
import { Result } from "../types/result.js";
|
|
4
|
+
import { BlobRef } from "../util/blobRef.js";
|
|
5
|
+
import type { TranscriptionResult } from "../transcription.js";
|
|
6
|
+
export declare const DEFAULT_TRANSCRIBE_BYTES: number;
|
|
7
|
+
export type TranscriptionClientConfig = {
|
|
8
|
+
model: string;
|
|
9
|
+
/** Resolved provider name. */
|
|
10
|
+
provider: string;
|
|
11
|
+
/** Resolved API key; empty string when none was found. */
|
|
12
|
+
apiKey: string;
|
|
13
|
+
/** Base-URL map (for OpenAI-compatible providers); read via resolveBaseUrl. */
|
|
14
|
+
baseUrl?: SmolConfig["baseUrl"];
|
|
15
|
+
modelData?: ModelDataBlob;
|
|
16
|
+
language?: string;
|
|
17
|
+
prompt?: string;
|
|
18
|
+
timestampGranularity?: "segment" | "word";
|
|
19
|
+
maxBytes?: number;
|
|
20
|
+
metadata?: Record<string, unknown>;
|
|
21
|
+
/** Abort the in-flight provider request when this signal fires. */
|
|
22
|
+
abortSignal?: AbortSignal;
|
|
23
|
+
};
|
|
24
|
+
/**
|
|
25
|
+
* Shared transcription behavior, mirroring BaseClient for text generation:
|
|
26
|
+
* the public transcribe() template method owns blob loading, model-data-driven
|
|
27
|
+
* validation, cost, and the single redacting/logging exception boundary.
|
|
28
|
+
* Subclasses implement only _transcribe(): SDK call + response mapping.
|
|
29
|
+
*/
|
|
30
|
+
export declare abstract class BaseTranscriptionClient {
|
|
31
|
+
protected config: TranscriptionClientConfig;
|
|
32
|
+
constructor(config: TranscriptionClientConfig);
|
|
33
|
+
transcribe(source: BlobRef): Promise<Result<TranscriptionResult>>;
|
|
34
|
+
/** Provider hook: SDK call + response mapping only; validation and cost live in the base. */
|
|
35
|
+
protected abstract _transcribe(data: Uint8Array, mimeType: string): Promise<Result<TranscriptionResult>>;
|
|
36
|
+
}
|
|
@@ -0,0 +1,133 @@
|
|
|
1
|
+
import { getModelForProvider, isSpeechToTextModel, modelSupportsInputModality, audioInputConstraints, } from "../models.js";
|
|
2
|
+
import { Model, calculateTranscriptionCost } from "../model.js";
|
|
3
|
+
import { success, failure } from "../types/result.js";
|
|
4
|
+
import { loadBlob } from "../util/blobRef.js";
|
|
5
|
+
import { audioFormatForMime, canonicalizeMime } from "../util/mime.js";
|
|
6
|
+
import { redactSecret } from "../util/redact.js";
|
|
7
|
+
import { getLogger } from "../util/logger.js";
|
|
8
|
+
export const DEFAULT_TRANSCRIBE_BYTES = 25 * 1024 * 1024;
|
|
9
|
+
/** Validate the declarative STT constraint block once before consuming it. */
|
|
10
|
+
function transcriptionConstraintError(modelName, c) {
|
|
11
|
+
const modelMaxBytes = c.maxBytes;
|
|
12
|
+
if (modelMaxBytes !== undefined &&
|
|
13
|
+
(typeof modelMaxBytes !== "number" || !Number.isFinite(modelMaxBytes) || modelMaxBytes <= 0)) {
|
|
14
|
+
return `Model "${modelName}" has an invalid maxBytes value.`;
|
|
15
|
+
}
|
|
16
|
+
const supportedMimeTypes = c.supportedMimeTypes;
|
|
17
|
+
if (supportedMimeTypes !== undefined &&
|
|
18
|
+
(!Array.isArray(supportedMimeTypes) ||
|
|
19
|
+
!supportedMimeTypes.every((mime) => typeof mime === "string"))) {
|
|
20
|
+
return `Model "${modelName}" has invalid supportedMimeTypes.`;
|
|
21
|
+
}
|
|
22
|
+
return null;
|
|
23
|
+
}
|
|
24
|
+
// The caller's maxBytes is a safety limit; the model's maxBytes is the
|
|
25
|
+
// provider's hard cap. Take the smaller of whichever are present so a caller
|
|
26
|
+
// can tighten the limit but never bypass the provider cap.
|
|
27
|
+
function resolveTranscriptionMaxBytes(callerMaxBytes, modelMaxBytes) {
|
|
28
|
+
if (callerMaxBytes !== undefined &&
|
|
29
|
+
(!Number.isFinite(callerMaxBytes) || callerMaxBytes <= 0)) {
|
|
30
|
+
return failure(`maxBytes must be a positive finite number (got ${callerMaxBytes}).`);
|
|
31
|
+
}
|
|
32
|
+
const limits = [];
|
|
33
|
+
if (callerMaxBytes !== undefined) {
|
|
34
|
+
limits.push(callerMaxBytes);
|
|
35
|
+
}
|
|
36
|
+
if (modelMaxBytes !== undefined) {
|
|
37
|
+
limits.push(modelMaxBytes);
|
|
38
|
+
}
|
|
39
|
+
if (limits.length === 0) {
|
|
40
|
+
return success(DEFAULT_TRANSCRIBE_BYTES);
|
|
41
|
+
}
|
|
42
|
+
return success(Math.min(...limits));
|
|
43
|
+
}
|
|
44
|
+
/**
|
|
45
|
+
* Shared transcription behavior, mirroring BaseClient for text generation:
|
|
46
|
+
* the public transcribe() template method owns blob loading, model-data-driven
|
|
47
|
+
* validation, cost, and the single redacting/logging exception boundary.
|
|
48
|
+
* Subclasses implement only _transcribe(): SDK call + response mapping.
|
|
49
|
+
*/
|
|
50
|
+
export class BaseTranscriptionClient {
|
|
51
|
+
config;
|
|
52
|
+
constructor(config) {
|
|
53
|
+
this.config = config;
|
|
54
|
+
}
|
|
55
|
+
async transcribe(source) {
|
|
56
|
+
// Already-aborted signal: stop before doing any paid work.
|
|
57
|
+
if (this.config.abortSignal?.aborted) {
|
|
58
|
+
return failure("Request was aborted");
|
|
59
|
+
}
|
|
60
|
+
try {
|
|
61
|
+
const model = getModelForProvider(this.config.provider, this.config.model, this.config.modelData);
|
|
62
|
+
// A model is a valid transcription target if it is a dedicated STT model,
|
|
63
|
+
// or a multimodal text model that accepts audio input (e.g. Gemini). An
|
|
64
|
+
// unknown model (undefined) flows through — the provider is authority.
|
|
65
|
+
const acceptsAudio = model === undefined ||
|
|
66
|
+
isSpeechToTextModel(model) ||
|
|
67
|
+
modelSupportsInputModality(this.config.model, "audio", this.config.modelData, this.config.provider) === true;
|
|
68
|
+
if (!acceptsAudio) {
|
|
69
|
+
return failure(`Model "${this.config.model}" cannot accept audio input (not a transcription model).`);
|
|
70
|
+
}
|
|
71
|
+
const constraints = model !== undefined ? audioInputConstraints(model) : {};
|
|
72
|
+
if (model !== undefined) {
|
|
73
|
+
const constraintError = transcriptionConstraintError(this.config.model, constraints);
|
|
74
|
+
if (constraintError !== null) {
|
|
75
|
+
return failure(constraintError);
|
|
76
|
+
}
|
|
77
|
+
}
|
|
78
|
+
const effectiveLimit = resolveTranscriptionMaxBytes(this.config.maxBytes, constraints.maxBytes);
|
|
79
|
+
if (!effectiveLimit.success) {
|
|
80
|
+
return effectiveLimit;
|
|
81
|
+
}
|
|
82
|
+
let loaded;
|
|
83
|
+
try {
|
|
84
|
+
loaded = await loadBlob(source, { maxBytes: effectiveLimit.value });
|
|
85
|
+
}
|
|
86
|
+
catch (err) {
|
|
87
|
+
return failure(`Failed to load audio for transcription: ${err.message}`);
|
|
88
|
+
}
|
|
89
|
+
const mimeType = loaded.mimeType ?? "application/octet-stream";
|
|
90
|
+
if (constraints.supportedMimeTypes !== undefined) {
|
|
91
|
+
const audioFormat = audioFormatForMime(mimeType);
|
|
92
|
+
const normalizedMime = audioFormat?.mimeType ?? canonicalizeMime(mimeType);
|
|
93
|
+
if (!constraints.supportedMimeTypes.includes(normalizedMime)) {
|
|
94
|
+
return failure(`Unsupported audio type "${mimeType}" for model "${this.config.model}". ` +
|
|
95
|
+
`Supported: ${constraints.supportedMimeTypes.join(", ")}.`);
|
|
96
|
+
}
|
|
97
|
+
}
|
|
98
|
+
// Re-check after the async preflight (blob load / MIME validation): the
|
|
99
|
+
// signal may have fired during it, and we must not dispatch a request.
|
|
100
|
+
if (this.config.abortSignal?.aborted) {
|
|
101
|
+
return failure("Request was aborted");
|
|
102
|
+
}
|
|
103
|
+
const result = await this._transcribe(loaded.data, mimeType);
|
|
104
|
+
if (!result.success) {
|
|
105
|
+
return result;
|
|
106
|
+
}
|
|
107
|
+
let cost = calculateTranscriptionCost(model, result.value.durationSeconds);
|
|
108
|
+
if (cost === undefined && result.value.usage !== undefined) {
|
|
109
|
+
// Token-billed providers (Gemini) price through the shared cost engine.
|
|
110
|
+
cost =
|
|
111
|
+
new Model(this.config.model, this.config.provider, this.config.modelData).calculateCost(result.value.usage) ?? undefined;
|
|
112
|
+
}
|
|
113
|
+
if (cost !== undefined) {
|
|
114
|
+
result.value.cost = cost;
|
|
115
|
+
}
|
|
116
|
+
return result;
|
|
117
|
+
}
|
|
118
|
+
catch (err) {
|
|
119
|
+
// Caller-initiated cancellation surfaces as a distinguishable failure
|
|
120
|
+
// (matching the chat path), not a redacted provider error.
|
|
121
|
+
if (this.config.abortSignal?.aborted) {
|
|
122
|
+
return failure("Request was aborted");
|
|
123
|
+
}
|
|
124
|
+
let msg = "transcribe() failed";
|
|
125
|
+
if (err instanceof Error) {
|
|
126
|
+
msg = err.message;
|
|
127
|
+
}
|
|
128
|
+
const redacted = redactSecret(msg, this.config.apiKey);
|
|
129
|
+
getLogger().error("transcribe() provider failed:", redacted);
|
|
130
|
+
return failure(redacted);
|
|
131
|
+
}
|
|
132
|
+
}
|
|
133
|
+
}
|
|
@@ -0,0 +1,6 @@
|
|
|
1
|
+
import { Result } from "../types/result.js";
|
|
2
|
+
import { BaseTranscriptionClient } from "./baseTranscriptionClient.js";
|
|
3
|
+
import type { TranscriptionResult } from "../transcription.js";
|
|
4
|
+
export declare class GoogleTranscriptionClient extends BaseTranscriptionClient {
|
|
5
|
+
protected _transcribe(data: Uint8Array, mimeType: string): Promise<Result<TranscriptionResult>>;
|
|
6
|
+
}
|
|
@@ -0,0 +1,56 @@
|
|
|
1
|
+
import { GoogleGenAI } from "@google/genai";
|
|
2
|
+
import { success, failure } from "../types/result.js";
|
|
3
|
+
import { BaseTranscriptionClient } from "./baseTranscriptionClient.js";
|
|
4
|
+
import { normalizeGoogleAudioUsage } from "../util/googleAudioUsage.js";
|
|
5
|
+
import { googleAudioWireMime } from "../util/audioMime.js";
|
|
6
|
+
// Gemini's inline request cap is 20 MB for the *entire* request (audio bytes,
|
|
7
|
+
// base64 expansion, prompt, and SDK envelope), not the raw file alone.
|
|
8
|
+
const GOOGLE_INLINE_REQUEST_MAX_BYTES = 20_000_000;
|
|
9
|
+
export class GoogleTranscriptionClient extends BaseTranscriptionClient {
|
|
10
|
+
// No try/catch: BaseTranscriptionClient.transcribe() is the exception boundary.
|
|
11
|
+
async _transcribe(data, mimeType) {
|
|
12
|
+
if (!this.config.apiKey) {
|
|
13
|
+
return failure("No Google API key provided. Set apiKey.google or GEMINI_API_KEY.");
|
|
14
|
+
}
|
|
15
|
+
if (this.config.timestampGranularity !== undefined) {
|
|
16
|
+
return failure("Gemini transcription does not support timestampGranularity.");
|
|
17
|
+
}
|
|
18
|
+
let instruction = "Transcribe the following audio verbatim. Output only the transcript text, with no commentary.";
|
|
19
|
+
if (this.config.language) {
|
|
20
|
+
instruction += ` The audio is in ${this.config.language}.`;
|
|
21
|
+
}
|
|
22
|
+
if (this.config.prompt) {
|
|
23
|
+
instruction += ` ${this.config.prompt}`;
|
|
24
|
+
}
|
|
25
|
+
const base64 = Buffer.from(data).toString("base64");
|
|
26
|
+
const request = {
|
|
27
|
+
model: this.config.model,
|
|
28
|
+
contents: [{
|
|
29
|
+
role: "user",
|
|
30
|
+
parts: [
|
|
31
|
+
{ inlineData: { mimeType: googleAudioWireMime(mimeType), data: base64 } },
|
|
32
|
+
{ text: instruction },
|
|
33
|
+
],
|
|
34
|
+
}],
|
|
35
|
+
};
|
|
36
|
+
if (Buffer.byteLength(JSON.stringify(request), "utf8") > GOOGLE_INLINE_REQUEST_MAX_BYTES) {
|
|
37
|
+
return failure("Audio and instructions exceed Gemini's 20 MB inline request limit; use a smaller source.");
|
|
38
|
+
}
|
|
39
|
+
const ai = new GoogleGenAI({ apiKey: this.config.apiKey });
|
|
40
|
+
// Signal is passed at call time (not part of `request`) so the encoded-size
|
|
41
|
+
// check above measures only the payload. Gemini's abortSignal is client-only:
|
|
42
|
+
// it tears down the request but does not stop server-side billing.
|
|
43
|
+
const res = await ai.models.generateContent({
|
|
44
|
+
...request,
|
|
45
|
+
config: { abortSignal: this.config.abortSignal },
|
|
46
|
+
});
|
|
47
|
+
const usage = normalizeGoogleAudioUsage(res.usageMetadata, "input");
|
|
48
|
+
const result = {
|
|
49
|
+
text: res.text ?? "",
|
|
50
|
+
raw: res,
|
|
51
|
+
};
|
|
52
|
+
if (usage)
|
|
53
|
+
result.usage = usage;
|
|
54
|
+
return success(result);
|
|
55
|
+
}
|
|
56
|
+
}
|
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
import OpenAI from "openai";
|
|
2
|
+
import { OpenAITranscriptionClient } from "./openai.js";
|
|
3
|
+
/**
|
|
4
|
+
* Groq exposes an OpenAI-compatible /audio/transcriptions endpoint (Whisper
|
|
5
|
+
* large-v3 / large-v3-turbo). Everything but the base URL is inherited.
|
|
6
|
+
*/
|
|
7
|
+
export declare class GroqTranscriptionClient extends OpenAITranscriptionClient {
|
|
8
|
+
protected makeClient(): OpenAI;
|
|
9
|
+
protected noKeyMessage(): string;
|
|
10
|
+
}
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
import OpenAI from "openai";
|
|
2
|
+
import { OpenAITranscriptionClient } from "./openai.js";
|
|
3
|
+
/**
|
|
4
|
+
* Groq exposes an OpenAI-compatible /audio/transcriptions endpoint (Whisper
|
|
5
|
+
* large-v3 / large-v3-turbo). Everything but the base URL is inherited.
|
|
6
|
+
*/
|
|
7
|
+
export class GroqTranscriptionClient extends OpenAITranscriptionClient {
|
|
8
|
+
makeClient() {
|
|
9
|
+
return new OpenAI({
|
|
10
|
+
apiKey: this.config.apiKey,
|
|
11
|
+
baseURL: "https://api.groq.com/openai/v1",
|
|
12
|
+
});
|
|
13
|
+
}
|
|
14
|
+
noKeyMessage() {
|
|
15
|
+
return "No Groq API key provided. Set apiKey.groq or GROQ_API_KEY.";
|
|
16
|
+
}
|
|
17
|
+
}
|
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
import OpenAI from "openai";
|
|
2
|
+
import { Result } from "../types/result.js";
|
|
3
|
+
import { BaseTranscriptionClient } from "./baseTranscriptionClient.js";
|
|
4
|
+
import type { TranscriptionResult } from "../transcription.js";
|
|
5
|
+
export declare class OpenAITranscriptionClient extends BaseTranscriptionClient {
|
|
6
|
+
/** Build the OpenAI SDK client. Subclasses override to point at a compatible base URL. */
|
|
7
|
+
protected makeClient(): OpenAI;
|
|
8
|
+
/** Provider-specific diagnostic when no API key is resolved. Subclasses override. */
|
|
9
|
+
protected noKeyMessage(): string;
|
|
10
|
+
protected _transcribe(data: Uint8Array, mimeType: string): Promise<Result<TranscriptionResult>>;
|
|
11
|
+
}
|
|
@@ -0,0 +1,67 @@
|
|
|
1
|
+
import OpenAI, { toFile } from "openai";
|
|
2
|
+
import { success, failure } from "../types/result.js";
|
|
3
|
+
import { transcriptionAudioType } from "../util/audioMime.js";
|
|
4
|
+
import { BaseTranscriptionClient } from "./baseTranscriptionClient.js";
|
|
5
|
+
export class OpenAITranscriptionClient extends BaseTranscriptionClient {
|
|
6
|
+
/** Build the OpenAI SDK client. Subclasses override to point at a compatible base URL. */
|
|
7
|
+
makeClient() {
|
|
8
|
+
return new OpenAI({ apiKey: this.config.apiKey });
|
|
9
|
+
}
|
|
10
|
+
/** Provider-specific diagnostic when no API key is resolved. Subclasses override. */
|
|
11
|
+
noKeyMessage() {
|
|
12
|
+
return "No OpenAI API key provided. Set apiKey.openAi or OPENAI_API_KEY.";
|
|
13
|
+
}
|
|
14
|
+
// No try/catch here: BaseTranscriptionClient.transcribe() is the single
|
|
15
|
+
// redacting/logging exception boundary.
|
|
16
|
+
async _transcribe(data, mimeType) {
|
|
17
|
+
if (!this.config.apiKey) {
|
|
18
|
+
return failure(this.noKeyMessage());
|
|
19
|
+
}
|
|
20
|
+
// Filename is an OpenAI upload detail, not part of the provider-neutral
|
|
21
|
+
// operation contract. Derive the synthetic name from the normalized MIME.
|
|
22
|
+
const filename = transcriptionAudioType(mimeType)?.filename ?? "audio.bin";
|
|
23
|
+
const client = this.makeClient();
|
|
24
|
+
const file = await toFile(data, filename, { type: mimeType });
|
|
25
|
+
const granularities = [];
|
|
26
|
+
if (this.config.timestampGranularity) {
|
|
27
|
+
granularities.push(this.config.timestampGranularity);
|
|
28
|
+
}
|
|
29
|
+
const requestBody = {
|
|
30
|
+
file,
|
|
31
|
+
model: this.config.model,
|
|
32
|
+
response_format: "verbose_json",
|
|
33
|
+
};
|
|
34
|
+
if (this.config.language) {
|
|
35
|
+
requestBody.language = this.config.language;
|
|
36
|
+
}
|
|
37
|
+
if (this.config.prompt) {
|
|
38
|
+
requestBody.prompt = this.config.prompt;
|
|
39
|
+
}
|
|
40
|
+
if (granularities.length > 0) {
|
|
41
|
+
requestBody.timestamp_granularities = granularities;
|
|
42
|
+
}
|
|
43
|
+
const res = (await client.audio.transcriptions.create(requestBody, { signal: this.config.abortSignal }));
|
|
44
|
+
const result = { text: res.text, raw: res };
|
|
45
|
+
if (res.language) {
|
|
46
|
+
result.language = res.language;
|
|
47
|
+
}
|
|
48
|
+
if (typeof res.duration === "number") {
|
|
49
|
+
result.durationSeconds = res.duration;
|
|
50
|
+
}
|
|
51
|
+
if (Array.isArray(res.segments)) {
|
|
52
|
+
result.segments = res.segments.map((segment) => ({
|
|
53
|
+
start: segment.start,
|
|
54
|
+
end: segment.end,
|
|
55
|
+
text: segment.text,
|
|
56
|
+
}));
|
|
57
|
+
}
|
|
58
|
+
if (Array.isArray(res.words)) {
|
|
59
|
+
result.words = res.words.map((word) => ({
|
|
60
|
+
start: word.start,
|
|
61
|
+
end: word.end,
|
|
62
|
+
word: word.word,
|
|
63
|
+
}));
|
|
64
|
+
}
|
|
65
|
+
return success(result);
|
|
66
|
+
}
|
|
67
|
+
}
|
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
import OpenAI from "openai";
|
|
2
|
+
import { OpenAITranscriptionClient } from "./openai.js";
|
|
3
|
+
/**
|
|
4
|
+
* Generic OpenAI-compatible transcription client. Point it at any provider
|
|
5
|
+
* exposing an OpenAI-shaped /audio/transcriptions endpoint via
|
|
6
|
+
* `config.baseUrl.openAiCompat` (or OPENAI_COMPAT_BASE_URL) and
|
|
7
|
+
* `config.apiKey.openAiCompat` (or OPENAI_COMPAT_API_KEY). Mirrors the chat
|
|
8
|
+
* `SmolOpenAiCompat` client.
|
|
9
|
+
*/
|
|
10
|
+
export declare class OpenAiCompatTranscriptionClient extends OpenAITranscriptionClient {
|
|
11
|
+
protected makeClient(): OpenAI;
|
|
12
|
+
protected noKeyMessage(): string;
|
|
13
|
+
}
|
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
import OpenAI from "openai";
|
|
2
|
+
import { OpenAITranscriptionClient } from "./openai.js";
|
|
3
|
+
import { resolveBaseUrl } from "../util/provider.js";
|
|
4
|
+
/**
|
|
5
|
+
* Generic OpenAI-compatible transcription client. Point it at any provider
|
|
6
|
+
* exposing an OpenAI-shaped /audio/transcriptions endpoint via
|
|
7
|
+
* `config.baseUrl.openAiCompat` (or OPENAI_COMPAT_BASE_URL) and
|
|
8
|
+
* `config.apiKey.openAiCompat` (or OPENAI_COMPAT_API_KEY). Mirrors the chat
|
|
9
|
+
* `SmolOpenAiCompat` client.
|
|
10
|
+
*/
|
|
11
|
+
export class OpenAiCompatTranscriptionClient extends OpenAITranscriptionClient {
|
|
12
|
+
makeClient() {
|
|
13
|
+
const baseURL = resolveBaseUrl("openai-compat", { baseUrl: this.config.baseUrl });
|
|
14
|
+
if (!baseURL) {
|
|
15
|
+
throw new Error("openai-compat: base URL required (config.baseUrl.openAiCompat or OPENAI_COMPAT_BASE_URL).");
|
|
16
|
+
}
|
|
17
|
+
return new OpenAI({ apiKey: this.config.apiKey, baseURL });
|
|
18
|
+
}
|
|
19
|
+
noKeyMessage() {
|
|
20
|
+
return "No API key provided. Set apiKey.openAiCompat or OPENAI_COMPAT_API_KEY.";
|
|
21
|
+
}
|
|
22
|
+
}
|
|
@@ -0,0 +1,54 @@
|
|
|
1
|
+
import type { ModelDataBlob } from "./modelData.js";
|
|
2
|
+
import type { SmolConfig } from "./types.js";
|
|
3
|
+
import { Result } from "./types/result.js";
|
|
4
|
+
import { TokenUsage } from "./types/tokenUsage.js";
|
|
5
|
+
import { CostEstimate } from "./types/costEstimate.js";
|
|
6
|
+
import { BlobRef } from "./util/blobRef.js";
|
|
7
|
+
import { BaseTranscriptionClient, TranscriptionClientConfig } from "./transcription/baseTranscriptionClient.js";
|
|
8
|
+
export { DEFAULT_TRANSCRIBE_BYTES } from "./transcription/baseTranscriptionClient.js";
|
|
9
|
+
export type TranscribeOptions = {
|
|
10
|
+
model: string;
|
|
11
|
+
provider?: string;
|
|
12
|
+
modelData?: ModelDataBlob;
|
|
13
|
+
apiKey?: SmolConfig["apiKey"];
|
|
14
|
+
baseUrl?: SmolConfig["baseUrl"];
|
|
15
|
+
language?: string;
|
|
16
|
+
prompt?: string;
|
|
17
|
+
timestampGranularity?: "segment" | "word";
|
|
18
|
+
maxBytes?: number;
|
|
19
|
+
metadata?: Record<string, unknown>;
|
|
20
|
+
/** Abort the in-flight provider request when this signal fires. */
|
|
21
|
+
abortSignal?: AbortSignal;
|
|
22
|
+
};
|
|
23
|
+
export type TranscriptionSegment = {
|
|
24
|
+
start: number;
|
|
25
|
+
end: number;
|
|
26
|
+
text: string;
|
|
27
|
+
};
|
|
28
|
+
export type TranscriptionWord = {
|
|
29
|
+
start: number;
|
|
30
|
+
end: number;
|
|
31
|
+
word: string;
|
|
32
|
+
};
|
|
33
|
+
export type TranscriptionResult = {
|
|
34
|
+
text: string;
|
|
35
|
+
language?: string;
|
|
36
|
+
durationSeconds?: number;
|
|
37
|
+
segments?: TranscriptionSegment[];
|
|
38
|
+
words?: TranscriptionWord[];
|
|
39
|
+
usage?: TokenUsage;
|
|
40
|
+
cost?: CostEstimate;
|
|
41
|
+
raw?: unknown;
|
|
42
|
+
};
|
|
43
|
+
export type TranscriptionClientClass = new (config: TranscriptionClientConfig) => BaseTranscriptionClient;
|
|
44
|
+
export declare function registerTranscriptionProvider(name: string, cls: TranscriptionClientClass): void;
|
|
45
|
+
/** Test-only: clear all registered custom providers so registrations don't leak across tests. */
|
|
46
|
+
export declare function _resetForTests(): void;
|
|
47
|
+
/**
|
|
48
|
+
* Resolve provider + API key and instantiate the matching transcription client
|
|
49
|
+
* for the declarative transcribe() operation. Never throws: a custom client
|
|
50
|
+
* class's constructor can throw, and this internal factory's catch redacts the
|
|
51
|
+
* resolved key so a constructor error cannot leak through the public wrapper.
|
|
52
|
+
*/
|
|
53
|
+
export declare function getTranscriptionClient(opts: TranscribeOptions): Result<BaseTranscriptionClient>;
|
|
54
|
+
export declare function transcribe(source: BlobRef, opts: TranscribeOptions): Promise<Result<TranscriptionResult>>;
|
|
@@ -0,0 +1,64 @@
|
|
|
1
|
+
import { success, failure } from "./types/result.js";
|
|
2
|
+
import { redactSecret } from "./util/redact.js";
|
|
3
|
+
import { getLogger } from "./util/logger.js";
|
|
4
|
+
import { resolveProvider, resolveApiKey } from "./util/provider.js";
|
|
5
|
+
import { OpenAITranscriptionClient } from "./transcription/openai.js";
|
|
6
|
+
import { GroqTranscriptionClient } from "./transcription/groq.js";
|
|
7
|
+
import { GoogleTranscriptionClient } from "./transcription/google.js";
|
|
8
|
+
import { OpenAiCompatTranscriptionClient } from "./transcription/openaiCompat.js";
|
|
9
|
+
export { DEFAULT_TRANSCRIBE_BYTES } from "./transcription/baseTranscriptionClient.js";
|
|
10
|
+
// Checked before the user registry so a registered "openai" can't hijack the built-in.
|
|
11
|
+
const builtinClients = Object.create(null);
|
|
12
|
+
builtinClients["openai"] = OpenAITranscriptionClient;
|
|
13
|
+
builtinClients["groq"] = GroqTranscriptionClient;
|
|
14
|
+
builtinClients["google"] = GoogleTranscriptionClient;
|
|
15
|
+
builtinClients["openai-compat"] = OpenAiCompatTranscriptionClient;
|
|
16
|
+
// Null-prototype so provider names like "toString"/"__proto__" can't collide
|
|
17
|
+
// with Object.prototype or pollute the registry.
|
|
18
|
+
const registered = Object.create(null);
|
|
19
|
+
export function registerTranscriptionProvider(name, cls) {
|
|
20
|
+
registered[name] = cls;
|
|
21
|
+
}
|
|
22
|
+
/** Test-only: clear all registered custom providers so registrations don't leak across tests. */
|
|
23
|
+
export function _resetForTests() {
|
|
24
|
+
for (const key of Object.keys(registered)) {
|
|
25
|
+
delete registered[key];
|
|
26
|
+
}
|
|
27
|
+
}
|
|
28
|
+
/**
|
|
29
|
+
* Resolve provider + API key and instantiate the matching transcription client
|
|
30
|
+
* for the declarative transcribe() operation. Never throws: a custom client
|
|
31
|
+
* class's constructor can throw, and this internal factory's catch redacts the
|
|
32
|
+
* resolved key so a constructor error cannot leak through the public wrapper.
|
|
33
|
+
*/
|
|
34
|
+
export function getTranscriptionClient(opts) {
|
|
35
|
+
let apiKeyForRedaction = "";
|
|
36
|
+
try {
|
|
37
|
+
const provider = resolveProvider(opts.model, opts.provider, opts.modelData);
|
|
38
|
+
const ClientClass = builtinClients[provider] ?? registered[provider];
|
|
39
|
+
if (ClientClass === undefined) {
|
|
40
|
+
return failure(`Provider "${provider}" has no transcription API. Register one with registerTranscriptionProvider(name, ClientClass).`);
|
|
41
|
+
}
|
|
42
|
+
const apiKey = resolveApiKey(provider, opts) ?? "";
|
|
43
|
+
apiKeyForRedaction = apiKey;
|
|
44
|
+
const { apiKey: _callerKeys, ...clientOpts } = opts;
|
|
45
|
+
const config = { ...clientOpts, provider, apiKey };
|
|
46
|
+
return success(new ClientClass(config));
|
|
47
|
+
}
|
|
48
|
+
catch (err) {
|
|
49
|
+
let msg = "getTranscriptionClient() failed";
|
|
50
|
+
if (err instanceof Error) {
|
|
51
|
+
msg = err.message;
|
|
52
|
+
}
|
|
53
|
+
const redacted = redactSecret(msg, apiKeyForRedaction);
|
|
54
|
+
getLogger().error("getTranscriptionClient() failed:", redacted);
|
|
55
|
+
return failure(redacted);
|
|
56
|
+
}
|
|
57
|
+
}
|
|
58
|
+
export async function transcribe(source, opts) {
|
|
59
|
+
const client = getTranscriptionClient(opts);
|
|
60
|
+
if (!client.success) {
|
|
61
|
+
return client;
|
|
62
|
+
}
|
|
63
|
+
return client.value.transcribe(source);
|
|
64
|
+
}
|
|
@@ -4,6 +4,8 @@ export type TokenUsage = {
|
|
|
4
4
|
outputTokens: number;
|
|
5
5
|
cachedInputTokens?: number;
|
|
6
6
|
cacheCreationInputTokens?: number;
|
|
7
|
+
inputAudioTokens?: number;
|
|
8
|
+
outputAudioTokens?: number;
|
|
7
9
|
totalTokens?: number;
|
|
8
10
|
};
|
|
9
11
|
export declare const TokenUsageSchema: z.ZodObject<{
|
|
@@ -11,6 +13,8 @@ export declare const TokenUsageSchema: z.ZodObject<{
|
|
|
11
13
|
outputTokens: z.ZodNumber;
|
|
12
14
|
cachedInputTokens: z.ZodOptional<z.ZodNumber>;
|
|
13
15
|
cacheCreationInputTokens: z.ZodOptional<z.ZodNumber>;
|
|
16
|
+
inputAudioTokens: z.ZodOptional<z.ZodNumber>;
|
|
17
|
+
outputAudioTokens: z.ZodOptional<z.ZodNumber>;
|
|
14
18
|
totalTokens: z.ZodOptional<z.ZodNumber>;
|
|
15
19
|
}, z.core.$strip>;
|
|
16
20
|
export declare function addTokenUsage(_a?: TokenUsage, _b?: TokenUsage): TokenUsage;
|