smoltalk 0.10.0 → 0.11.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +74 -6
- package/dist/client.d.ts +6 -0
- package/dist/client.js +12 -0
- package/dist/clients/llamaCppLoader.d.ts +42 -0
- package/dist/clients/llamaCppLoader.js +109 -0
- package/dist/functions.js +8 -1
- package/dist/index.d.ts +2 -0
- package/dist/index.js +2 -0
- package/dist/model.js +30 -11
- package/dist/models.d.ts +61 -3
- package/dist/models.js +74 -0
- package/dist/speech/baseSpeechClient.d.ts +5 -0
- package/dist/speech/baseSpeechClient.js +21 -2
- package/dist/speech/google.d.ts +6 -0
- package/dist/speech/google.js +54 -0
- package/dist/speech/groq.d.ts +11 -0
- package/dist/speech/groq.js +19 -0
- package/dist/speech/openai.d.ts +8 -0
- package/dist/speech/openai.js +16 -4
- package/dist/speech/openaiCompat.d.ts +13 -0
- package/dist/speech/openaiCompat.js +22 -0
- package/dist/speech.d.ts +5 -0
- package/dist/speech.js +6 -0
- package/dist/transcription/baseTranscriptionClient.d.ts +5 -0
- package/dist/transcription/baseTranscriptionClient.js +44 -18
- package/dist/transcription/google.d.ts +6 -0
- package/dist/transcription/google.js +56 -0
- package/dist/transcription/groq.d.ts +10 -0
- package/dist/transcription/groq.js +17 -0
- package/dist/transcription/openai.d.ts +5 -0
- package/dist/transcription/openai.js +11 -3
- package/dist/transcription/openaiCompat.d.ts +13 -0
- package/dist/transcription/openaiCompat.js +22 -0
- package/dist/transcription.d.ts +3 -0
- package/dist/transcription.js +6 -0
- package/dist/types.d.ts +1 -0
- package/dist/util/audioMime.d.ts +17 -0
- package/dist/util/audioMime.js +42 -1
- package/dist/util/googleAudioUsage.d.ts +14 -0
- package/dist/util/googleAudioUsage.js +52 -0
- package/dist/util/mime.js +2 -0
- package/dist/util/provider.d.ts +1 -0
- package/dist/util/provider.js +2 -0
- package/package.json +9 -1
|
@@ -1,4 +1,5 @@
|
|
|
1
1
|
import type { ModelDataBlob } from "../modelData.js";
|
|
2
|
+
import type { SmolConfig } from "../types.js";
|
|
2
3
|
import { Result } from "../types/result.js";
|
|
3
4
|
import type { SpeechResult } from "../speech.js";
|
|
4
5
|
export type SpeechClientConfig = {
|
|
@@ -7,12 +8,16 @@ export type SpeechClientConfig = {
|
|
|
7
8
|
provider: string;
|
|
8
9
|
/** Resolved API key; empty string when none was found. */
|
|
9
10
|
apiKey: string;
|
|
11
|
+
/** Base-URL map (for OpenAI-compatible providers); read via resolveBaseUrl. */
|
|
12
|
+
baseUrl?: SmolConfig["baseUrl"];
|
|
10
13
|
voice: string;
|
|
11
14
|
modelData?: ModelDataBlob;
|
|
12
15
|
/** Output format; provider-specific vocabulary (OpenAI: mp3/opus/aac/flac/wav/pcm). */
|
|
13
16
|
format?: string;
|
|
14
17
|
speed?: number;
|
|
15
18
|
metadata?: Record<string, unknown>;
|
|
19
|
+
/** Abort the in-flight provider request when this signal fires. */
|
|
20
|
+
abortSignal?: AbortSignal;
|
|
16
21
|
};
|
|
17
22
|
/**
|
|
18
23
|
* Shared TTS behavior, mirroring BaseClient for text generation: the public
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import { getModelForProvider, isTextToSpeechModel, } from "../models.js";
|
|
2
|
-
import { calculateSpeechCost } from "../model.js";
|
|
2
|
+
import { Model, calculateSpeechCost } from "../model.js";
|
|
3
3
|
import { failure } from "../types/result.js";
|
|
4
4
|
import { redactSecret } from "../util/redact.js";
|
|
5
5
|
import { getLogger } from "../util/logger.js";
|
|
@@ -49,6 +49,10 @@ export class BaseSpeechClient {
|
|
|
49
49
|
this.config = config;
|
|
50
50
|
}
|
|
51
51
|
async speak(text) {
|
|
52
|
+
// Already-aborted signal: stop before doing any paid work.
|
|
53
|
+
if (this.config.abortSignal?.aborted) {
|
|
54
|
+
return failure("Request was aborted");
|
|
55
|
+
}
|
|
52
56
|
try {
|
|
53
57
|
const model = getModelForProvider(this.config.provider, this.config.model, this.config.modelData);
|
|
54
58
|
if (model !== undefined && !isTextToSpeechModel(model)) {
|
|
@@ -75,17 +79,32 @@ export class BaseSpeechClient {
|
|
|
75
79
|
`Supported: ${model.formats.join(", ")}.`);
|
|
76
80
|
}
|
|
77
81
|
}
|
|
82
|
+
// Re-check after preflight validation: the signal may have fired during
|
|
83
|
+
// it, and we must not dispatch a request once cancelled.
|
|
84
|
+
if (this.config.abortSignal?.aborted) {
|
|
85
|
+
return failure("Request was aborted");
|
|
86
|
+
}
|
|
78
87
|
const result = await this._speak(text);
|
|
79
88
|
if (!result.success) {
|
|
80
89
|
return result;
|
|
81
90
|
}
|
|
82
|
-
|
|
91
|
+
let cost = calculateSpeechCost(model, [...text].length);
|
|
92
|
+
if (cost === undefined && result.value.usage !== undefined) {
|
|
93
|
+
// Token-billed providers (Gemini) price through the shared cost engine.
|
|
94
|
+
cost =
|
|
95
|
+
new Model(this.config.model, this.config.provider, this.config.modelData).calculateCost(result.value.usage) ?? undefined;
|
|
96
|
+
}
|
|
83
97
|
if (cost !== undefined) {
|
|
84
98
|
result.value.cost = cost;
|
|
85
99
|
}
|
|
86
100
|
return result;
|
|
87
101
|
}
|
|
88
102
|
catch (err) {
|
|
103
|
+
// Caller-initiated cancellation surfaces as a distinguishable failure
|
|
104
|
+
// (matching the chat path), not a redacted provider error.
|
|
105
|
+
if (this.config.abortSignal?.aborted) {
|
|
106
|
+
return failure("Request was aborted");
|
|
107
|
+
}
|
|
89
108
|
let msg = "speak() failed";
|
|
90
109
|
if (err instanceof Error) {
|
|
91
110
|
msg = err.message;
|
|
@@ -0,0 +1,6 @@
|
|
|
1
|
+
import { Result } from "../types/result.js";
|
|
2
|
+
import { BaseSpeechClient } from "./baseSpeechClient.js";
|
|
3
|
+
import type { SpeechResult } from "../speech.js";
|
|
4
|
+
export declare class GoogleSpeechClient extends BaseSpeechClient {
|
|
5
|
+
protected _speak(text: string): Promise<Result<SpeechResult>>;
|
|
6
|
+
}
|
|
@@ -0,0 +1,54 @@
|
|
|
1
|
+
import { GoogleGenAI } from "@google/genai";
|
|
2
|
+
import { success, failure } from "../types/result.js";
|
|
3
|
+
import { pcmToWav } from "../util/audioMime.js";
|
|
4
|
+
import { normalizeGoogleAudioUsage } from "../util/googleAudioUsage.js";
|
|
5
|
+
import { BaseSpeechClient } from "./baseSpeechClient.js";
|
|
6
|
+
const GEMINI_PCM = { sampleRateHz: 24000, sampleFormat: "s16le", channels: 1 };
|
|
7
|
+
export class GoogleSpeechClient extends BaseSpeechClient {
|
|
8
|
+
// No try/catch: BaseSpeechClient.speak() is the exception boundary.
|
|
9
|
+
async _speak(text) {
|
|
10
|
+
if (!this.config.apiKey) {
|
|
11
|
+
return failure("No Google API key provided. Set apiKey.google or GEMINI_API_KEY.");
|
|
12
|
+
}
|
|
13
|
+
// Gemini controls pacing via prompt style, not a numeric speed parameter.
|
|
14
|
+
if (this.config.speed !== undefined) {
|
|
15
|
+
return failure("Gemini TTS does not support the 'speed' option; control pacing via the prompt text.");
|
|
16
|
+
}
|
|
17
|
+
const format = this.config.format ?? "pcm";
|
|
18
|
+
if (format !== "pcm" && format !== "wav") {
|
|
19
|
+
return failure(`Gemini TTS only produces raw PCM. Supported formats: pcm (default), wav. Got "${format}".`);
|
|
20
|
+
}
|
|
21
|
+
const ai = new GoogleGenAI({ apiKey: this.config.apiKey });
|
|
22
|
+
const res = await ai.models.generateContent({
|
|
23
|
+
model: this.config.model,
|
|
24
|
+
contents: [{ role: "user", parts: [{ text }] }],
|
|
25
|
+
config: {
|
|
26
|
+
responseModalities: ["AUDIO"],
|
|
27
|
+
speechConfig: {
|
|
28
|
+
voiceConfig: { prebuiltVoiceConfig: { voiceName: this.config.voice } },
|
|
29
|
+
},
|
|
30
|
+
// Client-only cancellation: tears down the request, but Gemini still
|
|
31
|
+
// bills server-side work.
|
|
32
|
+
abortSignal: this.config.abortSignal,
|
|
33
|
+
},
|
|
34
|
+
});
|
|
35
|
+
const dataB64 = res.candidates?.[0]?.content?.parts?.find((part) => part.inlineData?.data !== undefined)?.inlineData?.data;
|
|
36
|
+
if (!dataB64) {
|
|
37
|
+
return failure("Gemini returned no audio data.");
|
|
38
|
+
}
|
|
39
|
+
const pcm = new Uint8Array(Buffer.from(dataB64, "base64"));
|
|
40
|
+
let audio = pcm;
|
|
41
|
+
let mimeType = "application/octet-stream";
|
|
42
|
+
if (format === "wav") {
|
|
43
|
+
audio = pcmToWav(pcm, { sampleRateHz: 24000, channels: 1, bitsPerSample: 16 });
|
|
44
|
+
mimeType = "audio/wav";
|
|
45
|
+
}
|
|
46
|
+
const usage = normalizeGoogleAudioUsage(res.usageMetadata, "output");
|
|
47
|
+
const result = { audio, mimeType, raw: res };
|
|
48
|
+
if (format === "pcm")
|
|
49
|
+
result.pcm = GEMINI_PCM;
|
|
50
|
+
if (usage)
|
|
51
|
+
result.usage = usage;
|
|
52
|
+
return success(result);
|
|
53
|
+
}
|
|
54
|
+
}
|
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
import OpenAI from "openai";
|
|
2
|
+
import { OpenAISpeechClient } from "./openai.js";
|
|
3
|
+
import type { SpeakFormat } from "../util/audioMime.js";
|
|
4
|
+
/**
|
|
5
|
+
* Groq exposes OpenAI-compatible Orpheus TTS and supports WAV only.
|
|
6
|
+
*/
|
|
7
|
+
export declare class GroqSpeechClient extends OpenAISpeechClient {
|
|
8
|
+
protected makeClient(): OpenAI;
|
|
9
|
+
protected defaultFormat(): SpeakFormat;
|
|
10
|
+
protected noKeyMessage(): string;
|
|
11
|
+
}
|
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
import OpenAI from "openai";
|
|
2
|
+
import { OpenAISpeechClient } from "./openai.js";
|
|
3
|
+
/**
|
|
4
|
+
* Groq exposes OpenAI-compatible Orpheus TTS and supports WAV only.
|
|
5
|
+
*/
|
|
6
|
+
export class GroqSpeechClient extends OpenAISpeechClient {
|
|
7
|
+
makeClient() {
|
|
8
|
+
return new OpenAI({
|
|
9
|
+
apiKey: this.config.apiKey,
|
|
10
|
+
baseURL: "https://api.groq.com/openai/v1",
|
|
11
|
+
});
|
|
12
|
+
}
|
|
13
|
+
defaultFormat() {
|
|
14
|
+
return "wav";
|
|
15
|
+
}
|
|
16
|
+
noKeyMessage() {
|
|
17
|
+
return "No Groq API key provided. Set apiKey.groq or GROQ_API_KEY.";
|
|
18
|
+
}
|
|
19
|
+
}
|
package/dist/speech/openai.d.ts
CHANGED
|
@@ -1,6 +1,14 @@
|
|
|
1
|
+
import OpenAI from "openai";
|
|
1
2
|
import { Result } from "../types/result.js";
|
|
3
|
+
import { type SpeakFormat } from "../util/audioMime.js";
|
|
2
4
|
import { BaseSpeechClient } from "./baseSpeechClient.js";
|
|
3
5
|
import type { SpeechResult } from "../speech.js";
|
|
4
6
|
export declare class OpenAISpeechClient extends BaseSpeechClient {
|
|
7
|
+
/** Build the OpenAI SDK client. Subclasses override to point at a compatible base URL. */
|
|
8
|
+
protected makeClient(): OpenAI;
|
|
9
|
+
/** Provider default used when the declarative call omits format. */
|
|
10
|
+
protected defaultFormat(): SpeakFormat;
|
|
11
|
+
/** Provider-specific diagnostic when no API key is resolved. Subclasses override. */
|
|
12
|
+
protected noKeyMessage(): string;
|
|
5
13
|
protected _speak(text: string): Promise<Result<SpeechResult>>;
|
|
6
14
|
}
|
package/dist/speech/openai.js
CHANGED
|
@@ -3,22 +3,34 @@ import { success, failure } from "../types/result.js";
|
|
|
3
3
|
import { SPEECH_FORMAT_TO_MIME, isSpeakFormat, } from "../util/audioMime.js";
|
|
4
4
|
import { BaseSpeechClient } from "./baseSpeechClient.js";
|
|
5
5
|
export class OpenAISpeechClient extends BaseSpeechClient {
|
|
6
|
+
/** Build the OpenAI SDK client. Subclasses override to point at a compatible base URL. */
|
|
7
|
+
makeClient() {
|
|
8
|
+
return new OpenAI({ apiKey: this.config.apiKey });
|
|
9
|
+
}
|
|
10
|
+
/** Provider default used when the declarative call omits format. */
|
|
11
|
+
defaultFormat() {
|
|
12
|
+
return "mp3";
|
|
13
|
+
}
|
|
14
|
+
/** Provider-specific diagnostic when no API key is resolved. Subclasses override. */
|
|
15
|
+
noKeyMessage() {
|
|
16
|
+
return "No OpenAI API key provided. Set apiKey.openAi or OPENAI_API_KEY.";
|
|
17
|
+
}
|
|
6
18
|
// No try/catch here: BaseSpeechClient.speak() is the single
|
|
7
19
|
// redacting/logging exception boundary.
|
|
8
20
|
async _speak(text) {
|
|
9
21
|
if (!this.config.apiKey) {
|
|
10
|
-
return failure(
|
|
22
|
+
return failure(this.noKeyMessage());
|
|
11
23
|
}
|
|
12
24
|
// The shared contract carries format as a plain string; narrow to OpenAI's
|
|
13
25
|
// closed union at runtime before indexing the MIME table.
|
|
14
|
-
const requestedFormat = this.config.format ??
|
|
26
|
+
const requestedFormat = this.config.format ?? this.defaultFormat();
|
|
15
27
|
if (!isSpeakFormat(requestedFormat)) {
|
|
16
28
|
return failure(`Format "${requestedFormat}" is not a supported OpenAI speech format. ` +
|
|
17
29
|
`Supported: ${Object.keys(SPEECH_FORMAT_TO_MIME).join(", ")}.`);
|
|
18
30
|
}
|
|
19
31
|
const format = requestedFormat;
|
|
20
32
|
const mimeType = SPEECH_FORMAT_TO_MIME[format];
|
|
21
|
-
const client =
|
|
33
|
+
const client = this.makeClient();
|
|
22
34
|
const params = {
|
|
23
35
|
model: this.config.model,
|
|
24
36
|
voice: this.config.voice,
|
|
@@ -28,7 +40,7 @@ export class OpenAISpeechClient extends BaseSpeechClient {
|
|
|
28
40
|
if (this.config.speed !== undefined) {
|
|
29
41
|
params.speed = this.config.speed;
|
|
30
42
|
}
|
|
31
|
-
const res = await client.audio.speech.create(params);
|
|
43
|
+
const res = await client.audio.speech.create(params, { signal: this.config.abortSignal });
|
|
32
44
|
const audio = new Uint8Array(await res.arrayBuffer());
|
|
33
45
|
const result = { audio, mimeType };
|
|
34
46
|
if (format === "pcm") {
|
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
import OpenAI from "openai";
|
|
2
|
+
import { OpenAISpeechClient } from "./openai.js";
|
|
3
|
+
/**
|
|
4
|
+
* Generic OpenAI-compatible speech (TTS) client. Point it at any provider
|
|
5
|
+
* exposing an OpenAI-shaped /audio/speech endpoint via
|
|
6
|
+
* `config.baseUrl.openAiCompat` (or OPENAI_COMPAT_BASE_URL) and
|
|
7
|
+
* `config.apiKey.openAiCompat` (or OPENAI_COMPAT_API_KEY). Mirrors the chat
|
|
8
|
+
* `SmolOpenAiCompat` client. Inherits OpenAI's `mp3` default format.
|
|
9
|
+
*/
|
|
10
|
+
export declare class OpenAiCompatSpeechClient extends OpenAISpeechClient {
|
|
11
|
+
protected makeClient(): OpenAI;
|
|
12
|
+
protected noKeyMessage(): string;
|
|
13
|
+
}
|
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
import OpenAI from "openai";
|
|
2
|
+
import { OpenAISpeechClient } from "./openai.js";
|
|
3
|
+
import { resolveBaseUrl } from "../util/provider.js";
|
|
4
|
+
/**
|
|
5
|
+
* Generic OpenAI-compatible speech (TTS) client. Point it at any provider
|
|
6
|
+
* exposing an OpenAI-shaped /audio/speech endpoint via
|
|
7
|
+
* `config.baseUrl.openAiCompat` (or OPENAI_COMPAT_BASE_URL) and
|
|
8
|
+
* `config.apiKey.openAiCompat` (or OPENAI_COMPAT_API_KEY). Mirrors the chat
|
|
9
|
+
* `SmolOpenAiCompat` client. Inherits OpenAI's `mp3` default format.
|
|
10
|
+
*/
|
|
11
|
+
export class OpenAiCompatSpeechClient extends OpenAISpeechClient {
|
|
12
|
+
makeClient() {
|
|
13
|
+
const baseURL = resolveBaseUrl("openai-compat", { baseUrl: this.config.baseUrl });
|
|
14
|
+
if (!baseURL) {
|
|
15
|
+
throw new Error("openai-compat: base URL required (config.baseUrl.openAiCompat or OPENAI_COMPAT_BASE_URL).");
|
|
16
|
+
}
|
|
17
|
+
return new OpenAI({ apiKey: this.config.apiKey, baseURL });
|
|
18
|
+
}
|
|
19
|
+
noKeyMessage() {
|
|
20
|
+
return "No API key provided. Set apiKey.openAiCompat or OPENAI_COMPAT_API_KEY.";
|
|
21
|
+
}
|
|
22
|
+
}
|
package/dist/speech.d.ts
CHANGED
|
@@ -2,6 +2,7 @@ import type { ModelDataBlob } from "./modelData.js";
|
|
|
2
2
|
import type { SmolConfig } from "./types.js";
|
|
3
3
|
import { Result } from "./types/result.js";
|
|
4
4
|
import { CostEstimate } from "./types/costEstimate.js";
|
|
5
|
+
import { TokenUsage } from "./types/tokenUsage.js";
|
|
5
6
|
import { BaseSpeechClient, SpeechClientConfig } from "./speech/baseSpeechClient.js";
|
|
6
7
|
export type SpeakOptions = {
|
|
7
8
|
model: string;
|
|
@@ -9,9 +10,12 @@ export type SpeakOptions = {
|
|
|
9
10
|
provider?: string;
|
|
10
11
|
modelData?: ModelDataBlob;
|
|
11
12
|
apiKey?: SmolConfig["apiKey"];
|
|
13
|
+
baseUrl?: SmolConfig["baseUrl"];
|
|
12
14
|
format?: string;
|
|
13
15
|
speed?: number;
|
|
14
16
|
metadata?: Record<string, unknown>;
|
|
17
|
+
/** Abort the in-flight provider request when this signal fires. */
|
|
18
|
+
abortSignal?: AbortSignal;
|
|
15
19
|
};
|
|
16
20
|
export type PcmAudioMetadata = {
|
|
17
21
|
sampleRateHz: number;
|
|
@@ -23,6 +27,7 @@ export type SpeechResult = {
|
|
|
23
27
|
audio: Uint8Array;
|
|
24
28
|
mimeType: string;
|
|
25
29
|
pcm?: PcmAudioMetadata;
|
|
30
|
+
usage?: TokenUsage;
|
|
26
31
|
cost?: CostEstimate;
|
|
27
32
|
raw?: unknown;
|
|
28
33
|
};
|
package/dist/speech.js
CHANGED
|
@@ -3,9 +3,15 @@ import { redactSecret } from "./util/redact.js";
|
|
|
3
3
|
import { getLogger } from "./util/logger.js";
|
|
4
4
|
import { resolveProvider, resolveApiKey } from "./util/provider.js";
|
|
5
5
|
import { OpenAISpeechClient } from "./speech/openai.js";
|
|
6
|
+
import { GroqSpeechClient } from "./speech/groq.js";
|
|
7
|
+
import { GoogleSpeechClient } from "./speech/google.js";
|
|
8
|
+
import { OpenAiCompatSpeechClient } from "./speech/openaiCompat.js";
|
|
6
9
|
// Checked before the user registry so a registered "openai" can't hijack the built-in.
|
|
7
10
|
const builtinClients = Object.create(null);
|
|
8
11
|
builtinClients["openai"] = OpenAISpeechClient;
|
|
12
|
+
builtinClients["groq"] = GroqSpeechClient;
|
|
13
|
+
builtinClients["google"] = GoogleSpeechClient;
|
|
14
|
+
builtinClients["openai-compat"] = OpenAiCompatSpeechClient;
|
|
9
15
|
// Null-prototype so provider names like "toString"/"__proto__" can't collide
|
|
10
16
|
// with Object.prototype or pollute the registry.
|
|
11
17
|
const registered = Object.create(null);
|
|
@@ -1,4 +1,5 @@
|
|
|
1
1
|
import type { ModelDataBlob } from "../modelData.js";
|
|
2
|
+
import type { SmolConfig } from "../types.js";
|
|
2
3
|
import { Result } from "../types/result.js";
|
|
3
4
|
import { BlobRef } from "../util/blobRef.js";
|
|
4
5
|
import type { TranscriptionResult } from "../transcription.js";
|
|
@@ -9,12 +10,16 @@ export type TranscriptionClientConfig = {
|
|
|
9
10
|
provider: string;
|
|
10
11
|
/** Resolved API key; empty string when none was found. */
|
|
11
12
|
apiKey: string;
|
|
13
|
+
/** Base-URL map (for OpenAI-compatible providers); read via resolveBaseUrl. */
|
|
14
|
+
baseUrl?: SmolConfig["baseUrl"];
|
|
12
15
|
modelData?: ModelDataBlob;
|
|
13
16
|
language?: string;
|
|
14
17
|
prompt?: string;
|
|
15
18
|
timestampGranularity?: "segment" | "word";
|
|
16
19
|
maxBytes?: number;
|
|
17
20
|
metadata?: Record<string, unknown>;
|
|
21
|
+
/** Abort the in-flight provider request when this signal fires. */
|
|
22
|
+
abortSignal?: AbortSignal;
|
|
18
23
|
};
|
|
19
24
|
/**
|
|
20
25
|
* Shared transcription behavior, mirroring BaseClient for text generation:
|
|
@@ -1,5 +1,5 @@
|
|
|
1
|
-
import { getModelForProvider, isSpeechToTextModel, } from "../models.js";
|
|
2
|
-
import { calculateTranscriptionCost } from "../model.js";
|
|
1
|
+
import { getModelForProvider, isSpeechToTextModel, modelSupportsInputModality, audioInputConstraints, } from "../models.js";
|
|
2
|
+
import { Model, calculateTranscriptionCost } from "../model.js";
|
|
3
3
|
import { success, failure } from "../types/result.js";
|
|
4
4
|
import { loadBlob } from "../util/blobRef.js";
|
|
5
5
|
import { audioFormatForMime, canonicalizeMime } from "../util/mime.js";
|
|
@@ -7,24 +7,24 @@ import { redactSecret } from "../util/redact.js";
|
|
|
7
7
|
import { getLogger } from "../util/logger.js";
|
|
8
8
|
export const DEFAULT_TRANSCRIBE_BYTES = 25 * 1024 * 1024;
|
|
9
9
|
/** Validate the declarative STT constraint block once before consuming it. */
|
|
10
|
-
function transcriptionConstraintError(
|
|
11
|
-
const modelMaxBytes =
|
|
10
|
+
function transcriptionConstraintError(modelName, c) {
|
|
11
|
+
const modelMaxBytes = c.maxBytes;
|
|
12
12
|
if (modelMaxBytes !== undefined &&
|
|
13
13
|
(typeof modelMaxBytes !== "number" || !Number.isFinite(modelMaxBytes) || modelMaxBytes <= 0)) {
|
|
14
|
-
return `Model "${
|
|
14
|
+
return `Model "${modelName}" has an invalid maxBytes value.`;
|
|
15
15
|
}
|
|
16
|
-
const supportedMimeTypes =
|
|
16
|
+
const supportedMimeTypes = c.supportedMimeTypes;
|
|
17
17
|
if (supportedMimeTypes !== undefined &&
|
|
18
18
|
(!Array.isArray(supportedMimeTypes) ||
|
|
19
19
|
!supportedMimeTypes.every((mime) => typeof mime === "string"))) {
|
|
20
|
-
return `Model "${
|
|
20
|
+
return `Model "${modelName}" has invalid supportedMimeTypes.`;
|
|
21
21
|
}
|
|
22
22
|
return null;
|
|
23
23
|
}
|
|
24
24
|
// The caller's maxBytes is a safety limit; the model's maxBytes is the
|
|
25
25
|
// provider's hard cap. Take the smaller of whichever are present so a caller
|
|
26
26
|
// can tighten the limit but never bypass the provider cap.
|
|
27
|
-
function resolveTranscriptionMaxBytes(callerMaxBytes,
|
|
27
|
+
function resolveTranscriptionMaxBytes(callerMaxBytes, modelMaxBytes) {
|
|
28
28
|
if (callerMaxBytes !== undefined &&
|
|
29
29
|
(!Number.isFinite(callerMaxBytes) || callerMaxBytes <= 0)) {
|
|
30
30
|
return failure(`maxBytes must be a positive finite number (got ${callerMaxBytes}).`);
|
|
@@ -33,8 +33,8 @@ function resolveTranscriptionMaxBytes(callerMaxBytes, model) {
|
|
|
33
33
|
if (callerMaxBytes !== undefined) {
|
|
34
34
|
limits.push(callerMaxBytes);
|
|
35
35
|
}
|
|
36
|
-
if (
|
|
37
|
-
limits.push(
|
|
36
|
+
if (modelMaxBytes !== undefined) {
|
|
37
|
+
limits.push(modelMaxBytes);
|
|
38
38
|
}
|
|
39
39
|
if (limits.length === 0) {
|
|
40
40
|
return success(DEFAULT_TRANSCRIBE_BYTES);
|
|
@@ -53,18 +53,29 @@ export class BaseTranscriptionClient {
|
|
|
53
53
|
this.config = config;
|
|
54
54
|
}
|
|
55
55
|
async transcribe(source) {
|
|
56
|
+
// Already-aborted signal: stop before doing any paid work.
|
|
57
|
+
if (this.config.abortSignal?.aborted) {
|
|
58
|
+
return failure("Request was aborted");
|
|
59
|
+
}
|
|
56
60
|
try {
|
|
57
61
|
const model = getModelForProvider(this.config.provider, this.config.model, this.config.modelData);
|
|
58
|
-
|
|
59
|
-
|
|
62
|
+
// A model is a valid transcription target if it is a dedicated STT model,
|
|
63
|
+
// or a multimodal text model that accepts audio input (e.g. Gemini). An
|
|
64
|
+
// unknown model (undefined) flows through — the provider is authority.
|
|
65
|
+
const acceptsAudio = model === undefined ||
|
|
66
|
+
isSpeechToTextModel(model) ||
|
|
67
|
+
modelSupportsInputModality(this.config.model, "audio", this.config.modelData, this.config.provider) === true;
|
|
68
|
+
if (!acceptsAudio) {
|
|
69
|
+
return failure(`Model "${this.config.model}" cannot accept audio input (not a transcription model).`);
|
|
60
70
|
}
|
|
71
|
+
const constraints = model !== undefined ? audioInputConstraints(model) : {};
|
|
61
72
|
if (model !== undefined) {
|
|
62
|
-
const constraintError = transcriptionConstraintError(model);
|
|
73
|
+
const constraintError = transcriptionConstraintError(this.config.model, constraints);
|
|
63
74
|
if (constraintError !== null) {
|
|
64
75
|
return failure(constraintError);
|
|
65
76
|
}
|
|
66
77
|
}
|
|
67
|
-
const effectiveLimit = resolveTranscriptionMaxBytes(this.config.maxBytes,
|
|
78
|
+
const effectiveLimit = resolveTranscriptionMaxBytes(this.config.maxBytes, constraints.maxBytes);
|
|
68
79
|
if (!effectiveLimit.success) {
|
|
69
80
|
return effectiveLimit;
|
|
70
81
|
}
|
|
@@ -76,25 +87,40 @@ export class BaseTranscriptionClient {
|
|
|
76
87
|
return failure(`Failed to load audio for transcription: ${err.message}`);
|
|
77
88
|
}
|
|
78
89
|
const mimeType = loaded.mimeType ?? "application/octet-stream";
|
|
79
|
-
if (
|
|
90
|
+
if (constraints.supportedMimeTypes !== undefined) {
|
|
80
91
|
const audioFormat = audioFormatForMime(mimeType);
|
|
81
92
|
const normalizedMime = audioFormat?.mimeType ?? canonicalizeMime(mimeType);
|
|
82
|
-
if (!
|
|
93
|
+
if (!constraints.supportedMimeTypes.includes(normalizedMime)) {
|
|
83
94
|
return failure(`Unsupported audio type "${mimeType}" for model "${this.config.model}". ` +
|
|
84
|
-
`Supported: ${
|
|
95
|
+
`Supported: ${constraints.supportedMimeTypes.join(", ")}.`);
|
|
85
96
|
}
|
|
86
97
|
}
|
|
98
|
+
// Re-check after the async preflight (blob load / MIME validation): the
|
|
99
|
+
// signal may have fired during it, and we must not dispatch a request.
|
|
100
|
+
if (this.config.abortSignal?.aborted) {
|
|
101
|
+
return failure("Request was aborted");
|
|
102
|
+
}
|
|
87
103
|
const result = await this._transcribe(loaded.data, mimeType);
|
|
88
104
|
if (!result.success) {
|
|
89
105
|
return result;
|
|
90
106
|
}
|
|
91
|
-
|
|
107
|
+
let cost = calculateTranscriptionCost(model, result.value.durationSeconds);
|
|
108
|
+
if (cost === undefined && result.value.usage !== undefined) {
|
|
109
|
+
// Token-billed providers (Gemini) price through the shared cost engine.
|
|
110
|
+
cost =
|
|
111
|
+
new Model(this.config.model, this.config.provider, this.config.modelData).calculateCost(result.value.usage) ?? undefined;
|
|
112
|
+
}
|
|
92
113
|
if (cost !== undefined) {
|
|
93
114
|
result.value.cost = cost;
|
|
94
115
|
}
|
|
95
116
|
return result;
|
|
96
117
|
}
|
|
97
118
|
catch (err) {
|
|
119
|
+
// Caller-initiated cancellation surfaces as a distinguishable failure
|
|
120
|
+
// (matching the chat path), not a redacted provider error.
|
|
121
|
+
if (this.config.abortSignal?.aborted) {
|
|
122
|
+
return failure("Request was aborted");
|
|
123
|
+
}
|
|
98
124
|
let msg = "transcribe() failed";
|
|
99
125
|
if (err instanceof Error) {
|
|
100
126
|
msg = err.message;
|
|
@@ -0,0 +1,6 @@
|
|
|
1
|
+
import { Result } from "../types/result.js";
|
|
2
|
+
import { BaseTranscriptionClient } from "./baseTranscriptionClient.js";
|
|
3
|
+
import type { TranscriptionResult } from "../transcription.js";
|
|
4
|
+
export declare class GoogleTranscriptionClient extends BaseTranscriptionClient {
|
|
5
|
+
protected _transcribe(data: Uint8Array, mimeType: string): Promise<Result<TranscriptionResult>>;
|
|
6
|
+
}
|
|
@@ -0,0 +1,56 @@
|
|
|
1
|
+
import { GoogleGenAI } from "@google/genai";
|
|
2
|
+
import { success, failure } from "../types/result.js";
|
|
3
|
+
import { BaseTranscriptionClient } from "./baseTranscriptionClient.js";
|
|
4
|
+
import { normalizeGoogleAudioUsage } from "../util/googleAudioUsage.js";
|
|
5
|
+
import { googleAudioWireMime } from "../util/audioMime.js";
|
|
6
|
+
// Gemini's inline request cap is 20 MB for the *entire* request (audio bytes,
|
|
7
|
+
// base64 expansion, prompt, and SDK envelope), not the raw file alone.
|
|
8
|
+
const GOOGLE_INLINE_REQUEST_MAX_BYTES = 20_000_000;
|
|
9
|
+
export class GoogleTranscriptionClient extends BaseTranscriptionClient {
|
|
10
|
+
// No try/catch: BaseTranscriptionClient.transcribe() is the exception boundary.
|
|
11
|
+
async _transcribe(data, mimeType) {
|
|
12
|
+
if (!this.config.apiKey) {
|
|
13
|
+
return failure("No Google API key provided. Set apiKey.google or GEMINI_API_KEY.");
|
|
14
|
+
}
|
|
15
|
+
if (this.config.timestampGranularity !== undefined) {
|
|
16
|
+
return failure("Gemini transcription does not support timestampGranularity.");
|
|
17
|
+
}
|
|
18
|
+
let instruction = "Transcribe the following audio verbatim. Output only the transcript text, with no commentary.";
|
|
19
|
+
if (this.config.language) {
|
|
20
|
+
instruction += ` The audio is in ${this.config.language}.`;
|
|
21
|
+
}
|
|
22
|
+
if (this.config.prompt) {
|
|
23
|
+
instruction += ` ${this.config.prompt}`;
|
|
24
|
+
}
|
|
25
|
+
const base64 = Buffer.from(data).toString("base64");
|
|
26
|
+
const request = {
|
|
27
|
+
model: this.config.model,
|
|
28
|
+
contents: [{
|
|
29
|
+
role: "user",
|
|
30
|
+
parts: [
|
|
31
|
+
{ inlineData: { mimeType: googleAudioWireMime(mimeType), data: base64 } },
|
|
32
|
+
{ text: instruction },
|
|
33
|
+
],
|
|
34
|
+
}],
|
|
35
|
+
};
|
|
36
|
+
if (Buffer.byteLength(JSON.stringify(request), "utf8") > GOOGLE_INLINE_REQUEST_MAX_BYTES) {
|
|
37
|
+
return failure("Audio and instructions exceed Gemini's 20 MB inline request limit; use a smaller source.");
|
|
38
|
+
}
|
|
39
|
+
const ai = new GoogleGenAI({ apiKey: this.config.apiKey });
|
|
40
|
+
// Signal is passed at call time (not part of `request`) so the encoded-size
|
|
41
|
+
// check above measures only the payload. Gemini's abortSignal is client-only:
|
|
42
|
+
// it tears down the request but does not stop server-side billing.
|
|
43
|
+
const res = await ai.models.generateContent({
|
|
44
|
+
...request,
|
|
45
|
+
config: { abortSignal: this.config.abortSignal },
|
|
46
|
+
});
|
|
47
|
+
const usage = normalizeGoogleAudioUsage(res.usageMetadata, "input");
|
|
48
|
+
const result = {
|
|
49
|
+
text: res.text ?? "",
|
|
50
|
+
raw: res,
|
|
51
|
+
};
|
|
52
|
+
if (usage)
|
|
53
|
+
result.usage = usage;
|
|
54
|
+
return success(result);
|
|
55
|
+
}
|
|
56
|
+
}
|
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
import OpenAI from "openai";
|
|
2
|
+
import { OpenAITranscriptionClient } from "./openai.js";
|
|
3
|
+
/**
|
|
4
|
+
* Groq exposes an OpenAI-compatible /audio/transcriptions endpoint (Whisper
|
|
5
|
+
* large-v3 / large-v3-turbo). Everything but the base URL is inherited.
|
|
6
|
+
*/
|
|
7
|
+
export declare class GroqTranscriptionClient extends OpenAITranscriptionClient {
|
|
8
|
+
protected makeClient(): OpenAI;
|
|
9
|
+
protected noKeyMessage(): string;
|
|
10
|
+
}
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
import OpenAI from "openai";
|
|
2
|
+
import { OpenAITranscriptionClient } from "./openai.js";
|
|
3
|
+
/**
|
|
4
|
+
* Groq exposes an OpenAI-compatible /audio/transcriptions endpoint (Whisper
|
|
5
|
+
* large-v3 / large-v3-turbo). Everything but the base URL is inherited.
|
|
6
|
+
*/
|
|
7
|
+
export class GroqTranscriptionClient extends OpenAITranscriptionClient {
|
|
8
|
+
makeClient() {
|
|
9
|
+
return new OpenAI({
|
|
10
|
+
apiKey: this.config.apiKey,
|
|
11
|
+
baseURL: "https://api.groq.com/openai/v1",
|
|
12
|
+
});
|
|
13
|
+
}
|
|
14
|
+
noKeyMessage() {
|
|
15
|
+
return "No Groq API key provided. Set apiKey.groq or GROQ_API_KEY.";
|
|
16
|
+
}
|
|
17
|
+
}
|
|
@@ -1,6 +1,11 @@
|
|
|
1
|
+
import OpenAI from "openai";
|
|
1
2
|
import { Result } from "../types/result.js";
|
|
2
3
|
import { BaseTranscriptionClient } from "./baseTranscriptionClient.js";
|
|
3
4
|
import type { TranscriptionResult } from "../transcription.js";
|
|
4
5
|
export declare class OpenAITranscriptionClient extends BaseTranscriptionClient {
|
|
6
|
+
/** Build the OpenAI SDK client. Subclasses override to point at a compatible base URL. */
|
|
7
|
+
protected makeClient(): OpenAI;
|
|
8
|
+
/** Provider-specific diagnostic when no API key is resolved. Subclasses override. */
|
|
9
|
+
protected noKeyMessage(): string;
|
|
5
10
|
protected _transcribe(data: Uint8Array, mimeType: string): Promise<Result<TranscriptionResult>>;
|
|
6
11
|
}
|
|
@@ -3,16 +3,24 @@ import { success, failure } from "../types/result.js";
|
|
|
3
3
|
import { transcriptionAudioType } from "../util/audioMime.js";
|
|
4
4
|
import { BaseTranscriptionClient } from "./baseTranscriptionClient.js";
|
|
5
5
|
export class OpenAITranscriptionClient extends BaseTranscriptionClient {
|
|
6
|
+
/** Build the OpenAI SDK client. Subclasses override to point at a compatible base URL. */
|
|
7
|
+
makeClient() {
|
|
8
|
+
return new OpenAI({ apiKey: this.config.apiKey });
|
|
9
|
+
}
|
|
10
|
+
/** Provider-specific diagnostic when no API key is resolved. Subclasses override. */
|
|
11
|
+
noKeyMessage() {
|
|
12
|
+
return "No OpenAI API key provided. Set apiKey.openAi or OPENAI_API_KEY.";
|
|
13
|
+
}
|
|
6
14
|
// No try/catch here: BaseTranscriptionClient.transcribe() is the single
|
|
7
15
|
// redacting/logging exception boundary.
|
|
8
16
|
async _transcribe(data, mimeType) {
|
|
9
17
|
if (!this.config.apiKey) {
|
|
10
|
-
return failure(
|
|
18
|
+
return failure(this.noKeyMessage());
|
|
11
19
|
}
|
|
12
20
|
// Filename is an OpenAI upload detail, not part of the provider-neutral
|
|
13
21
|
// operation contract. Derive the synthetic name from the normalized MIME.
|
|
14
22
|
const filename = transcriptionAudioType(mimeType)?.filename ?? "audio.bin";
|
|
15
|
-
const client =
|
|
23
|
+
const client = this.makeClient();
|
|
16
24
|
const file = await toFile(data, filename, { type: mimeType });
|
|
17
25
|
const granularities = [];
|
|
18
26
|
if (this.config.timestampGranularity) {
|
|
@@ -32,7 +40,7 @@ export class OpenAITranscriptionClient extends BaseTranscriptionClient {
|
|
|
32
40
|
if (granularities.length > 0) {
|
|
33
41
|
requestBody.timestamp_granularities = granularities;
|
|
34
42
|
}
|
|
35
|
-
const res = (await client.audio.transcriptions.create(requestBody));
|
|
43
|
+
const res = (await client.audio.transcriptions.create(requestBody, { signal: this.config.abortSignal }));
|
|
36
44
|
const result = { text: res.text, raw: res };
|
|
37
45
|
if (res.language) {
|
|
38
46
|
result.language = res.language;
|
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
import OpenAI from "openai";
|
|
2
|
+
import { OpenAITranscriptionClient } from "./openai.js";
|
|
3
|
+
/**
|
|
4
|
+
* Generic OpenAI-compatible transcription client. Point it at any provider
|
|
5
|
+
* exposing an OpenAI-shaped /audio/transcriptions endpoint via
|
|
6
|
+
* `config.baseUrl.openAiCompat` (or OPENAI_COMPAT_BASE_URL) and
|
|
7
|
+
* `config.apiKey.openAiCompat` (or OPENAI_COMPAT_API_KEY). Mirrors the chat
|
|
8
|
+
* `SmolOpenAiCompat` client.
|
|
9
|
+
*/
|
|
10
|
+
export declare class OpenAiCompatTranscriptionClient extends OpenAITranscriptionClient {
|
|
11
|
+
protected makeClient(): OpenAI;
|
|
12
|
+
protected noKeyMessage(): string;
|
|
13
|
+
}
|