smoltalk 0.10.0 → 0.10.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +40 -6
- package/dist/model.js +30 -11
- package/dist/models.d.ts +61 -3
- package/dist/models.js +74 -0
- package/dist/speech/baseSpeechClient.d.ts +5 -0
- package/dist/speech/baseSpeechClient.js +21 -2
- package/dist/speech/google.d.ts +6 -0
- package/dist/speech/google.js +54 -0
- package/dist/speech/groq.d.ts +11 -0
- package/dist/speech/groq.js +19 -0
- package/dist/speech/openai.d.ts +8 -0
- package/dist/speech/openai.js +16 -4
- package/dist/speech/openaiCompat.d.ts +13 -0
- package/dist/speech/openaiCompat.js +22 -0
- package/dist/speech.d.ts +5 -0
- package/dist/speech.js +6 -0
- package/dist/transcription/baseTranscriptionClient.d.ts +5 -0
- package/dist/transcription/baseTranscriptionClient.js +44 -18
- package/dist/transcription/google.d.ts +6 -0
- package/dist/transcription/google.js +56 -0
- package/dist/transcription/groq.d.ts +10 -0
- package/dist/transcription/groq.js +17 -0
- package/dist/transcription/openai.d.ts +5 -0
- package/dist/transcription/openai.js +11 -3
- package/dist/transcription/openaiCompat.d.ts +13 -0
- package/dist/transcription/openaiCompat.js +22 -0
- package/dist/transcription.d.ts +3 -0
- package/dist/transcription.js +6 -0
- package/dist/types.d.ts +1 -0
- package/dist/util/audioMime.d.ts +17 -0
- package/dist/util/audioMime.js +42 -1
- package/dist/util/googleAudioUsage.d.ts +14 -0
- package/dist/util/googleAudioUsage.js +52 -0
- package/dist/util/mime.js +2 -0
- package/dist/util/provider.d.ts +1 -0
- package/dist/util/provider.js +2 -0
- package/package.json +1 -1
package/dist/speech.d.ts
CHANGED
|
@@ -2,6 +2,7 @@ import type { ModelDataBlob } from "./modelData.js";
|
|
|
2
2
|
import type { SmolConfig } from "./types.js";
|
|
3
3
|
import { Result } from "./types/result.js";
|
|
4
4
|
import { CostEstimate } from "./types/costEstimate.js";
|
|
5
|
+
import { TokenUsage } from "./types/tokenUsage.js";
|
|
5
6
|
import { BaseSpeechClient, SpeechClientConfig } from "./speech/baseSpeechClient.js";
|
|
6
7
|
export type SpeakOptions = {
|
|
7
8
|
model: string;
|
|
@@ -9,9 +10,12 @@ export type SpeakOptions = {
|
|
|
9
10
|
provider?: string;
|
|
10
11
|
modelData?: ModelDataBlob;
|
|
11
12
|
apiKey?: SmolConfig["apiKey"];
|
|
13
|
+
baseUrl?: SmolConfig["baseUrl"];
|
|
12
14
|
format?: string;
|
|
13
15
|
speed?: number;
|
|
14
16
|
metadata?: Record<string, unknown>;
|
|
17
|
+
/** Abort the in-flight provider request when this signal fires. */
|
|
18
|
+
abortSignal?: AbortSignal;
|
|
15
19
|
};
|
|
16
20
|
export type PcmAudioMetadata = {
|
|
17
21
|
sampleRateHz: number;
|
|
@@ -23,6 +27,7 @@ export type SpeechResult = {
|
|
|
23
27
|
audio: Uint8Array;
|
|
24
28
|
mimeType: string;
|
|
25
29
|
pcm?: PcmAudioMetadata;
|
|
30
|
+
usage?: TokenUsage;
|
|
26
31
|
cost?: CostEstimate;
|
|
27
32
|
raw?: unknown;
|
|
28
33
|
};
|
package/dist/speech.js
CHANGED
|
@@ -3,9 +3,15 @@ import { redactSecret } from "./util/redact.js";
|
|
|
3
3
|
import { getLogger } from "./util/logger.js";
|
|
4
4
|
import { resolveProvider, resolveApiKey } from "./util/provider.js";
|
|
5
5
|
import { OpenAISpeechClient } from "./speech/openai.js";
|
|
6
|
+
import { GroqSpeechClient } from "./speech/groq.js";
|
|
7
|
+
import { GoogleSpeechClient } from "./speech/google.js";
|
|
8
|
+
import { OpenAiCompatSpeechClient } from "./speech/openaiCompat.js";
|
|
6
9
|
// Checked before the user registry so a registered "openai" can't hijack the built-in.
|
|
7
10
|
const builtinClients = Object.create(null);
|
|
8
11
|
builtinClients["openai"] = OpenAISpeechClient;
|
|
12
|
+
builtinClients["groq"] = GroqSpeechClient;
|
|
13
|
+
builtinClients["google"] = GoogleSpeechClient;
|
|
14
|
+
builtinClients["openai-compat"] = OpenAiCompatSpeechClient;
|
|
9
15
|
// Null-prototype so provider names like "toString"/"__proto__" can't collide
|
|
10
16
|
// with Object.prototype or pollute the registry.
|
|
11
17
|
const registered = Object.create(null);
|
|
@@ -1,4 +1,5 @@
|
|
|
1
1
|
import type { ModelDataBlob } from "../modelData.js";
|
|
2
|
+
import type { SmolConfig } from "../types.js";
|
|
2
3
|
import { Result } from "../types/result.js";
|
|
3
4
|
import { BlobRef } from "../util/blobRef.js";
|
|
4
5
|
import type { TranscriptionResult } from "../transcription.js";
|
|
@@ -9,12 +10,16 @@ export type TranscriptionClientConfig = {
|
|
|
9
10
|
provider: string;
|
|
10
11
|
/** Resolved API key; empty string when none was found. */
|
|
11
12
|
apiKey: string;
|
|
13
|
+
/** Base-URL map (for OpenAI-compatible providers); read via resolveBaseUrl. */
|
|
14
|
+
baseUrl?: SmolConfig["baseUrl"];
|
|
12
15
|
modelData?: ModelDataBlob;
|
|
13
16
|
language?: string;
|
|
14
17
|
prompt?: string;
|
|
15
18
|
timestampGranularity?: "segment" | "word";
|
|
16
19
|
maxBytes?: number;
|
|
17
20
|
metadata?: Record<string, unknown>;
|
|
21
|
+
/** Abort the in-flight provider request when this signal fires. */
|
|
22
|
+
abortSignal?: AbortSignal;
|
|
18
23
|
};
|
|
19
24
|
/**
|
|
20
25
|
* Shared transcription behavior, mirroring BaseClient for text generation:
|
|
@@ -1,5 +1,5 @@
|
|
|
1
|
-
import { getModelForProvider, isSpeechToTextModel, } from "../models.js";
|
|
2
|
-
import { calculateTranscriptionCost } from "../model.js";
|
|
1
|
+
import { getModelForProvider, isSpeechToTextModel, modelSupportsInputModality, audioInputConstraints, } from "../models.js";
|
|
2
|
+
import { Model, calculateTranscriptionCost } from "../model.js";
|
|
3
3
|
import { success, failure } from "../types/result.js";
|
|
4
4
|
import { loadBlob } from "../util/blobRef.js";
|
|
5
5
|
import { audioFormatForMime, canonicalizeMime } from "../util/mime.js";
|
|
@@ -7,24 +7,24 @@ import { redactSecret } from "../util/redact.js";
|
|
|
7
7
|
import { getLogger } from "../util/logger.js";
|
|
8
8
|
export const DEFAULT_TRANSCRIBE_BYTES = 25 * 1024 * 1024;
|
|
9
9
|
/** Validate the declarative STT constraint block once before consuming it. */
|
|
10
|
-
function transcriptionConstraintError(
|
|
11
|
-
const modelMaxBytes =
|
|
10
|
+
function transcriptionConstraintError(modelName, c) {
|
|
11
|
+
const modelMaxBytes = c.maxBytes;
|
|
12
12
|
if (modelMaxBytes !== undefined &&
|
|
13
13
|
(typeof modelMaxBytes !== "number" || !Number.isFinite(modelMaxBytes) || modelMaxBytes <= 0)) {
|
|
14
|
-
return `Model "${
|
|
14
|
+
return `Model "${modelName}" has an invalid maxBytes value.`;
|
|
15
15
|
}
|
|
16
|
-
const supportedMimeTypes =
|
|
16
|
+
const supportedMimeTypes = c.supportedMimeTypes;
|
|
17
17
|
if (supportedMimeTypes !== undefined &&
|
|
18
18
|
(!Array.isArray(supportedMimeTypes) ||
|
|
19
19
|
!supportedMimeTypes.every((mime) => typeof mime === "string"))) {
|
|
20
|
-
return `Model "${
|
|
20
|
+
return `Model "${modelName}" has invalid supportedMimeTypes.`;
|
|
21
21
|
}
|
|
22
22
|
return null;
|
|
23
23
|
}
|
|
24
24
|
// The caller's maxBytes is a safety limit; the model's maxBytes is the
|
|
25
25
|
// provider's hard cap. Take the smaller of whichever are present so a caller
|
|
26
26
|
// can tighten the limit but never bypass the provider cap.
|
|
27
|
-
function resolveTranscriptionMaxBytes(callerMaxBytes,
|
|
27
|
+
function resolveTranscriptionMaxBytes(callerMaxBytes, modelMaxBytes) {
|
|
28
28
|
if (callerMaxBytes !== undefined &&
|
|
29
29
|
(!Number.isFinite(callerMaxBytes) || callerMaxBytes <= 0)) {
|
|
30
30
|
return failure(`maxBytes must be a positive finite number (got ${callerMaxBytes}).`);
|
|
@@ -33,8 +33,8 @@ function resolveTranscriptionMaxBytes(callerMaxBytes, model) {
|
|
|
33
33
|
if (callerMaxBytes !== undefined) {
|
|
34
34
|
limits.push(callerMaxBytes);
|
|
35
35
|
}
|
|
36
|
-
if (
|
|
37
|
-
limits.push(
|
|
36
|
+
if (modelMaxBytes !== undefined) {
|
|
37
|
+
limits.push(modelMaxBytes);
|
|
38
38
|
}
|
|
39
39
|
if (limits.length === 0) {
|
|
40
40
|
return success(DEFAULT_TRANSCRIBE_BYTES);
|
|
@@ -53,18 +53,29 @@ export class BaseTranscriptionClient {
|
|
|
53
53
|
this.config = config;
|
|
54
54
|
}
|
|
55
55
|
async transcribe(source) {
|
|
56
|
+
// Already-aborted signal: stop before doing any paid work.
|
|
57
|
+
if (this.config.abortSignal?.aborted) {
|
|
58
|
+
return failure("Request was aborted");
|
|
59
|
+
}
|
|
56
60
|
try {
|
|
57
61
|
const model = getModelForProvider(this.config.provider, this.config.model, this.config.modelData);
|
|
58
|
-
|
|
59
|
-
|
|
62
|
+
// A model is a valid transcription target if it is a dedicated STT model,
|
|
63
|
+
// or a multimodal text model that accepts audio input (e.g. Gemini). An
|
|
64
|
+
// unknown model (undefined) flows through — the provider is authority.
|
|
65
|
+
const acceptsAudio = model === undefined ||
|
|
66
|
+
isSpeechToTextModel(model) ||
|
|
67
|
+
modelSupportsInputModality(this.config.model, "audio", this.config.modelData, this.config.provider) === true;
|
|
68
|
+
if (!acceptsAudio) {
|
|
69
|
+
return failure(`Model "${this.config.model}" cannot accept audio input (not a transcription model).`);
|
|
60
70
|
}
|
|
71
|
+
const constraints = model !== undefined ? audioInputConstraints(model) : {};
|
|
61
72
|
if (model !== undefined) {
|
|
62
|
-
const constraintError = transcriptionConstraintError(model);
|
|
73
|
+
const constraintError = transcriptionConstraintError(this.config.model, constraints);
|
|
63
74
|
if (constraintError !== null) {
|
|
64
75
|
return failure(constraintError);
|
|
65
76
|
}
|
|
66
77
|
}
|
|
67
|
-
const effectiveLimit = resolveTranscriptionMaxBytes(this.config.maxBytes,
|
|
78
|
+
const effectiveLimit = resolveTranscriptionMaxBytes(this.config.maxBytes, constraints.maxBytes);
|
|
68
79
|
if (!effectiveLimit.success) {
|
|
69
80
|
return effectiveLimit;
|
|
70
81
|
}
|
|
@@ -76,25 +87,40 @@ export class BaseTranscriptionClient {
|
|
|
76
87
|
return failure(`Failed to load audio for transcription: ${err.message}`);
|
|
77
88
|
}
|
|
78
89
|
const mimeType = loaded.mimeType ?? "application/octet-stream";
|
|
79
|
-
if (
|
|
90
|
+
if (constraints.supportedMimeTypes !== undefined) {
|
|
80
91
|
const audioFormat = audioFormatForMime(mimeType);
|
|
81
92
|
const normalizedMime = audioFormat?.mimeType ?? canonicalizeMime(mimeType);
|
|
82
|
-
if (!
|
|
93
|
+
if (!constraints.supportedMimeTypes.includes(normalizedMime)) {
|
|
83
94
|
return failure(`Unsupported audio type "${mimeType}" for model "${this.config.model}". ` +
|
|
84
|
-
`Supported: ${
|
|
95
|
+
`Supported: ${constraints.supportedMimeTypes.join(", ")}.`);
|
|
85
96
|
}
|
|
86
97
|
}
|
|
98
|
+
// Re-check after the async preflight (blob load / MIME validation): the
|
|
99
|
+
// signal may have fired during it, and we must not dispatch a request.
|
|
100
|
+
if (this.config.abortSignal?.aborted) {
|
|
101
|
+
return failure("Request was aborted");
|
|
102
|
+
}
|
|
87
103
|
const result = await this._transcribe(loaded.data, mimeType);
|
|
88
104
|
if (!result.success) {
|
|
89
105
|
return result;
|
|
90
106
|
}
|
|
91
|
-
|
|
107
|
+
let cost = calculateTranscriptionCost(model, result.value.durationSeconds);
|
|
108
|
+
if (cost === undefined && result.value.usage !== undefined) {
|
|
109
|
+
// Token-billed providers (Gemini) price through the shared cost engine.
|
|
110
|
+
cost =
|
|
111
|
+
new Model(this.config.model, this.config.provider, this.config.modelData).calculateCost(result.value.usage) ?? undefined;
|
|
112
|
+
}
|
|
92
113
|
if (cost !== undefined) {
|
|
93
114
|
result.value.cost = cost;
|
|
94
115
|
}
|
|
95
116
|
return result;
|
|
96
117
|
}
|
|
97
118
|
catch (err) {
|
|
119
|
+
// Caller-initiated cancellation surfaces as a distinguishable failure
|
|
120
|
+
// (matching the chat path), not a redacted provider error.
|
|
121
|
+
if (this.config.abortSignal?.aborted) {
|
|
122
|
+
return failure("Request was aborted");
|
|
123
|
+
}
|
|
98
124
|
let msg = "transcribe() failed";
|
|
99
125
|
if (err instanceof Error) {
|
|
100
126
|
msg = err.message;
|
|
@@ -0,0 +1,6 @@
|
|
|
1
|
+
import { Result } from "../types/result.js";
|
|
2
|
+
import { BaseTranscriptionClient } from "./baseTranscriptionClient.js";
|
|
3
|
+
import type { TranscriptionResult } from "../transcription.js";
|
|
4
|
+
export declare class GoogleTranscriptionClient extends BaseTranscriptionClient {
|
|
5
|
+
protected _transcribe(data: Uint8Array, mimeType: string): Promise<Result<TranscriptionResult>>;
|
|
6
|
+
}
|
|
@@ -0,0 +1,56 @@
|
|
|
1
|
+
import { GoogleGenAI } from "@google/genai";
|
|
2
|
+
import { success, failure } from "../types/result.js";
|
|
3
|
+
import { BaseTranscriptionClient } from "./baseTranscriptionClient.js";
|
|
4
|
+
import { normalizeGoogleAudioUsage } from "../util/googleAudioUsage.js";
|
|
5
|
+
import { googleAudioWireMime } from "../util/audioMime.js";
|
|
6
|
+
// Gemini's inline request cap is 20 MB for the *entire* request (audio bytes,
|
|
7
|
+
// base64 expansion, prompt, and SDK envelope), not the raw file alone.
|
|
8
|
+
const GOOGLE_INLINE_REQUEST_MAX_BYTES = 20_000_000;
|
|
9
|
+
export class GoogleTranscriptionClient extends BaseTranscriptionClient {
|
|
10
|
+
// No try/catch: BaseTranscriptionClient.transcribe() is the exception boundary.
|
|
11
|
+
async _transcribe(data, mimeType) {
|
|
12
|
+
if (!this.config.apiKey) {
|
|
13
|
+
return failure("No Google API key provided. Set apiKey.google or GEMINI_API_KEY.");
|
|
14
|
+
}
|
|
15
|
+
if (this.config.timestampGranularity !== undefined) {
|
|
16
|
+
return failure("Gemini transcription does not support timestampGranularity.");
|
|
17
|
+
}
|
|
18
|
+
let instruction = "Transcribe the following audio verbatim. Output only the transcript text, with no commentary.";
|
|
19
|
+
if (this.config.language) {
|
|
20
|
+
instruction += ` The audio is in ${this.config.language}.`;
|
|
21
|
+
}
|
|
22
|
+
if (this.config.prompt) {
|
|
23
|
+
instruction += ` ${this.config.prompt}`;
|
|
24
|
+
}
|
|
25
|
+
const base64 = Buffer.from(data).toString("base64");
|
|
26
|
+
const request = {
|
|
27
|
+
model: this.config.model,
|
|
28
|
+
contents: [{
|
|
29
|
+
role: "user",
|
|
30
|
+
parts: [
|
|
31
|
+
{ inlineData: { mimeType: googleAudioWireMime(mimeType), data: base64 } },
|
|
32
|
+
{ text: instruction },
|
|
33
|
+
],
|
|
34
|
+
}],
|
|
35
|
+
};
|
|
36
|
+
if (Buffer.byteLength(JSON.stringify(request), "utf8") > GOOGLE_INLINE_REQUEST_MAX_BYTES) {
|
|
37
|
+
return failure("Audio and instructions exceed Gemini's 20 MB inline request limit; use a smaller source.");
|
|
38
|
+
}
|
|
39
|
+
const ai = new GoogleGenAI({ apiKey: this.config.apiKey });
|
|
40
|
+
// Signal is passed at call time (not part of `request`) so the encoded-size
|
|
41
|
+
// check above measures only the payload. Gemini's abortSignal is client-only:
|
|
42
|
+
// it tears down the request but does not stop server-side billing.
|
|
43
|
+
const res = await ai.models.generateContent({
|
|
44
|
+
...request,
|
|
45
|
+
config: { abortSignal: this.config.abortSignal },
|
|
46
|
+
});
|
|
47
|
+
const usage = normalizeGoogleAudioUsage(res.usageMetadata, "input");
|
|
48
|
+
const result = {
|
|
49
|
+
text: res.text ?? "",
|
|
50
|
+
raw: res,
|
|
51
|
+
};
|
|
52
|
+
if (usage)
|
|
53
|
+
result.usage = usage;
|
|
54
|
+
return success(result);
|
|
55
|
+
}
|
|
56
|
+
}
|
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
import OpenAI from "openai";
|
|
2
|
+
import { OpenAITranscriptionClient } from "./openai.js";
|
|
3
|
+
/**
|
|
4
|
+
* Groq exposes an OpenAI-compatible /audio/transcriptions endpoint (Whisper
|
|
5
|
+
* large-v3 / large-v3-turbo). Everything but the base URL is inherited.
|
|
6
|
+
*/
|
|
7
|
+
export declare class GroqTranscriptionClient extends OpenAITranscriptionClient {
|
|
8
|
+
protected makeClient(): OpenAI;
|
|
9
|
+
protected noKeyMessage(): string;
|
|
10
|
+
}
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
import OpenAI from "openai";
|
|
2
|
+
import { OpenAITranscriptionClient } from "./openai.js";
|
|
3
|
+
/**
|
|
4
|
+
* Groq exposes an OpenAI-compatible /audio/transcriptions endpoint (Whisper
|
|
5
|
+
* large-v3 / large-v3-turbo). Everything but the base URL is inherited.
|
|
6
|
+
*/
|
|
7
|
+
export class GroqTranscriptionClient extends OpenAITranscriptionClient {
|
|
8
|
+
makeClient() {
|
|
9
|
+
return new OpenAI({
|
|
10
|
+
apiKey: this.config.apiKey,
|
|
11
|
+
baseURL: "https://api.groq.com/openai/v1",
|
|
12
|
+
});
|
|
13
|
+
}
|
|
14
|
+
noKeyMessage() {
|
|
15
|
+
return "No Groq API key provided. Set apiKey.groq or GROQ_API_KEY.";
|
|
16
|
+
}
|
|
17
|
+
}
|
|
@@ -1,6 +1,11 @@
|
|
|
1
|
+
import OpenAI from "openai";
|
|
1
2
|
import { Result } from "../types/result.js";
|
|
2
3
|
import { BaseTranscriptionClient } from "./baseTranscriptionClient.js";
|
|
3
4
|
import type { TranscriptionResult } from "../transcription.js";
|
|
4
5
|
export declare class OpenAITranscriptionClient extends BaseTranscriptionClient {
|
|
6
|
+
/** Build the OpenAI SDK client. Subclasses override to point at a compatible base URL. */
|
|
7
|
+
protected makeClient(): OpenAI;
|
|
8
|
+
/** Provider-specific diagnostic when no API key is resolved. Subclasses override. */
|
|
9
|
+
protected noKeyMessage(): string;
|
|
5
10
|
protected _transcribe(data: Uint8Array, mimeType: string): Promise<Result<TranscriptionResult>>;
|
|
6
11
|
}
|
|
@@ -3,16 +3,24 @@ import { success, failure } from "../types/result.js";
|
|
|
3
3
|
import { transcriptionAudioType } from "../util/audioMime.js";
|
|
4
4
|
import { BaseTranscriptionClient } from "./baseTranscriptionClient.js";
|
|
5
5
|
export class OpenAITranscriptionClient extends BaseTranscriptionClient {
|
|
6
|
+
/** Build the OpenAI SDK client. Subclasses override to point at a compatible base URL. */
|
|
7
|
+
makeClient() {
|
|
8
|
+
return new OpenAI({ apiKey: this.config.apiKey });
|
|
9
|
+
}
|
|
10
|
+
/** Provider-specific diagnostic when no API key is resolved. Subclasses override. */
|
|
11
|
+
noKeyMessage() {
|
|
12
|
+
return "No OpenAI API key provided. Set apiKey.openAi or OPENAI_API_KEY.";
|
|
13
|
+
}
|
|
6
14
|
// No try/catch here: BaseTranscriptionClient.transcribe() is the single
|
|
7
15
|
// redacting/logging exception boundary.
|
|
8
16
|
async _transcribe(data, mimeType) {
|
|
9
17
|
if (!this.config.apiKey) {
|
|
10
|
-
return failure(
|
|
18
|
+
return failure(this.noKeyMessage());
|
|
11
19
|
}
|
|
12
20
|
// Filename is an OpenAI upload detail, not part of the provider-neutral
|
|
13
21
|
// operation contract. Derive the synthetic name from the normalized MIME.
|
|
14
22
|
const filename = transcriptionAudioType(mimeType)?.filename ?? "audio.bin";
|
|
15
|
-
const client =
|
|
23
|
+
const client = this.makeClient();
|
|
16
24
|
const file = await toFile(data, filename, { type: mimeType });
|
|
17
25
|
const granularities = [];
|
|
18
26
|
if (this.config.timestampGranularity) {
|
|
@@ -32,7 +40,7 @@ export class OpenAITranscriptionClient extends BaseTranscriptionClient {
|
|
|
32
40
|
if (granularities.length > 0) {
|
|
33
41
|
requestBody.timestamp_granularities = granularities;
|
|
34
42
|
}
|
|
35
|
-
const res = (await client.audio.transcriptions.create(requestBody));
|
|
43
|
+
const res = (await client.audio.transcriptions.create(requestBody, { signal: this.config.abortSignal }));
|
|
36
44
|
const result = { text: res.text, raw: res };
|
|
37
45
|
if (res.language) {
|
|
38
46
|
result.language = res.language;
|
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
import OpenAI from "openai";
|
|
2
|
+
import { OpenAITranscriptionClient } from "./openai.js";
|
|
3
|
+
/**
|
|
4
|
+
* Generic OpenAI-compatible transcription client. Point it at any provider
|
|
5
|
+
* exposing an OpenAI-shaped /audio/transcriptions endpoint via
|
|
6
|
+
* `config.baseUrl.openAiCompat` (or OPENAI_COMPAT_BASE_URL) and
|
|
7
|
+
* `config.apiKey.openAiCompat` (or OPENAI_COMPAT_API_KEY). Mirrors the chat
|
|
8
|
+
* `SmolOpenAiCompat` client.
|
|
9
|
+
*/
|
|
10
|
+
export declare class OpenAiCompatTranscriptionClient extends OpenAITranscriptionClient {
|
|
11
|
+
protected makeClient(): OpenAI;
|
|
12
|
+
protected noKeyMessage(): string;
|
|
13
|
+
}
|
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
import OpenAI from "openai";
|
|
2
|
+
import { OpenAITranscriptionClient } from "./openai.js";
|
|
3
|
+
import { resolveBaseUrl } from "../util/provider.js";
|
|
4
|
+
/**
|
|
5
|
+
* Generic OpenAI-compatible transcription client. Point it at any provider
|
|
6
|
+
* exposing an OpenAI-shaped /audio/transcriptions endpoint via
|
|
7
|
+
* `config.baseUrl.openAiCompat` (or OPENAI_COMPAT_BASE_URL) and
|
|
8
|
+
* `config.apiKey.openAiCompat` (or OPENAI_COMPAT_API_KEY). Mirrors the chat
|
|
9
|
+
* `SmolOpenAiCompat` client.
|
|
10
|
+
*/
|
|
11
|
+
export class OpenAiCompatTranscriptionClient extends OpenAITranscriptionClient {
|
|
12
|
+
makeClient() {
|
|
13
|
+
const baseURL = resolveBaseUrl("openai-compat", { baseUrl: this.config.baseUrl });
|
|
14
|
+
if (!baseURL) {
|
|
15
|
+
throw new Error("openai-compat: base URL required (config.baseUrl.openAiCompat or OPENAI_COMPAT_BASE_URL).");
|
|
16
|
+
}
|
|
17
|
+
return new OpenAI({ apiKey: this.config.apiKey, baseURL });
|
|
18
|
+
}
|
|
19
|
+
noKeyMessage() {
|
|
20
|
+
return "No API key provided. Set apiKey.openAiCompat or OPENAI_COMPAT_API_KEY.";
|
|
21
|
+
}
|
|
22
|
+
}
|
package/dist/transcription.d.ts
CHANGED
|
@@ -11,11 +11,14 @@ export type TranscribeOptions = {
|
|
|
11
11
|
provider?: string;
|
|
12
12
|
modelData?: ModelDataBlob;
|
|
13
13
|
apiKey?: SmolConfig["apiKey"];
|
|
14
|
+
baseUrl?: SmolConfig["baseUrl"];
|
|
14
15
|
language?: string;
|
|
15
16
|
prompt?: string;
|
|
16
17
|
timestampGranularity?: "segment" | "word";
|
|
17
18
|
maxBytes?: number;
|
|
18
19
|
metadata?: Record<string, unknown>;
|
|
20
|
+
/** Abort the in-flight provider request when this signal fires. */
|
|
21
|
+
abortSignal?: AbortSignal;
|
|
19
22
|
};
|
|
20
23
|
export type TranscriptionSegment = {
|
|
21
24
|
start: number;
|
package/dist/transcription.js
CHANGED
|
@@ -3,10 +3,16 @@ import { redactSecret } from "./util/redact.js";
|
|
|
3
3
|
import { getLogger } from "./util/logger.js";
|
|
4
4
|
import { resolveProvider, resolveApiKey } from "./util/provider.js";
|
|
5
5
|
import { OpenAITranscriptionClient } from "./transcription/openai.js";
|
|
6
|
+
import { GroqTranscriptionClient } from "./transcription/groq.js";
|
|
7
|
+
import { GoogleTranscriptionClient } from "./transcription/google.js";
|
|
8
|
+
import { OpenAiCompatTranscriptionClient } from "./transcription/openaiCompat.js";
|
|
6
9
|
export { DEFAULT_TRANSCRIBE_BYTES } from "./transcription/baseTranscriptionClient.js";
|
|
7
10
|
// Checked before the user registry so a registered "openai" can't hijack the built-in.
|
|
8
11
|
const builtinClients = Object.create(null);
|
|
9
12
|
builtinClients["openai"] = OpenAITranscriptionClient;
|
|
13
|
+
builtinClients["groq"] = GroqTranscriptionClient;
|
|
14
|
+
builtinClients["google"] = GoogleTranscriptionClient;
|
|
15
|
+
builtinClients["openai-compat"] = OpenAiCompatTranscriptionClient;
|
|
10
16
|
// Null-prototype so provider names like "toString"/"__proto__" can't collide
|
|
11
17
|
// with Object.prototype or pollute the registry.
|
|
12
18
|
const registered = Object.create(null);
|
package/dist/types.d.ts
CHANGED
|
@@ -30,6 +30,7 @@ export type SmolConfig = {
|
|
|
30
30
|
deepInfra?: string;
|
|
31
31
|
liteLlm?: string;
|
|
32
32
|
openAiCompat?: string;
|
|
33
|
+
groq?: string;
|
|
33
34
|
/** Arbitrary provider names, for keys targeting a custom-registered provider
|
|
34
35
|
* (e.g. registerProvider("acme", ...)), keyed by the exact registered name. */
|
|
35
36
|
[provider: string]: string | undefined;
|
package/dist/util/audioMime.d.ts
CHANGED
|
@@ -7,3 +7,20 @@ export declare function transcriptionAudioType(mime: string): TranscriptionAudio
|
|
|
7
7
|
export declare function chatAudioFormat(mime: string): "mp3" | "wav" | null;
|
|
8
8
|
export declare const SPEECH_FORMAT_TO_MIME: Record<SpeakFormat, string>;
|
|
9
9
|
export declare function isSpeakFormat(value: string): value is SpeakFormat;
|
|
10
|
+
/**
|
|
11
|
+
* Translate a canonical/alias audio MIME to the wire value Google expects.
|
|
12
|
+
* Google documents MP3 as `audio/mp3`, whereas this repo canonicalizes it to
|
|
13
|
+
* `audio/mpeg`; everything else passes through canonicalized.
|
|
14
|
+
*/
|
|
15
|
+
export declare function googleAudioWireMime(mimeType: string): string;
|
|
16
|
+
export type PcmWavOptions = {
|
|
17
|
+
sampleRateHz: number;
|
|
18
|
+
channels: number;
|
|
19
|
+
bitsPerSample: number;
|
|
20
|
+
};
|
|
21
|
+
/**
|
|
22
|
+
* Wrap raw signed-integer little-endian PCM in a 44-byte RIFF/WAVE header so it
|
|
23
|
+
* becomes a directly-playable .wav. Pure function, no dependency. Used for
|
|
24
|
+
* Gemini TTS output, which is only ever raw PCM.
|
|
25
|
+
*/
|
|
26
|
+
export declare function pcmToWav(pcm: Uint8Array, opts: PcmWavOptions): Uint8Array;
|
package/dist/util/audioMime.js
CHANGED
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { audioFormatForMime } from "./mime.js";
|
|
1
|
+
import { audioFormatForMime, canonicalizeMime } from "./mime.js";
|
|
2
2
|
export function transcriptionAudioType(mime) {
|
|
3
3
|
const format = audioFormatForMime(mime);
|
|
4
4
|
if (format === null) {
|
|
@@ -31,3 +31,44 @@ export const SPEECH_FORMAT_TO_MIME = {
|
|
|
31
31
|
export function isSpeakFormat(value) {
|
|
32
32
|
return Object.hasOwn(SPEECH_FORMAT_TO_MIME, value);
|
|
33
33
|
}
|
|
34
|
+
/**
|
|
35
|
+
* Translate a canonical/alias audio MIME to the wire value Google expects.
|
|
36
|
+
* Google documents MP3 as `audio/mp3`, whereas this repo canonicalizes it to
|
|
37
|
+
* `audio/mpeg`; everything else passes through canonicalized.
|
|
38
|
+
*/
|
|
39
|
+
export function googleAudioWireMime(mimeType) {
|
|
40
|
+
const format = audioFormatForMime(mimeType);
|
|
41
|
+
const canonical = format?.mimeType ?? canonicalizeMime(mimeType);
|
|
42
|
+
return canonical === "audio/mpeg" ? "audio/mp3" : canonical;
|
|
43
|
+
}
|
|
44
|
+
/**
|
|
45
|
+
* Wrap raw signed-integer little-endian PCM in a 44-byte RIFF/WAVE header so it
|
|
46
|
+
* becomes a directly-playable .wav. Pure function, no dependency. Used for
|
|
47
|
+
* Gemini TTS output, which is only ever raw PCM.
|
|
48
|
+
*/
|
|
49
|
+
export function pcmToWav(pcm, opts) {
|
|
50
|
+
const { sampleRateHz, channels, bitsPerSample } = opts;
|
|
51
|
+
const blockAlign = (channels * bitsPerSample) / 8;
|
|
52
|
+
const byteRate = sampleRateHz * blockAlign;
|
|
53
|
+
const out = new Uint8Array(44 + pcm.length);
|
|
54
|
+
const view = new DataView(out.buffer);
|
|
55
|
+
const writeAscii = (offset, s) => {
|
|
56
|
+
for (let i = 0; i < s.length; i++)
|
|
57
|
+
view.setUint8(offset + i, s.charCodeAt(i));
|
|
58
|
+
};
|
|
59
|
+
writeAscii(0, "RIFF");
|
|
60
|
+
view.setUint32(4, 36 + pcm.length, true);
|
|
61
|
+
writeAscii(8, "WAVE");
|
|
62
|
+
writeAscii(12, "fmt ");
|
|
63
|
+
view.setUint32(16, 16, true); // PCM fmt chunk size
|
|
64
|
+
view.setUint16(20, 1, true); // audio format = PCM
|
|
65
|
+
view.setUint16(22, channels, true);
|
|
66
|
+
view.setUint32(24, sampleRateHz, true);
|
|
67
|
+
view.setUint32(28, byteRate, true);
|
|
68
|
+
view.setUint16(32, blockAlign, true);
|
|
69
|
+
view.setUint16(34, bitsPerSample, true);
|
|
70
|
+
writeAscii(36, "data");
|
|
71
|
+
view.setUint32(40, pcm.length, true);
|
|
72
|
+
out.set(pcm, 44);
|
|
73
|
+
return out;
|
|
74
|
+
}
|
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
import type { GenerateContentResponseUsageMetadata } from "@google/genai";
|
|
2
|
+
import type { TokenUsage } from "../types/tokenUsage.js";
|
|
3
|
+
export type GoogleAudioDirection = "input" | "output";
|
|
4
|
+
/**
|
|
5
|
+
* Normalize Gemini's usage metadata into smoltalk's TokenUsage, separating the
|
|
6
|
+
* audio bucket for the given direction so the token-priced cost engine can bill
|
|
7
|
+
* it at the audio rate. STT sets audioDirection "input" (audio prompt tokens);
|
|
8
|
+
* TTS sets "output" (audio candidate tokens).
|
|
9
|
+
*
|
|
10
|
+
* Thinking tokens (`thoughtsTokenCount`) are reported separately from
|
|
11
|
+
* `candidatesTokenCount` and are always text — they go into the text-output
|
|
12
|
+
* bucket, never the audio-output bucket.
|
|
13
|
+
*/
|
|
14
|
+
export declare function normalizeGoogleAudioUsage(metadata: GenerateContentResponseUsageMetadata | undefined, audioDirection: GoogleAudioDirection): TokenUsage | undefined;
|
|
@@ -0,0 +1,52 @@
|
|
|
1
|
+
function modalityTokens(details, modality) {
|
|
2
|
+
return (details ?? [])
|
|
3
|
+
// `modality` is a MediaModality enum; compare by its string value.
|
|
4
|
+
.filter((detail) => String(detail.modality) === modality)
|
|
5
|
+
.reduce((sum, detail) => sum + (detail.tokenCount ?? 0), 0);
|
|
6
|
+
}
|
|
7
|
+
/**
|
|
8
|
+
* Normalize Gemini's usage metadata into smoltalk's TokenUsage, separating the
|
|
9
|
+
* audio bucket for the given direction so the token-priced cost engine can bill
|
|
10
|
+
* it at the audio rate. STT sets audioDirection "input" (audio prompt tokens);
|
|
11
|
+
* TTS sets "output" (audio candidate tokens).
|
|
12
|
+
*
|
|
13
|
+
* Thinking tokens (`thoughtsTokenCount`) are reported separately from
|
|
14
|
+
* `candidatesTokenCount` and are always text — they go into the text-output
|
|
15
|
+
* bucket, never the audio-output bucket.
|
|
16
|
+
*/
|
|
17
|
+
export function normalizeGoogleAudioUsage(metadata, audioDirection) {
|
|
18
|
+
if (metadata === undefined) {
|
|
19
|
+
return undefined;
|
|
20
|
+
}
|
|
21
|
+
const prompt = metadata.promptTokenCount ?? 0;
|
|
22
|
+
const candidates = metadata.candidatesTokenCount ?? 0;
|
|
23
|
+
const thoughts = metadata.thoughtsTokenCount ?? 0;
|
|
24
|
+
if (audioDirection === "input") {
|
|
25
|
+
const audioIn = modalityTokens(metadata.promptTokensDetails, "AUDIO");
|
|
26
|
+
const usage = {
|
|
27
|
+
inputTokens: Math.max(0, prompt - audioIn),
|
|
28
|
+
outputTokens: candidates + thoughts,
|
|
29
|
+
};
|
|
30
|
+
if (audioIn > 0)
|
|
31
|
+
usage.inputAudioTokens = audioIn;
|
|
32
|
+
if (metadata.totalTokenCount !== undefined)
|
|
33
|
+
usage.totalTokens = metadata.totalTokenCount;
|
|
34
|
+
return usage;
|
|
35
|
+
}
|
|
36
|
+
// Output (TTS): split the audio bucket out of candidate tokens. Only assume
|
|
37
|
+
// "all candidates are audio" when the details array is missing/empty; when it
|
|
38
|
+
// is present, trust its AUDIO total even if that total is 0 (otherwise
|
|
39
|
+
// purely-text candidate tokens would be mispriced at the audio rate).
|
|
40
|
+
const details = metadata.candidatesTokensDetails;
|
|
41
|
+
const detailsPresent = Array.isArray(details) && details.length > 0;
|
|
42
|
+
const audioOut = detailsPresent ? modalityTokens(details, "AUDIO") : candidates;
|
|
43
|
+
const usage = {
|
|
44
|
+
inputTokens: prompt,
|
|
45
|
+
outputTokens: Math.max(0, candidates - audioOut) + thoughts,
|
|
46
|
+
};
|
|
47
|
+
if (audioOut > 0)
|
|
48
|
+
usage.outputAudioTokens = audioOut;
|
|
49
|
+
if (metadata.totalTokenCount !== undefined)
|
|
50
|
+
usage.totalTokens = metadata.totalTokenCount;
|
|
51
|
+
return usage;
|
|
52
|
+
}
|
package/dist/util/mime.js
CHANGED
|
@@ -11,6 +11,8 @@ export const AUDIO_FORMATS = [
|
|
|
11
11
|
{ extension: "ogg", mimeType: "audio/ogg", aliasMimeTypes: [], aliasExtensions: [] },
|
|
12
12
|
{ extension: "flac", mimeType: "audio/flac", aliasMimeTypes: [], aliasExtensions: [] },
|
|
13
13
|
{ extension: "webm", mimeType: "audio/webm", aliasMimeTypes: [], aliasExtensions: [] },
|
|
14
|
+
{ extension: "aac", mimeType: "audio/aac", aliasMimeTypes: [], aliasExtensions: [] },
|
|
15
|
+
{ extension: "aiff", mimeType: "audio/aiff", aliasMimeTypes: ["audio/x-aiff"], aliasExtensions: ["aif"] },
|
|
14
16
|
];
|
|
15
17
|
const IMAGE_AND_DOCUMENT_EXT_TO_MIME = {
|
|
16
18
|
".png": "image/png",
|
package/dist/util/provider.d.ts
CHANGED
package/dist/util/provider.js
CHANGED
|
@@ -37,6 +37,8 @@ export function resolveApiKey(provider, config) {
|
|
|
37
37
|
return k?.liteLlm || process.env.LITELLM_API_KEY;
|
|
38
38
|
case "openai-compat":
|
|
39
39
|
return k?.openAiCompat || process.env.OPENAI_COMPAT_API_KEY;
|
|
40
|
+
case "groq":
|
|
41
|
+
return k?.groq || process.env.GROQ_API_KEY;
|
|
40
42
|
default:
|
|
41
43
|
return config.apiKey?.[provider];
|
|
42
44
|
}
|