smoltalk 0.10.0 → 0.11.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +74 -6
- package/dist/client.d.ts +6 -0
- package/dist/client.js +12 -0
- package/dist/clients/llamaCppLoader.d.ts +42 -0
- package/dist/clients/llamaCppLoader.js +109 -0
- package/dist/functions.js +8 -1
- package/dist/index.d.ts +2 -0
- package/dist/index.js +2 -0
- package/dist/model.js +30 -11
- package/dist/models.d.ts +61 -3
- package/dist/models.js +74 -0
- package/dist/speech/baseSpeechClient.d.ts +5 -0
- package/dist/speech/baseSpeechClient.js +21 -2
- package/dist/speech/google.d.ts +6 -0
- package/dist/speech/google.js +54 -0
- package/dist/speech/groq.d.ts +11 -0
- package/dist/speech/groq.js +19 -0
- package/dist/speech/openai.d.ts +8 -0
- package/dist/speech/openai.js +16 -4
- package/dist/speech/openaiCompat.d.ts +13 -0
- package/dist/speech/openaiCompat.js +22 -0
- package/dist/speech.d.ts +5 -0
- package/dist/speech.js +6 -0
- package/dist/transcription/baseTranscriptionClient.d.ts +5 -0
- package/dist/transcription/baseTranscriptionClient.js +44 -18
- package/dist/transcription/google.d.ts +6 -0
- package/dist/transcription/google.js +56 -0
- package/dist/transcription/groq.d.ts +10 -0
- package/dist/transcription/groq.js +17 -0
- package/dist/transcription/openai.d.ts +5 -0
- package/dist/transcription/openai.js +11 -3
- package/dist/transcription/openaiCompat.d.ts +13 -0
- package/dist/transcription/openaiCompat.js +22 -0
- package/dist/transcription.d.ts +3 -0
- package/dist/transcription.js +6 -0
- package/dist/types.d.ts +1 -0
- package/dist/util/audioMime.d.ts +17 -0
- package/dist/util/audioMime.js +42 -1
- package/dist/util/googleAudioUsage.d.ts +14 -0
- package/dist/util/googleAudioUsage.js +52 -0
- package/dist/util/mime.js +2 -0
- package/dist/util/provider.d.ts +1 -0
- package/dist/util/provider.js +2 -0
- package/package.json +9 -1
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
import OpenAI from "openai";
|
|
2
|
+
import { OpenAITranscriptionClient } from "./openai.js";
|
|
3
|
+
import { resolveBaseUrl } from "../util/provider.js";
|
|
4
|
+
/**
|
|
5
|
+
* Generic OpenAI-compatible transcription client. Point it at any provider
|
|
6
|
+
* exposing an OpenAI-shaped /audio/transcriptions endpoint via
|
|
7
|
+
* `config.baseUrl.openAiCompat` (or OPENAI_COMPAT_BASE_URL) and
|
|
8
|
+
* `config.apiKey.openAiCompat` (or OPENAI_COMPAT_API_KEY). Mirrors the chat
|
|
9
|
+
* `SmolOpenAiCompat` client.
|
|
10
|
+
*/
|
|
11
|
+
export class OpenAiCompatTranscriptionClient extends OpenAITranscriptionClient {
|
|
12
|
+
makeClient() {
|
|
13
|
+
const baseURL = resolveBaseUrl("openai-compat", { baseUrl: this.config.baseUrl });
|
|
14
|
+
if (!baseURL) {
|
|
15
|
+
throw new Error("openai-compat: base URL required (config.baseUrl.openAiCompat or OPENAI_COMPAT_BASE_URL).");
|
|
16
|
+
}
|
|
17
|
+
return new OpenAI({ apiKey: this.config.apiKey, baseURL });
|
|
18
|
+
}
|
|
19
|
+
noKeyMessage() {
|
|
20
|
+
return "No API key provided. Set apiKey.openAiCompat or OPENAI_COMPAT_API_KEY.";
|
|
21
|
+
}
|
|
22
|
+
}
|
package/dist/transcription.d.ts
CHANGED
|
@@ -11,11 +11,14 @@ export type TranscribeOptions = {
|
|
|
11
11
|
provider?: string;
|
|
12
12
|
modelData?: ModelDataBlob;
|
|
13
13
|
apiKey?: SmolConfig["apiKey"];
|
|
14
|
+
baseUrl?: SmolConfig["baseUrl"];
|
|
14
15
|
language?: string;
|
|
15
16
|
prompt?: string;
|
|
16
17
|
timestampGranularity?: "segment" | "word";
|
|
17
18
|
maxBytes?: number;
|
|
18
19
|
metadata?: Record<string, unknown>;
|
|
20
|
+
/** Abort the in-flight provider request when this signal fires. */
|
|
21
|
+
abortSignal?: AbortSignal;
|
|
19
22
|
};
|
|
20
23
|
export type TranscriptionSegment = {
|
|
21
24
|
start: number;
|
package/dist/transcription.js
CHANGED
|
@@ -3,10 +3,16 @@ import { redactSecret } from "./util/redact.js";
|
|
|
3
3
|
import { getLogger } from "./util/logger.js";
|
|
4
4
|
import { resolveProvider, resolveApiKey } from "./util/provider.js";
|
|
5
5
|
import { OpenAITranscriptionClient } from "./transcription/openai.js";
|
|
6
|
+
import { GroqTranscriptionClient } from "./transcription/groq.js";
|
|
7
|
+
import { GoogleTranscriptionClient } from "./transcription/google.js";
|
|
8
|
+
import { OpenAiCompatTranscriptionClient } from "./transcription/openaiCompat.js";
|
|
6
9
|
export { DEFAULT_TRANSCRIBE_BYTES } from "./transcription/baseTranscriptionClient.js";
|
|
7
10
|
// Checked before the user registry so a registered "openai" can't hijack the built-in.
|
|
8
11
|
const builtinClients = Object.create(null);
|
|
9
12
|
builtinClients["openai"] = OpenAITranscriptionClient;
|
|
13
|
+
builtinClients["groq"] = GroqTranscriptionClient;
|
|
14
|
+
builtinClients["google"] = GoogleTranscriptionClient;
|
|
15
|
+
builtinClients["openai-compat"] = OpenAiCompatTranscriptionClient;
|
|
10
16
|
// Null-prototype so provider names like "toString"/"__proto__" can't collide
|
|
11
17
|
// with Object.prototype or pollute the registry.
|
|
12
18
|
const registered = Object.create(null);
|
package/dist/types.d.ts
CHANGED
|
@@ -30,6 +30,7 @@ export type SmolConfig = {
|
|
|
30
30
|
deepInfra?: string;
|
|
31
31
|
liteLlm?: string;
|
|
32
32
|
openAiCompat?: string;
|
|
33
|
+
groq?: string;
|
|
33
34
|
/** Arbitrary provider names, for keys targeting a custom-registered provider
|
|
34
35
|
* (e.g. registerProvider("acme", ...)), keyed by the exact registered name. */
|
|
35
36
|
[provider: string]: string | undefined;
|
package/dist/util/audioMime.d.ts
CHANGED
|
@@ -7,3 +7,20 @@ export declare function transcriptionAudioType(mime: string): TranscriptionAudio
|
|
|
7
7
|
export declare function chatAudioFormat(mime: string): "mp3" | "wav" | null;
|
|
8
8
|
export declare const SPEECH_FORMAT_TO_MIME: Record<SpeakFormat, string>;
|
|
9
9
|
export declare function isSpeakFormat(value: string): value is SpeakFormat;
|
|
10
|
+
/**
|
|
11
|
+
* Translate a canonical/alias audio MIME to the wire value Google expects.
|
|
12
|
+
* Google documents MP3 as `audio/mp3`, whereas this repo canonicalizes it to
|
|
13
|
+
* `audio/mpeg`; everything else passes through canonicalized.
|
|
14
|
+
*/
|
|
15
|
+
export declare function googleAudioWireMime(mimeType: string): string;
|
|
16
|
+
export type PcmWavOptions = {
|
|
17
|
+
sampleRateHz: number;
|
|
18
|
+
channels: number;
|
|
19
|
+
bitsPerSample: number;
|
|
20
|
+
};
|
|
21
|
+
/**
|
|
22
|
+
* Wrap raw signed-integer little-endian PCM in a 44-byte RIFF/WAVE header so it
|
|
23
|
+
* becomes a directly-playable .wav. Pure function, no dependency. Used for
|
|
24
|
+
* Gemini TTS output, which is only ever raw PCM.
|
|
25
|
+
*/
|
|
26
|
+
export declare function pcmToWav(pcm: Uint8Array, opts: PcmWavOptions): Uint8Array;
|
package/dist/util/audioMime.js
CHANGED
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { audioFormatForMime } from "./mime.js";
|
|
1
|
+
import { audioFormatForMime, canonicalizeMime } from "./mime.js";
|
|
2
2
|
export function transcriptionAudioType(mime) {
|
|
3
3
|
const format = audioFormatForMime(mime);
|
|
4
4
|
if (format === null) {
|
|
@@ -31,3 +31,44 @@ export const SPEECH_FORMAT_TO_MIME = {
|
|
|
31
31
|
export function isSpeakFormat(value) {
|
|
32
32
|
return Object.hasOwn(SPEECH_FORMAT_TO_MIME, value);
|
|
33
33
|
}
|
|
34
|
+
/**
|
|
35
|
+
* Translate a canonical/alias audio MIME to the wire value Google expects.
|
|
36
|
+
* Google documents MP3 as `audio/mp3`, whereas this repo canonicalizes it to
|
|
37
|
+
* `audio/mpeg`; everything else passes through canonicalized.
|
|
38
|
+
*/
|
|
39
|
+
export function googleAudioWireMime(mimeType) {
|
|
40
|
+
const format = audioFormatForMime(mimeType);
|
|
41
|
+
const canonical = format?.mimeType ?? canonicalizeMime(mimeType);
|
|
42
|
+
return canonical === "audio/mpeg" ? "audio/mp3" : canonical;
|
|
43
|
+
}
|
|
44
|
+
/**
|
|
45
|
+
* Wrap raw signed-integer little-endian PCM in a 44-byte RIFF/WAVE header so it
|
|
46
|
+
* becomes a directly-playable .wav. Pure function, no dependency. Used for
|
|
47
|
+
* Gemini TTS output, which is only ever raw PCM.
|
|
48
|
+
*/
|
|
49
|
+
export function pcmToWav(pcm, opts) {
|
|
50
|
+
const { sampleRateHz, channels, bitsPerSample } = opts;
|
|
51
|
+
const blockAlign = (channels * bitsPerSample) / 8;
|
|
52
|
+
const byteRate = sampleRateHz * blockAlign;
|
|
53
|
+
const out = new Uint8Array(44 + pcm.length);
|
|
54
|
+
const view = new DataView(out.buffer);
|
|
55
|
+
const writeAscii = (offset, s) => {
|
|
56
|
+
for (let i = 0; i < s.length; i++)
|
|
57
|
+
view.setUint8(offset + i, s.charCodeAt(i));
|
|
58
|
+
};
|
|
59
|
+
writeAscii(0, "RIFF");
|
|
60
|
+
view.setUint32(4, 36 + pcm.length, true);
|
|
61
|
+
writeAscii(8, "WAVE");
|
|
62
|
+
writeAscii(12, "fmt ");
|
|
63
|
+
view.setUint32(16, 16, true); // PCM fmt chunk size
|
|
64
|
+
view.setUint16(20, 1, true); // audio format = PCM
|
|
65
|
+
view.setUint16(22, channels, true);
|
|
66
|
+
view.setUint32(24, sampleRateHz, true);
|
|
67
|
+
view.setUint32(28, byteRate, true);
|
|
68
|
+
view.setUint16(32, blockAlign, true);
|
|
69
|
+
view.setUint16(34, bitsPerSample, true);
|
|
70
|
+
writeAscii(36, "data");
|
|
71
|
+
view.setUint32(40, pcm.length, true);
|
|
72
|
+
out.set(pcm, 44);
|
|
73
|
+
return out;
|
|
74
|
+
}
|
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
import type { GenerateContentResponseUsageMetadata } from "@google/genai";
|
|
2
|
+
import type { TokenUsage } from "../types/tokenUsage.js";
|
|
3
|
+
export type GoogleAudioDirection = "input" | "output";
|
|
4
|
+
/**
|
|
5
|
+
* Normalize Gemini's usage metadata into smoltalk's TokenUsage, separating the
|
|
6
|
+
* audio bucket for the given direction so the token-priced cost engine can bill
|
|
7
|
+
* it at the audio rate. STT sets audioDirection "input" (audio prompt tokens);
|
|
8
|
+
* TTS sets "output" (audio candidate tokens).
|
|
9
|
+
*
|
|
10
|
+
* Thinking tokens (`thoughtsTokenCount`) are reported separately from
|
|
11
|
+
* `candidatesTokenCount` and are always text — they go into the text-output
|
|
12
|
+
* bucket, never the audio-output bucket.
|
|
13
|
+
*/
|
|
14
|
+
export declare function normalizeGoogleAudioUsage(metadata: GenerateContentResponseUsageMetadata | undefined, audioDirection: GoogleAudioDirection): TokenUsage | undefined;
|
|
@@ -0,0 +1,52 @@
|
|
|
1
|
+
function modalityTokens(details, modality) {
|
|
2
|
+
return (details ?? [])
|
|
3
|
+
// `modality` is a MediaModality enum; compare by its string value.
|
|
4
|
+
.filter((detail) => String(detail.modality) === modality)
|
|
5
|
+
.reduce((sum, detail) => sum + (detail.tokenCount ?? 0), 0);
|
|
6
|
+
}
|
|
7
|
+
/**
|
|
8
|
+
* Normalize Gemini's usage metadata into smoltalk's TokenUsage, separating the
|
|
9
|
+
* audio bucket for the given direction so the token-priced cost engine can bill
|
|
10
|
+
* it at the audio rate. STT sets audioDirection "input" (audio prompt tokens);
|
|
11
|
+
* TTS sets "output" (audio candidate tokens).
|
|
12
|
+
*
|
|
13
|
+
* Thinking tokens (`thoughtsTokenCount`) are reported separately from
|
|
14
|
+
* `candidatesTokenCount` and are always text — they go into the text-output
|
|
15
|
+
* bucket, never the audio-output bucket.
|
|
16
|
+
*/
|
|
17
|
+
export function normalizeGoogleAudioUsage(metadata, audioDirection) {
|
|
18
|
+
if (metadata === undefined) {
|
|
19
|
+
return undefined;
|
|
20
|
+
}
|
|
21
|
+
const prompt = metadata.promptTokenCount ?? 0;
|
|
22
|
+
const candidates = metadata.candidatesTokenCount ?? 0;
|
|
23
|
+
const thoughts = metadata.thoughtsTokenCount ?? 0;
|
|
24
|
+
if (audioDirection === "input") {
|
|
25
|
+
const audioIn = modalityTokens(metadata.promptTokensDetails, "AUDIO");
|
|
26
|
+
const usage = {
|
|
27
|
+
inputTokens: Math.max(0, prompt - audioIn),
|
|
28
|
+
outputTokens: candidates + thoughts,
|
|
29
|
+
};
|
|
30
|
+
if (audioIn > 0)
|
|
31
|
+
usage.inputAudioTokens = audioIn;
|
|
32
|
+
if (metadata.totalTokenCount !== undefined)
|
|
33
|
+
usage.totalTokens = metadata.totalTokenCount;
|
|
34
|
+
return usage;
|
|
35
|
+
}
|
|
36
|
+
// Output (TTS): split the audio bucket out of candidate tokens. Only assume
|
|
37
|
+
// "all candidates are audio" when the details array is missing/empty; when it
|
|
38
|
+
// is present, trust its AUDIO total even if that total is 0 (otherwise
|
|
39
|
+
// purely-text candidate tokens would be mispriced at the audio rate).
|
|
40
|
+
const details = metadata.candidatesTokensDetails;
|
|
41
|
+
const detailsPresent = Array.isArray(details) && details.length > 0;
|
|
42
|
+
const audioOut = detailsPresent ? modalityTokens(details, "AUDIO") : candidates;
|
|
43
|
+
const usage = {
|
|
44
|
+
inputTokens: prompt,
|
|
45
|
+
outputTokens: Math.max(0, candidates - audioOut) + thoughts,
|
|
46
|
+
};
|
|
47
|
+
if (audioOut > 0)
|
|
48
|
+
usage.outputAudioTokens = audioOut;
|
|
49
|
+
if (metadata.totalTokenCount !== undefined)
|
|
50
|
+
usage.totalTokens = metadata.totalTokenCount;
|
|
51
|
+
return usage;
|
|
52
|
+
}
|
package/dist/util/mime.js
CHANGED
|
@@ -11,6 +11,8 @@ export const AUDIO_FORMATS = [
|
|
|
11
11
|
{ extension: "ogg", mimeType: "audio/ogg", aliasMimeTypes: [], aliasExtensions: [] },
|
|
12
12
|
{ extension: "flac", mimeType: "audio/flac", aliasMimeTypes: [], aliasExtensions: [] },
|
|
13
13
|
{ extension: "webm", mimeType: "audio/webm", aliasMimeTypes: [], aliasExtensions: [] },
|
|
14
|
+
{ extension: "aac", mimeType: "audio/aac", aliasMimeTypes: [], aliasExtensions: [] },
|
|
15
|
+
{ extension: "aiff", mimeType: "audio/aiff", aliasMimeTypes: ["audio/x-aiff"], aliasExtensions: ["aif"] },
|
|
14
16
|
];
|
|
15
17
|
const IMAGE_AND_DOCUMENT_EXT_TO_MIME = {
|
|
16
18
|
".png": "image/png",
|
package/dist/util/provider.d.ts
CHANGED
package/dist/util/provider.js
CHANGED
|
@@ -37,6 +37,8 @@ export function resolveApiKey(provider, config) {
|
|
|
37
37
|
return k?.liteLlm || process.env.LITELLM_API_KEY;
|
|
38
38
|
case "openai-compat":
|
|
39
39
|
return k?.openAiCompat || process.env.OPENAI_COMPAT_API_KEY;
|
|
40
|
+
case "groq":
|
|
41
|
+
return k?.groq || process.env.GROQ_API_KEY;
|
|
40
42
|
default:
|
|
41
43
|
return config.apiKey?.[provider];
|
|
42
44
|
}
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "smoltalk",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.11.0",
|
|
4
4
|
"description": "A common interface for LLM APIs",
|
|
5
5
|
"homepage": "https://github.com/egonSchiele/smoltalk",
|
|
6
6
|
"files": [
|
|
@@ -38,6 +38,14 @@
|
|
|
38
38
|
"devDependencies": {
|
|
39
39
|
"tsx": "^4.19.2"
|
|
40
40
|
},
|
|
41
|
+
"peerDependencies": {
|
|
42
|
+
"smoltalk-llama-cpp": ">=0.2.0 <1.0.0"
|
|
43
|
+
},
|
|
44
|
+
"peerDependenciesMeta": {
|
|
45
|
+
"smoltalk-llama-cpp": {
|
|
46
|
+
"optional": true
|
|
47
|
+
}
|
|
48
|
+
},
|
|
41
49
|
"scripts": {
|
|
42
50
|
"test": "vitest --exclude=**/*.live.test.ts",
|
|
43
51
|
"test:live": "vitest run lib/clients/*.live.test.ts lib/embed/*.live.test.ts lib/image/*.live.test.ts",
|