@tanstack/ai-groq 0.5.3 → 0.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +1 -0
- package/dist/esm/adapters/summarize.d.ts +51 -0
- package/dist/esm/adapters/summarize.js +55 -0
- package/dist/esm/adapters/summarize.js.map +1 -0
- package/dist/esm/adapters/text.js +89 -61
- package/dist/esm/adapters/text.js.map +1 -1
- package/dist/esm/adapters/transcription.js +210 -172
- package/dist/esm/adapters/transcription.js.map +1 -1
- package/dist/esm/adapters/tts.js +128 -66
- package/dist/esm/adapters/tts.js.map +1 -1
- package/dist/esm/audio/audio-provider-options.js +11 -8
- package/dist/esm/audio/audio-provider-options.js.map +1 -1
- package/dist/esm/index.d.ts +1 -0
- package/dist/esm/index.js +2 -15
- package/dist/esm/model-meta.js +300 -69
- package/dist/esm/model-meta.js.map +1 -1
- package/dist/esm/tools/function-tool.js +31 -26
- package/dist/esm/tools/function-tool.js.map +1 -1
- package/dist/esm/tools/index.js +1 -5
- package/dist/esm/tools/tool-converter.js +12 -7
- package/dist/esm/tools/tool-converter.js.map +1 -1
- package/dist/esm/utils/client.js +23 -16
- package/dist/esm/utils/client.js.map +1 -1
- package/dist/esm/utils/schema-converter.d.ts +0 -2
- package/dist/esm/utils/schema-converter.js +66 -66
- package/dist/esm/utils/schema-converter.js.map +1 -1
- package/package.json +7 -7
- package/src/adapters/summarize.ts +77 -0
- package/src/index.ts +8 -0
- package/src/utils/schema-converter.ts +0 -3
- package/dist/esm/index.js.map +0 -1
- package/dist/esm/tools/index.js.map +0 -1
|
@@ -1,179 +1,217 @@
|
|
|
1
|
+
import { getGroqApiKeyFromEnv, withGroqDefaults } from "../utils/client.js";
|
|
2
|
+
import { base64ToArrayBuffer, generateId } from "@tanstack/ai-utils";
|
|
1
3
|
import { BaseTranscriptionAdapter } from "@tanstack/ai/adapters";
|
|
2
|
-
|
|
3
|
-
|
|
4
|
+
//#region src/adapters/transcription.ts
|
|
5
|
+
/**
|
|
6
|
+
* Flattens the `openai` SDK's `HeadersLike` config value into a plain record so
|
|
7
|
+
* it can be merged into the raw `fetch` request this adapter issues. Handles
|
|
8
|
+
* the shapes callers actually pass (`Headers`, an entries array, or a plain
|
|
9
|
+
* object); null/undefined values are dropped.
|
|
10
|
+
*
|
|
11
|
+
* ponytail: doesn't unwrap the SDK's internal `NullableHeaders` class; forward
|
|
12
|
+
* that shape here if the SDK ever hands it to adapter config.
|
|
13
|
+
*/
|
|
4
14
|
function normalizeHeaders(headers) {
|
|
5
|
-
|
|
6
|
-
|
|
7
|
-
|
|
8
|
-
|
|
9
|
-
|
|
10
|
-
|
|
11
|
-
|
|
12
|
-
|
|
13
|
-
|
|
14
|
-
} else {
|
|
15
|
-
for (const [key, value] of Object.entries(headers)) assign(key, value);
|
|
16
|
-
}
|
|
17
|
-
return out;
|
|
18
|
-
}
|
|
19
|
-
class GroqTranscriptionAdapter extends BaseTranscriptionAdapter {
|
|
20
|
-
name = "groq";
|
|
21
|
-
apiKey;
|
|
22
|
-
baseURL;
|
|
23
|
-
defaultHeaders;
|
|
24
|
-
constructor(config, model) {
|
|
25
|
-
super(model, {});
|
|
26
|
-
const resolved = withGroqDefaults(config);
|
|
27
|
-
this.apiKey = resolved.apiKey;
|
|
28
|
-
this.baseURL = resolved.baseURL ?? "https://api.groq.com/openai/v1";
|
|
29
|
-
this.defaultHeaders = normalizeHeaders(resolved.defaultHeaders);
|
|
30
|
-
}
|
|
31
|
-
async transcribe(options) {
|
|
32
|
-
const { model, audio, language, prompt, responseFormat, modelOptions } = options;
|
|
33
|
-
if (responseFormat === "srt" || responseFormat === "vtt") {
|
|
34
|
-
throw new Error(
|
|
35
|
-
`Groq transcription does not support responseFormat='${responseFormat}'. Supported values: 'json', 'text', 'verbose_json'.`
|
|
36
|
-
);
|
|
37
|
-
}
|
|
38
|
-
const effectiveFormat = responseFormat ?? "verbose_json";
|
|
39
|
-
const useVerbose = effectiveFormat === "verbose_json";
|
|
40
|
-
const form = new FormData();
|
|
41
|
-
form.append("model", model);
|
|
42
|
-
form.append("response_format", effectiveFormat);
|
|
43
|
-
if (language !== void 0) form.append("language", language);
|
|
44
|
-
if (prompt !== void 0) form.append("prompt", prompt);
|
|
45
|
-
if (modelOptions?.temperature !== void 0) {
|
|
46
|
-
form.append("temperature", String(modelOptions.temperature));
|
|
47
|
-
}
|
|
48
|
-
if (modelOptions?.timestamp_granularities !== void 0) {
|
|
49
|
-
for (const g of modelOptions.timestamp_granularities) {
|
|
50
|
-
form.append("timestamp_granularities[]", g);
|
|
51
|
-
}
|
|
52
|
-
}
|
|
53
|
-
if (typeof audio === "string" && /^https?:\/\//.test(audio)) {
|
|
54
|
-
form.append("url", audio);
|
|
55
|
-
} else {
|
|
56
|
-
form.append("file", this.prepareAudioFile(audio));
|
|
57
|
-
}
|
|
58
|
-
try {
|
|
59
|
-
options.logger.request(
|
|
60
|
-
`activity=transcription provider=${this.name} model=${model} verbose=${useVerbose}`,
|
|
61
|
-
{ provider: this.name, model }
|
|
62
|
-
);
|
|
63
|
-
const response = await fetch(`${this.baseURL}/audio/transcriptions`, {
|
|
64
|
-
method: "POST",
|
|
65
|
-
headers: {
|
|
66
|
-
...this.defaultHeaders,
|
|
67
|
-
Authorization: `Bearer ${this.apiKey}`
|
|
68
|
-
},
|
|
69
|
-
body: form
|
|
70
|
-
});
|
|
71
|
-
if (!response.ok) {
|
|
72
|
-
const body = await response.json().catch(() => null);
|
|
73
|
-
const message = body?.error?.message ?? `Groq API error ${response.status}`;
|
|
74
|
-
throw new Error(message);
|
|
75
|
-
}
|
|
76
|
-
if (useVerbose) {
|
|
77
|
-
const data = await response.json();
|
|
78
|
-
const requestId = data.x_groq?.id ?? generateId(this.name);
|
|
79
|
-
const segments = data.segments?.map(
|
|
80
|
-
(seg) => ({
|
|
81
|
-
id: seg.id,
|
|
82
|
-
start: seg.start,
|
|
83
|
-
end: seg.end,
|
|
84
|
-
text: seg.text,
|
|
85
|
-
confidence: Math.exp(seg.avg_logprob)
|
|
86
|
-
})
|
|
87
|
-
);
|
|
88
|
-
const words = data.words?.map((w) => ({
|
|
89
|
-
word: w.word,
|
|
90
|
-
start: w.start,
|
|
91
|
-
end: w.end
|
|
92
|
-
}));
|
|
93
|
-
return {
|
|
94
|
-
id: requestId,
|
|
95
|
-
model,
|
|
96
|
-
text: data.text,
|
|
97
|
-
...data.language !== void 0 && { language: data.language },
|
|
98
|
-
...data.duration !== void 0 && { duration: data.duration },
|
|
99
|
-
...segments !== void 0 && { segments },
|
|
100
|
-
...words !== void 0 && { words }
|
|
101
|
-
};
|
|
102
|
-
} else if (effectiveFormat === "text") {
|
|
103
|
-
const text = await response.text();
|
|
104
|
-
return {
|
|
105
|
-
id: generateId(this.name),
|
|
106
|
-
model,
|
|
107
|
-
text,
|
|
108
|
-
...language !== void 0 && { language }
|
|
109
|
-
};
|
|
110
|
-
} else {
|
|
111
|
-
const data = await response.json();
|
|
112
|
-
return {
|
|
113
|
-
id: data.x_groq?.id ?? generateId(this.name),
|
|
114
|
-
model,
|
|
115
|
-
text: data.text,
|
|
116
|
-
...language !== void 0 && { language }
|
|
117
|
-
};
|
|
118
|
-
}
|
|
119
|
-
} catch (error) {
|
|
120
|
-
options.logger.errors(`${this.name}.transcribe fatal`, {
|
|
121
|
-
error,
|
|
122
|
-
source: `${this.name}.transcribe`
|
|
123
|
-
});
|
|
124
|
-
throw error;
|
|
125
|
-
}
|
|
126
|
-
}
|
|
127
|
-
prepareAudioFile(audio) {
|
|
128
|
-
if (typeof File !== "undefined" && audio instanceof File) {
|
|
129
|
-
return audio;
|
|
130
|
-
}
|
|
131
|
-
if (typeof Blob !== "undefined" && audio instanceof Blob) {
|
|
132
|
-
this.ensureFileSupport();
|
|
133
|
-
return new File([audio], "audio.mp3", {
|
|
134
|
-
type: audio.type || "audio/mpeg"
|
|
135
|
-
});
|
|
136
|
-
}
|
|
137
|
-
if (typeof ArrayBuffer !== "undefined" && audio instanceof ArrayBuffer) {
|
|
138
|
-
this.ensureFileSupport();
|
|
139
|
-
return new File([audio], "audio.mp3", { type: "audio/mpeg" });
|
|
140
|
-
}
|
|
141
|
-
if (typeof audio === "string") {
|
|
142
|
-
this.ensureFileSupport();
|
|
143
|
-
if (audio.startsWith("data:")) {
|
|
144
|
-
const parts = audio.split(",");
|
|
145
|
-
const header = parts[0];
|
|
146
|
-
const base64Data = parts[1] || "";
|
|
147
|
-
const mimeMatch = header?.match(/data:([^;]+)/);
|
|
148
|
-
const mimeType = mimeMatch?.[1] || "audio/mpeg";
|
|
149
|
-
const bytes2 = base64ToArrayBuffer(base64Data);
|
|
150
|
-
const extension = mimeType.split("/")[1] || "mp3";
|
|
151
|
-
return new File([bytes2], `audio.${extension}`, { type: mimeType });
|
|
152
|
-
}
|
|
153
|
-
const bytes = base64ToArrayBuffer(audio);
|
|
154
|
-
return new File([bytes], "audio.mp3", { type: "audio/mpeg" });
|
|
155
|
-
}
|
|
156
|
-
throw new Error("Invalid audio input type");
|
|
157
|
-
}
|
|
158
|
-
// Throws on Node < 20 where the global `File` constructor is unavailable.
|
|
159
|
-
ensureFileSupport() {
|
|
160
|
-
if (typeof File === "undefined") {
|
|
161
|
-
throw new Error(
|
|
162
|
-
"`File` is not available in this environment. Use Node.js 20 or newer, or pass a File object directly."
|
|
163
|
-
);
|
|
164
|
-
}
|
|
165
|
-
}
|
|
15
|
+
const out = {};
|
|
16
|
+
if (!headers) return out;
|
|
17
|
+
const assign = (key, value) => {
|
|
18
|
+
if (value != null) out[key] = String(value);
|
|
19
|
+
};
|
|
20
|
+
if (headers instanceof Headers) headers.forEach((value, key) => assign(key, value));
|
|
21
|
+
else if (Array.isArray(headers)) for (const [key, value] of headers) assign(key, value);
|
|
22
|
+
else for (const [key, value] of Object.entries(headers)) assign(key, value);
|
|
23
|
+
return out;
|
|
166
24
|
}
|
|
25
|
+
/**
|
|
26
|
+
* Groq Transcription (Speech-to-Text) Adapter
|
|
27
|
+
*
|
|
28
|
+
* Tree-shakeable adapter for Groq audio transcription. Supports
|
|
29
|
+
* whisper-large-v3 and whisper-large-v3-turbo.
|
|
30
|
+
*
|
|
31
|
+
* Features:
|
|
32
|
+
* - Audio file uploads (File, Blob, ArrayBuffer, base64/data URL)
|
|
33
|
+
* - Remote audio URLs passed directly via Groq's `url` field — no upload needed
|
|
34
|
+
* - Verbose JSON response with segment and word timestamps
|
|
35
|
+
* - Language detection or specification (ISO-639-1)
|
|
36
|
+
* - Confidence scores derived from segment avg_logprob
|
|
37
|
+
*/
|
|
38
|
+
var GroqTranscriptionAdapter = class extends BaseTranscriptionAdapter {
|
|
39
|
+
name = "groq";
|
|
40
|
+
apiKey;
|
|
41
|
+
baseURL;
|
|
42
|
+
defaultHeaders;
|
|
43
|
+
constructor(config, model) {
|
|
44
|
+
super(model, {});
|
|
45
|
+
const resolved = withGroqDefaults(config);
|
|
46
|
+
this.apiKey = resolved.apiKey;
|
|
47
|
+
this.baseURL = resolved.baseURL ?? "https://api.groq.com/openai/v1";
|
|
48
|
+
this.defaultHeaders = normalizeHeaders(resolved.defaultHeaders);
|
|
49
|
+
}
|
|
50
|
+
async transcribe(options) {
|
|
51
|
+
const { model, audio, language, prompt, responseFormat, modelOptions } = options;
|
|
52
|
+
if (responseFormat === "srt" || responseFormat === "vtt") throw new Error(`Groq transcription does not support responseFormat='${responseFormat}'. Supported values: 'json', 'text', 'verbose_json'.`);
|
|
53
|
+
const effectiveFormat = responseFormat ?? "verbose_json";
|
|
54
|
+
const useVerbose = effectiveFormat === "verbose_json";
|
|
55
|
+
const form = new FormData();
|
|
56
|
+
form.append("model", model);
|
|
57
|
+
form.append("response_format", effectiveFormat);
|
|
58
|
+
if (language !== void 0) form.append("language", language);
|
|
59
|
+
if (prompt !== void 0) form.append("prompt", prompt);
|
|
60
|
+
if (modelOptions?.temperature !== void 0) form.append("temperature", String(modelOptions.temperature));
|
|
61
|
+
if (modelOptions?.timestamp_granularities !== void 0) for (const g of modelOptions.timestamp_granularities) form.append("timestamp_granularities[]", g);
|
|
62
|
+
if (typeof audio === "string" && /^https?:\/\//.test(audio)) form.append("url", audio);
|
|
63
|
+
else form.append("file", this.prepareAudioFile(audio));
|
|
64
|
+
try {
|
|
65
|
+
options.logger.request(`activity=transcription provider=${this.name} model=${model} verbose=${useVerbose}`, {
|
|
66
|
+
provider: this.name,
|
|
67
|
+
model
|
|
68
|
+
});
|
|
69
|
+
const response = await fetch(`${this.baseURL}/audio/transcriptions`, {
|
|
70
|
+
method: "POST",
|
|
71
|
+
headers: {
|
|
72
|
+
...this.defaultHeaders,
|
|
73
|
+
Authorization: `Bearer ${this.apiKey}`
|
|
74
|
+
},
|
|
75
|
+
body: form
|
|
76
|
+
});
|
|
77
|
+
if (!response.ok) {
|
|
78
|
+
const message = ((await response.json().catch(() => null))?.error)?.message ?? `Groq API error ${response.status}`;
|
|
79
|
+
throw new Error(message);
|
|
80
|
+
}
|
|
81
|
+
if (useVerbose) {
|
|
82
|
+
const data = await response.json();
|
|
83
|
+
const requestId = data.x_groq?.id ?? generateId(this.name);
|
|
84
|
+
const segments = data.segments?.map((seg) => ({
|
|
85
|
+
id: seg.id,
|
|
86
|
+
start: seg.start,
|
|
87
|
+
end: seg.end,
|
|
88
|
+
text: seg.text,
|
|
89
|
+
confidence: Math.exp(seg.avg_logprob)
|
|
90
|
+
}));
|
|
91
|
+
const words = data.words?.map((w) => ({
|
|
92
|
+
word: w.word,
|
|
93
|
+
start: w.start,
|
|
94
|
+
end: w.end
|
|
95
|
+
}));
|
|
96
|
+
return {
|
|
97
|
+
id: requestId,
|
|
98
|
+
model,
|
|
99
|
+
text: data.text,
|
|
100
|
+
...data.language !== void 0 && { language: data.language },
|
|
101
|
+
...data.duration !== void 0 && { duration: data.duration },
|
|
102
|
+
...segments !== void 0 && { segments },
|
|
103
|
+
...words !== void 0 && { words }
|
|
104
|
+
};
|
|
105
|
+
} else if (effectiveFormat === "text") {
|
|
106
|
+
const text = await response.text();
|
|
107
|
+
return {
|
|
108
|
+
id: generateId(this.name),
|
|
109
|
+
model,
|
|
110
|
+
text,
|
|
111
|
+
...language !== void 0 && { language }
|
|
112
|
+
};
|
|
113
|
+
} else {
|
|
114
|
+
const data = await response.json();
|
|
115
|
+
return {
|
|
116
|
+
id: data.x_groq?.id ?? generateId(this.name),
|
|
117
|
+
model,
|
|
118
|
+
text: data.text,
|
|
119
|
+
...language !== void 0 && { language }
|
|
120
|
+
};
|
|
121
|
+
}
|
|
122
|
+
} catch (error) {
|
|
123
|
+
options.logger.errors(`${this.name}.transcribe fatal`, {
|
|
124
|
+
error,
|
|
125
|
+
source: `${this.name}.transcribe`
|
|
126
|
+
});
|
|
127
|
+
throw error;
|
|
128
|
+
}
|
|
129
|
+
}
|
|
130
|
+
prepareAudioFile(audio) {
|
|
131
|
+
if (typeof File !== "undefined" && audio instanceof File) return audio;
|
|
132
|
+
if (typeof Blob !== "undefined" && audio instanceof Blob) {
|
|
133
|
+
this.ensureFileSupport();
|
|
134
|
+
return new File([audio], "audio.mp3", { type: audio.type || "audio/mpeg" });
|
|
135
|
+
}
|
|
136
|
+
if (typeof ArrayBuffer !== "undefined" && audio instanceof ArrayBuffer) {
|
|
137
|
+
this.ensureFileSupport();
|
|
138
|
+
return new File([audio], "audio.mp3", { type: "audio/mpeg" });
|
|
139
|
+
}
|
|
140
|
+
if (typeof audio === "string") {
|
|
141
|
+
this.ensureFileSupport();
|
|
142
|
+
if (audio.startsWith("data:")) {
|
|
143
|
+
const parts = audio.split(",");
|
|
144
|
+
const header = parts[0];
|
|
145
|
+
const base64Data = parts[1] || "";
|
|
146
|
+
const mimeType = (header?.match(/data:([^;]+)/))?.[1] || "audio/mpeg";
|
|
147
|
+
const bytes = base64ToArrayBuffer(base64Data);
|
|
148
|
+
const extension = mimeType.split("/")[1] || "mp3";
|
|
149
|
+
return new File([bytes], `audio.${extension}`, { type: mimeType });
|
|
150
|
+
}
|
|
151
|
+
const bytes = base64ToArrayBuffer(audio);
|
|
152
|
+
return new File([bytes], "audio.mp3", { type: "audio/mpeg" });
|
|
153
|
+
}
|
|
154
|
+
throw new Error("Invalid audio input type");
|
|
155
|
+
}
|
|
156
|
+
ensureFileSupport() {
|
|
157
|
+
if (typeof File === "undefined") throw new Error("`File` is not available in this environment. Use Node.js 20 or newer, or pass a File object directly.");
|
|
158
|
+
}
|
|
159
|
+
};
|
|
160
|
+
/**
|
|
161
|
+
* Creates a Groq transcription adapter with an explicit API key.
|
|
162
|
+
* Type resolution happens here at the call site.
|
|
163
|
+
*
|
|
164
|
+
* @param model - The model name (e.g., 'whisper-large-v3-turbo')
|
|
165
|
+
* @param apiKey - Your Groq API key
|
|
166
|
+
* @param config - Optional additional configuration
|
|
167
|
+
* @returns Configured Groq transcription adapter instance
|
|
168
|
+
*
|
|
169
|
+
* @example
|
|
170
|
+
* ```typescript
|
|
171
|
+
* const adapter = createGroqTranscription('whisper-large-v3-turbo', 'gsk_...');
|
|
172
|
+
*
|
|
173
|
+
* const result = await generateTranscription({
|
|
174
|
+
* adapter,
|
|
175
|
+
* audio: audioFile,
|
|
176
|
+
* language: 'en',
|
|
177
|
+
* });
|
|
178
|
+
* ```
|
|
179
|
+
*/
|
|
167
180
|
function createGroqTranscription(model, apiKey, config) {
|
|
168
|
-
|
|
181
|
+
return new GroqTranscriptionAdapter({
|
|
182
|
+
apiKey,
|
|
183
|
+
...config
|
|
184
|
+
}, model);
|
|
169
185
|
}
|
|
186
|
+
/**
|
|
187
|
+
* Creates a Groq transcription adapter using the `GROQ_API_KEY` environment
|
|
188
|
+
* variable. Type resolution happens here at the call site.
|
|
189
|
+
*
|
|
190
|
+
* Looks for `GROQ_API_KEY` in:
|
|
191
|
+
* - `process.env` (Node.js)
|
|
192
|
+
* - `window.env` (browser with injected env)
|
|
193
|
+
*
|
|
194
|
+
* @param model - The model name (e.g., 'whisper-large-v3-turbo')
|
|
195
|
+
* @param config - Optional configuration (excluding apiKey which is auto-detected)
|
|
196
|
+
* @returns Configured Groq transcription adapter instance
|
|
197
|
+
* @throws Error if GROQ_API_KEY is not found in environment
|
|
198
|
+
*
|
|
199
|
+
* @example
|
|
200
|
+
* ```typescript
|
|
201
|
+
* const adapter = groqTranscription('whisper-large-v3-turbo');
|
|
202
|
+
*
|
|
203
|
+
* const result = await generateTranscription({
|
|
204
|
+
* adapter,
|
|
205
|
+
* audio: 'https://example.com/audio.mp3',
|
|
206
|
+
* });
|
|
207
|
+
*
|
|
208
|
+
* console.log(result.text)
|
|
209
|
+
* ```
|
|
210
|
+
*/
|
|
170
211
|
function groqTranscription(model, config) {
|
|
171
|
-
|
|
172
|
-
return createGroqTranscription(model, apiKey, config);
|
|
212
|
+
return createGroqTranscription(model, getGroqApiKeyFromEnv(), config);
|
|
173
213
|
}
|
|
174
|
-
|
|
175
|
-
|
|
176
|
-
|
|
177
|
-
|
|
178
|
-
};
|
|
179
|
-
//# sourceMappingURL=transcription.js.map
|
|
214
|
+
//#endregion
|
|
215
|
+
export { GroqTranscriptionAdapter, createGroqTranscription, groqTranscription };
|
|
216
|
+
|
|
217
|
+
//# sourceMappingURL=transcription.js.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"transcription.js","sources":["../../../src/adapters/transcription.ts"],"sourcesContent":["import { BaseTranscriptionAdapter } from '@tanstack/ai/adapters'\nimport { base64ToArrayBuffer, generateId } from '@tanstack/ai-utils'\nimport { getGroqApiKeyFromEnv, withGroqDefaults } from '../utils/client'\nimport type {\n TranscriptionOptions,\n TranscriptionResult,\n TranscriptionSegment,\n} from '@tanstack/ai'\nimport type { GroqTranscriptionModel } from '../model-meta'\nimport type { GroqTranscriptionProviderOptions } from '../audio/transcription-provider-options'\nimport type { GroqClientConfig } from '../utils/client'\n\n/**\n * Configuration for the Groq Transcription adapter.\n */\nexport interface GroqTranscriptionConfig extends GroqClientConfig {}\n\n/**\n * Flattens the `openai` SDK's `HeadersLike` config value into a plain record so\n * it can be merged into the raw `fetch` request this adapter issues. Handles\n * the shapes callers actually pass (`Headers`, an entries array, or a plain\n * object); null/undefined values are dropped.\n *\n * ponytail: doesn't unwrap the SDK's internal `NullableHeaders` class; forward\n * that shape here if the SDK ever hands it to adapter config.\n */\nfunction normalizeHeaders(\n headers: GroqTranscriptionConfig['defaultHeaders'],\n): Record<string, string> {\n const out: Record<string, string> = {}\n if (!headers) return out\n const assign = (key: string, value: unknown) => {\n if (value != null) out[key] = String(value)\n }\n if (headers instanceof Headers) {\n headers.forEach((value, key) => assign(key, value))\n } else if (Array.isArray(headers)) {\n for (const [key, value] of headers) assign(key, value)\n } else {\n for (const [key, value] of Object.entries(headers)) assign(key, value)\n }\n return out\n}\n\n// Shape of Groq's verbose_json transcription response\ninterface GroqVerboseTranscriptionResponse {\n task?: string\n language?: string\n duration?: number\n text: string\n segments?: Array<{\n id: number\n seek?: number\n start: number\n end: number\n text: string\n tokens?: Array<number>\n temperature?: number\n avg_logprob: number\n compression_ratio?: number\n no_speech_prob?: number\n }>\n words?: Array<{ word: string; start: number; end: number }>\n x_groq?: { id?: string }\n}\n\n// Shape of Groq's json transcription response\ninterface GroqJsonTranscriptionResponse {\n text: string\n x_groq?: { id?: string }\n}\n\n/**\n * Groq Transcription (Speech-to-Text) Adapter\n *\n * Tree-shakeable adapter for Groq audio transcription. Supports\n * whisper-large-v3 and whisper-large-v3-turbo.\n *\n * Features:\n * - Audio file uploads (File, Blob, ArrayBuffer, base64/data URL)\n * - Remote audio URLs passed directly via Groq's `url` field — no upload needed\n * - Verbose JSON response with segment and word timestamps\n * - Language detection or specification (ISO-639-1)\n * - Confidence scores derived from segment avg_logprob\n */\nexport class GroqTranscriptionAdapter<\n TModel extends GroqTranscriptionModel,\n> extends BaseTranscriptionAdapter<TModel, GroqTranscriptionProviderOptions> {\n readonly name = 'groq' as const\n\n private readonly apiKey: string\n private readonly baseURL: string\n private readonly defaultHeaders: Record<string, string>\n\n constructor(config: GroqTranscriptionConfig, model: TModel) {\n super(model, {})\n const resolved = withGroqDefaults(config)\n this.apiKey = resolved.apiKey\n this.baseURL = resolved.baseURL ?? 'https://api.groq.com/openai/v1'\n this.defaultHeaders = normalizeHeaders(resolved.defaultHeaders)\n }\n\n async transcribe(\n options: TranscriptionOptions<GroqTranscriptionProviderOptions>,\n ): Promise<TranscriptionResult> {\n const { model, audio, language, prompt, responseFormat, modelOptions } =\n options\n\n // Groq's transcription endpoint only accepts 'json', 'text', and\n // 'verbose_json'. Reject 'srt'/'vtt' up front so callers get a clear\n // message instead of an opaque Groq HTTP error.\n if (responseFormat === 'srt' || responseFormat === 'vtt') {\n throw new Error(\n `Groq transcription does not support responseFormat='${responseFormat}'. ` +\n `Supported values: 'json', 'text', 'verbose_json'.`,\n )\n }\n\n // Default to verbose_json so callers get language, duration, and timestamps\n // without having to opt in explicitly. Both Groq whisper models support it.\n const effectiveFormat = responseFormat ?? 'verbose_json'\n const useVerbose = effectiveFormat === 'verbose_json'\n\n const form = new FormData()\n form.append('model', model)\n form.append('response_format', effectiveFormat)\n if (language !== undefined) form.append('language', language)\n if (prompt !== undefined) form.append('prompt', prompt)\n if (modelOptions?.temperature !== undefined) {\n form.append('temperature', String(modelOptions.temperature))\n }\n if (modelOptions?.timestamp_granularities !== undefined) {\n for (const g of modelOptions.timestamp_granularities) {\n form.append('timestamp_granularities[]', g)\n }\n }\n\n // HTTP/HTTPS URLs are forwarded directly via Groq's `url` field, which\n // avoids a round-trip upload. All other inputs (File, Blob, ArrayBuffer,\n // base64, data URL) are converted to a File and sent as `file`.\n if (typeof audio === 'string' && /^https?:\\/\\//.test(audio)) {\n form.append('url', audio)\n } else {\n form.append('file', this.prepareAudioFile(audio))\n }\n\n try {\n options.logger.request(\n `activity=transcription provider=${this.name} model=${model} verbose=${useVerbose}`,\n { provider: this.name, model },\n )\n\n const response = await fetch(`${this.baseURL}/audio/transcriptions`, {\n method: 'POST',\n headers: {\n ...this.defaultHeaders,\n Authorization: `Bearer ${this.apiKey}`,\n },\n body: form,\n })\n\n if (!response.ok) {\n const body = await response\n .json()\n .catch(() => null as Record<string, unknown> | null)\n const message =\n (body?.error as { message?: string } | undefined)?.message ??\n `Groq API error ${response.status}`\n throw new Error(message)\n }\n\n if (useVerbose) {\n const data = (await response.json()) as GroqVerboseTranscriptionResponse\n const requestId = data.x_groq?.id ?? generateId(this.name)\n\n // `TranscriptionResult` declares optional fields without `| undefined`,\n // so under exactOptionalPropertyTypes we must omit absent fields rather\n // than assigning `undefined`.\n const segments = data.segments?.map(\n (seg): TranscriptionSegment => ({\n id: seg.id,\n start: seg.start,\n end: seg.end,\n text: seg.text,\n confidence: Math.exp(seg.avg_logprob),\n }),\n )\n const words = data.words?.map((w) => ({\n word: w.word,\n start: w.start,\n end: w.end,\n }))\n\n return {\n id: requestId,\n model,\n text: data.text,\n ...(data.language !== undefined && { language: data.language }),\n ...(data.duration !== undefined && { duration: data.duration }),\n ...(segments !== undefined && { segments }),\n ...(words !== undefined && { words }),\n }\n } else if (effectiveFormat === 'text') {\n const text = await response.text()\n return {\n id: generateId(this.name),\n model,\n text,\n ...(language !== undefined && { language }),\n }\n } else {\n const data = (await response.json()) as GroqJsonTranscriptionResponse\n return {\n id: data.x_groq?.id ?? generateId(this.name),\n model,\n text: data.text,\n ...(language !== undefined && { language }),\n }\n }\n } catch (error: unknown) {\n options.logger.errors(`${this.name}.transcribe fatal`, {\n error,\n source: `${this.name}.transcribe`,\n })\n throw error\n }\n }\n\n private prepareAudioFile(audio: string | File | Blob | ArrayBuffer): File {\n if (typeof File !== 'undefined' && audio instanceof File) {\n return audio\n }\n if (typeof Blob !== 'undefined' && audio instanceof Blob) {\n this.ensureFileSupport()\n return new File([audio], 'audio.mp3', {\n type: audio.type || 'audio/mpeg',\n })\n }\n if (typeof ArrayBuffer !== 'undefined' && audio instanceof ArrayBuffer) {\n this.ensureFileSupport()\n return new File([audio], 'audio.mp3', { type: 'audio/mpeg' })\n }\n if (typeof audio === 'string') {\n this.ensureFileSupport()\n\n if (audio.startsWith('data:')) {\n const parts = audio.split(',')\n const header = parts[0]\n const base64Data = parts[1] || ''\n const mimeMatch = header?.match(/data:([^;]+)/)\n const mimeType = mimeMatch?.[1] || 'audio/mpeg'\n const bytes = base64ToArrayBuffer(base64Data)\n const extension = mimeType.split('/')[1] || 'mp3'\n return new File([bytes], `audio.${extension}`, { type: mimeType })\n }\n\n const bytes = base64ToArrayBuffer(audio)\n return new File([bytes], 'audio.mp3', { type: 'audio/mpeg' })\n }\n\n throw new Error('Invalid audio input type')\n }\n\n // Throws on Node < 20 where the global `File` constructor is unavailable.\n private ensureFileSupport(): void {\n if (typeof File === 'undefined') {\n throw new Error(\n '`File` is not available in this environment. ' +\n 'Use Node.js 20 or newer, or pass a File object directly.',\n )\n }\n }\n}\n\n/**\n * Creates a Groq transcription adapter with an explicit API key.\n * Type resolution happens here at the call site.\n *\n * @param model - The model name (e.g., 'whisper-large-v3-turbo')\n * @param apiKey - Your Groq API key\n * @param config - Optional additional configuration\n * @returns Configured Groq transcription adapter instance\n *\n * @example\n * ```typescript\n * const adapter = createGroqTranscription('whisper-large-v3-turbo', 'gsk_...');\n *\n * const result = await generateTranscription({\n * adapter,\n * audio: audioFile,\n * language: 'en',\n * });\n * ```\n */\nexport function createGroqTranscription<TModel extends GroqTranscriptionModel>(\n model: TModel,\n apiKey: string,\n config?: Omit<GroqTranscriptionConfig, 'apiKey'>,\n): GroqTranscriptionAdapter<TModel> {\n return new GroqTranscriptionAdapter({ apiKey, ...config }, model)\n}\n\n/**\n * Creates a Groq transcription adapter using the `GROQ_API_KEY` environment\n * variable. Type resolution happens here at the call site.\n *\n * Looks for `GROQ_API_KEY` in:\n * - `process.env` (Node.js)\n * - `window.env` (browser with injected env)\n *\n * @param model - The model name (e.g., 'whisper-large-v3-turbo')\n * @param config - Optional configuration (excluding apiKey which is auto-detected)\n * @returns Configured Groq transcription adapter instance\n * @throws Error if GROQ_API_KEY is not found in environment\n *\n * @example\n * ```typescript\n * const adapter = groqTranscription('whisper-large-v3-turbo');\n *\n * const result = await generateTranscription({\n * adapter,\n * audio: 'https://example.com/audio.mp3',\n * });\n *\n * console.log(result.text)\n * ```\n */\nexport function groqTranscription<TModel extends GroqTranscriptionModel>(\n model: TModel,\n config?: Omit<GroqTranscriptionConfig, 'apiKey'>,\n): GroqTranscriptionAdapter<TModel> {\n const apiKey = getGroqApiKeyFromEnv()\n return createGroqTranscription(model, apiKey, config)\n}\n"],"names":["bytes"],"mappings":";;;AA0BA,SAAS,iBACP,SACwB;AACxB,QAAM,MAA8B,CAAA;AACpC,MAAI,CAAC,QAAS,QAAO;AACrB,QAAM,SAAS,CAAC,KAAa,UAAmB;AAC9C,QAAI,SAAS,KAAM,KAAI,GAAG,IAAI,OAAO,KAAK;AAAA,EAC5C;AACA,MAAI,mBAAmB,SAAS;AAC9B,YAAQ,QAAQ,CAAC,OAAO,QAAQ,OAAO,KAAK,KAAK,CAAC;AAAA,EACpD,WAAW,MAAM,QAAQ,OAAO,GAAG;AACjC,eAAW,CAAC,KAAK,KAAK,KAAK,QAAS,QAAO,KAAK,KAAK;AAAA,EACvD,OAAO;AACL,eAAW,CAAC,KAAK,KAAK,KAAK,OAAO,QAAQ,OAAO,EAAG,QAAO,KAAK,KAAK;AAAA,EACvE;AACA,SAAO;AACT;AA2CO,MAAM,iCAEH,yBAAmE;AAAA,EAClE,OAAO;AAAA,EAEC;AAAA,EACA;AAAA,EACA;AAAA,EAEjB,YAAY,QAAiC,OAAe;AAC1D,UAAM,OAAO,EAAE;AACf,UAAM,WAAW,iBAAiB,MAAM;AACxC,SAAK,SAAS,SAAS;AACvB,SAAK,UAAU,SAAS,WAAW;AACnC,SAAK,iBAAiB,iBAAiB,SAAS,cAAc;AAAA,EAChE;AAAA,EAEA,MAAM,WACJ,SAC8B;AAC9B,UAAM,EAAE,OAAO,OAAO,UAAU,QAAQ,gBAAgB,iBACtD;AAKF,QAAI,mBAAmB,SAAS,mBAAmB,OAAO;AACxD,YAAM,IAAI;AAAA,QACR,uDAAuD,cAAc;AAAA,MAAA;AAAA,IAGzE;AAIA,UAAM,kBAAkB,kBAAkB;AAC1C,UAAM,aAAa,oBAAoB;AAEvC,UAAM,OAAO,IAAI,SAAA;AACjB,SAAK,OAAO,SAAS,KAAK;AAC1B,SAAK,OAAO,mBAAmB,eAAe;AAC9C,QAAI,aAAa,OAAW,MAAK,OAAO,YAAY,QAAQ;AAC5D,QAAI,WAAW,OAAW,MAAK,OAAO,UAAU,MAAM;AACtD,QAAI,cAAc,gBAAgB,QAAW;AAC3C,WAAK,OAAO,eAAe,OAAO,aAAa,WAAW,CAAC;AAAA,IAC7D;AACA,QAAI,cAAc,4BAA4B,QAAW;AACvD,iBAAW,KAAK,aAAa,yBAAyB;AACpD,aAAK,OAAO,6BAA6B,CAAC;AAAA,MAC5C;AAAA,IACF;AAKA,QAAI,OAAO,UAAU,YAAY,eAAe,KAAK,KAAK,GAAG;AAC3D,WAAK,OAAO,OAAO,KAAK;AAAA,IAC1B,OAAO;AACL,WAAK,OAAO,QAAQ,KAAK,iBAAiB,KAAK,CAAC;AAAA,IAClD;AAEA,QAAI;AACF,cAAQ,OAAO;AAAA,QACb,mCAAmC,KAAK,IAAI,UAAU,KAAK,YAAY,UAAU;AAAA,QACjF,EAAE,UAAU,KAAK,MAAM,MAAA;AAAA,MAAM;AAG/B,YAAM,WAAW,MAAM,MAAM,GAAG,KAAK,OAAO,yBAAyB;AAAA,QACnE,QAAQ;AAAA,QACR,SAAS;AAAA,UACP,GAAG,KAAK;AAAA,UACR,eAAe,UAAU,KAAK,MAAM;AAAA,QAAA;AAAA,QAEtC,MAAM;AAAA,MAAA,CACP;AAED,UAAI,CAAC,SAAS,IAAI;AAChB,cAAM,OAAO,MAAM,SAChB,OACA,MAAM,MAAM,IAAsC;AACrD,cAAM,UACH,MAAM,OAA4C,WACnD,kBAAkB,SAAS,MAAM;AACnC,cAAM,IAAI,MAAM,OAAO;AAAA,MACzB;AAEA,UAAI,YAAY;AACd,cAAM,OAAQ,MAAM,SAAS,KAAA;AAC7B,cAAM,YAAY,KAAK,QAAQ,MAAM,WAAW,KAAK,IAAI;AAKzD,cAAM,WAAW,KAAK,UAAU;AAAA,UAC9B,CAAC,SAA+B;AAAA,YAC9B,IAAI,IAAI;AAAA,YACR,OAAO,IAAI;AAAA,YACX,KAAK,IAAI;AAAA,YACT,MAAM,IAAI;AAAA,YACV,YAAY,KAAK,IAAI,IAAI,WAAW;AAAA,UAAA;AAAA,QACtC;AAEF,cAAM,QAAQ,KAAK,OAAO,IAAI,CAAC,OAAO;AAAA,UACpC,MAAM,EAAE;AAAA,UACR,OAAO,EAAE;AAAA,UACT,KAAK,EAAE;AAAA,QAAA,EACP;AAEF,eAAO;AAAA,UACL,IAAI;AAAA,UACJ;AAAA,UACA,MAAM,KAAK;AAAA,UACX,GAAI,KAAK,aAAa,UAAa,EAAE,UAAU,KAAK,SAAA;AAAA,UACpD,GAAI,KAAK,aAAa,UAAa,EAAE,UAAU,KAAK,SAAA;AAAA,UACpD,GAAI,aAAa,UAAa,EAAE,SAAA;AAAA,UAChC,GAAI,UAAU,UAAa,EAAE,MAAA;AAAA,QAAM;AAAA,MAEvC,WAAW,oBAAoB,QAAQ;AACrC,cAAM,OAAO,MAAM,SAAS,KAAA;AAC5B,eAAO;AAAA,UACL,IAAI,WAAW,KAAK,IAAI;AAAA,UACxB;AAAA,UACA;AAAA,UACA,GAAI,aAAa,UAAa,EAAE,SAAA;AAAA,QAAS;AAAA,MAE7C,OAAO;AACL,cAAM,OAAQ,MAAM,SAAS,KAAA;AAC7B,eAAO;AAAA,UACL,IAAI,KAAK,QAAQ,MAAM,WAAW,KAAK,IAAI;AAAA,UAC3C;AAAA,UACA,MAAM,KAAK;AAAA,UACX,GAAI,aAAa,UAAa,EAAE,SAAA;AAAA,QAAS;AAAA,MAE7C;AAAA,IACF,SAAS,OAAgB;AACvB,cAAQ,OAAO,OAAO,GAAG,KAAK,IAAI,qBAAqB;AAAA,QACrD;AAAA,QACA,QAAQ,GAAG,KAAK,IAAI;AAAA,MAAA,CACrB;AACD,YAAM;AAAA,IACR;AAAA,EACF;AAAA,EAEQ,iBAAiB,OAAiD;AACxE,QAAI,OAAO,SAAS,eAAe,iBAAiB,MAAM;AACxD,aAAO;AAAA,IACT;AACA,QAAI,OAAO,SAAS,eAAe,iBAAiB,MAAM;AACxD,WAAK,kBAAA;AACL,aAAO,IAAI,KAAK,CAAC,KAAK,GAAG,aAAa;AAAA,QACpC,MAAM,MAAM,QAAQ;AAAA,MAAA,CACrB;AAAA,IACH;AACA,QAAI,OAAO,gBAAgB,eAAe,iBAAiB,aAAa;AACtE,WAAK,kBAAA;AACL,aAAO,IAAI,KAAK,CAAC,KAAK,GAAG,aAAa,EAAE,MAAM,cAAc;AAAA,IAC9D;AACA,QAAI,OAAO,UAAU,UAAU;AAC7B,WAAK,kBAAA;AAEL,UAAI,MAAM,WAAW,OAAO,GAAG;AAC7B,cAAM,QAAQ,MAAM,MAAM,GAAG;AAC7B,cAAM,SAAS,MAAM,CAAC;AACtB,cAAM,aAAa,MAAM,CAAC,KAAK;AAC/B,cAAM,YAAY,QAAQ,MAAM,cAAc;AAC9C,cAAM,WAAW,YAAY,CAAC,KAAK;AACnC,cAAMA,SAAQ,oBAAoB,UAAU;AAC5C,cAAM,YAAY,SAAS,MAAM,GAAG,EAAE,CAAC,KAAK;AAC5C,eAAO,IAAI,KAAK,CAACA,MAAK,GAAG,SAAS,SAAS,IAAI,EAAE,MAAM,UAAU;AAAA,MACnE;AAEA,YAAM,QAAQ,oBAAoB,KAAK;AACvC,aAAO,IAAI,KAAK,CAAC,KAAK,GAAG,aAAa,EAAE,MAAM,cAAc;AAAA,IAC9D;AAEA,UAAM,IAAI,MAAM,0BAA0B;AAAA,EAC5C;AAAA;AAAA,EAGQ,oBAA0B;AAChC,QAAI,OAAO,SAAS,aAAa;AAC/B,YAAM,IAAI;AAAA,QACR;AAAA,MAAA;AAAA,IAGJ;AAAA,EACF;AACF;AAsBO,SAAS,wBACd,OACA,QACA,QACkC;AAClC,SAAO,IAAI,yBAAyB,EAAE,QAAQ,GAAG,OAAA,GAAU,KAAK;AAClE;AA2BO,SAAS,kBACd,OACA,QACkC;AAClC,QAAM,SAAS,qBAAA;AACf,SAAO,wBAAwB,OAAO,QAAQ,MAAM;AACtD;"}
|
|
1
|
+
{"version":3,"file":"transcription.js","names":[],"sources":["../../../src/adapters/transcription.ts"],"sourcesContent":["import { BaseTranscriptionAdapter } from '@tanstack/ai/adapters'\nimport { base64ToArrayBuffer, generateId } from '@tanstack/ai-utils'\nimport { getGroqApiKeyFromEnv, withGroqDefaults } from '../utils/client'\nimport type {\n TranscriptionOptions,\n TranscriptionResult,\n TranscriptionSegment,\n} from '@tanstack/ai'\nimport type { GroqTranscriptionModel } from '../model-meta'\nimport type { GroqTranscriptionProviderOptions } from '../audio/transcription-provider-options'\nimport type { GroqClientConfig } from '../utils/client'\n\n/**\n * Configuration for the Groq Transcription adapter.\n */\nexport interface GroqTranscriptionConfig extends GroqClientConfig {}\n\n/**\n * Flattens the `openai` SDK's `HeadersLike` config value into a plain record so\n * it can be merged into the raw `fetch` request this adapter issues. Handles\n * the shapes callers actually pass (`Headers`, an entries array, or a plain\n * object); null/undefined values are dropped.\n *\n * ponytail: doesn't unwrap the SDK's internal `NullableHeaders` class; forward\n * that shape here if the SDK ever hands it to adapter config.\n */\nfunction normalizeHeaders(\n headers: GroqTranscriptionConfig['defaultHeaders'],\n): Record<string, string> {\n const out: Record<string, string> = {}\n if (!headers) return out\n const assign = (key: string, value: unknown) => {\n if (value != null) out[key] = String(value)\n }\n if (headers instanceof Headers) {\n headers.forEach((value, key) => assign(key, value))\n } else if (Array.isArray(headers)) {\n for (const [key, value] of headers) assign(key, value)\n } else {\n for (const [key, value] of Object.entries(headers)) assign(key, value)\n }\n return out\n}\n\n// Shape of Groq's verbose_json transcription response\ninterface GroqVerboseTranscriptionResponse {\n task?: string\n language?: string\n duration?: number\n text: string\n segments?: Array<{\n id: number\n seek?: number\n start: number\n end: number\n text: string\n tokens?: Array<number>\n temperature?: number\n avg_logprob: number\n compression_ratio?: number\n no_speech_prob?: number\n }>\n words?: Array<{ word: string; start: number; end: number }>\n x_groq?: { id?: string }\n}\n\n// Shape of Groq's json transcription response\ninterface GroqJsonTranscriptionResponse {\n text: string\n x_groq?: { id?: string }\n}\n\n/**\n * Groq Transcription (Speech-to-Text) Adapter\n *\n * Tree-shakeable adapter for Groq audio transcription. Supports\n * whisper-large-v3 and whisper-large-v3-turbo.\n *\n * Features:\n * - Audio file uploads (File, Blob, ArrayBuffer, base64/data URL)\n * - Remote audio URLs passed directly via Groq's `url` field — no upload needed\n * - Verbose JSON response with segment and word timestamps\n * - Language detection or specification (ISO-639-1)\n * - Confidence scores derived from segment avg_logprob\n */\nexport class GroqTranscriptionAdapter<\n TModel extends GroqTranscriptionModel,\n> extends BaseTranscriptionAdapter<TModel, GroqTranscriptionProviderOptions> {\n readonly name = 'groq' as const\n\n private readonly apiKey: string\n private readonly baseURL: string\n private readonly defaultHeaders: Record<string, string>\n\n constructor(config: GroqTranscriptionConfig, model: TModel) {\n super(model, {})\n const resolved = withGroqDefaults(config)\n this.apiKey = resolved.apiKey\n this.baseURL = resolved.baseURL ?? 'https://api.groq.com/openai/v1'\n this.defaultHeaders = normalizeHeaders(resolved.defaultHeaders)\n }\n\n async transcribe(\n options: TranscriptionOptions<GroqTranscriptionProviderOptions>,\n ): Promise<TranscriptionResult> {\n const { model, audio, language, prompt, responseFormat, modelOptions } =\n options\n\n // Groq's transcription endpoint only accepts 'json', 'text', and\n // 'verbose_json'. Reject 'srt'/'vtt' up front so callers get a clear\n // message instead of an opaque Groq HTTP error.\n if (responseFormat === 'srt' || responseFormat === 'vtt') {\n throw new Error(\n `Groq transcription does not support responseFormat='${responseFormat}'. ` +\n `Supported values: 'json', 'text', 'verbose_json'.`,\n )\n }\n\n // Default to verbose_json so callers get language, duration, and timestamps\n // without having to opt in explicitly. Both Groq whisper models support it.\n const effectiveFormat = responseFormat ?? 'verbose_json'\n const useVerbose = effectiveFormat === 'verbose_json'\n\n const form = new FormData()\n form.append('model', model)\n form.append('response_format', effectiveFormat)\n if (language !== undefined) form.append('language', language)\n if (prompt !== undefined) form.append('prompt', prompt)\n if (modelOptions?.temperature !== undefined) {\n form.append('temperature', String(modelOptions.temperature))\n }\n if (modelOptions?.timestamp_granularities !== undefined) {\n for (const g of modelOptions.timestamp_granularities) {\n form.append('timestamp_granularities[]', g)\n }\n }\n\n // HTTP/HTTPS URLs are forwarded directly via Groq's `url` field, which\n // avoids a round-trip upload. All other inputs (File, Blob, ArrayBuffer,\n // base64, data URL) are converted to a File and sent as `file`.\n if (typeof audio === 'string' && /^https?:\\/\\//.test(audio)) {\n form.append('url', audio)\n } else {\n form.append('file', this.prepareAudioFile(audio))\n }\n\n try {\n options.logger.request(\n `activity=transcription provider=${this.name} model=${model} verbose=${useVerbose}`,\n { provider: this.name, model },\n )\n\n const response = await fetch(`${this.baseURL}/audio/transcriptions`, {\n method: 'POST',\n headers: {\n ...this.defaultHeaders,\n Authorization: `Bearer ${this.apiKey}`,\n },\n body: form,\n })\n\n if (!response.ok) {\n const body = await response\n .json()\n .catch(() => null as Record<string, unknown> | null)\n const message =\n (body?.error as { message?: string } | undefined)?.message ??\n `Groq API error ${response.status}`\n throw new Error(message)\n }\n\n if (useVerbose) {\n const data = (await response.json()) as GroqVerboseTranscriptionResponse\n const requestId = data.x_groq?.id ?? generateId(this.name)\n\n // `TranscriptionResult` declares optional fields without `| undefined`,\n // so under exactOptionalPropertyTypes we must omit absent fields rather\n // than assigning `undefined`.\n const segments = data.segments?.map(\n (seg): TranscriptionSegment => ({\n id: seg.id,\n start: seg.start,\n end: seg.end,\n text: seg.text,\n confidence: Math.exp(seg.avg_logprob),\n }),\n )\n const words = data.words?.map((w) => ({\n word: w.word,\n start: w.start,\n end: w.end,\n }))\n\n return {\n id: requestId,\n model,\n text: data.text,\n ...(data.language !== undefined && { language: data.language }),\n ...(data.duration !== undefined && { duration: data.duration }),\n ...(segments !== undefined && { segments }),\n ...(words !== undefined && { words }),\n }\n } else if (effectiveFormat === 'text') {\n const text = await response.text()\n return {\n id: generateId(this.name),\n model,\n text,\n ...(language !== undefined && { language }),\n }\n } else {\n const data = (await response.json()) as GroqJsonTranscriptionResponse\n return {\n id: data.x_groq?.id ?? generateId(this.name),\n model,\n text: data.text,\n ...(language !== undefined && { language }),\n }\n }\n } catch (error: unknown) {\n options.logger.errors(`${this.name}.transcribe fatal`, {\n error,\n source: `${this.name}.transcribe`,\n })\n throw error\n }\n }\n\n private prepareAudioFile(audio: string | File | Blob | ArrayBuffer): File {\n if (typeof File !== 'undefined' && audio instanceof File) {\n return audio\n }\n if (typeof Blob !== 'undefined' && audio instanceof Blob) {\n this.ensureFileSupport()\n return new File([audio], 'audio.mp3', {\n type: audio.type || 'audio/mpeg',\n })\n }\n if (typeof ArrayBuffer !== 'undefined' && audio instanceof ArrayBuffer) {\n this.ensureFileSupport()\n return new File([audio], 'audio.mp3', { type: 'audio/mpeg' })\n }\n if (typeof audio === 'string') {\n this.ensureFileSupport()\n\n if (audio.startsWith('data:')) {\n const parts = audio.split(',')\n const header = parts[0]\n const base64Data = parts[1] || ''\n const mimeMatch = header?.match(/data:([^;]+)/)\n const mimeType = mimeMatch?.[1] || 'audio/mpeg'\n const bytes = base64ToArrayBuffer(base64Data)\n const extension = mimeType.split('/')[1] || 'mp3'\n return new File([bytes], `audio.${extension}`, { type: mimeType })\n }\n\n const bytes = base64ToArrayBuffer(audio)\n return new File([bytes], 'audio.mp3', { type: 'audio/mpeg' })\n }\n\n throw new Error('Invalid audio input type')\n }\n\n // Throws on Node < 20 where the global `File` constructor is unavailable.\n private ensureFileSupport(): void {\n if (typeof File === 'undefined') {\n throw new Error(\n '`File` is not available in this environment. ' +\n 'Use Node.js 20 or newer, or pass a File object directly.',\n )\n }\n }\n}\n\n/**\n * Creates a Groq transcription adapter with an explicit API key.\n * Type resolution happens here at the call site.\n *\n * @param model - The model name (e.g., 'whisper-large-v3-turbo')\n * @param apiKey - Your Groq API key\n * @param config - Optional additional configuration\n * @returns Configured Groq transcription adapter instance\n *\n * @example\n * ```typescript\n * const adapter = createGroqTranscription('whisper-large-v3-turbo', 'gsk_...');\n *\n * const result = await generateTranscription({\n * adapter,\n * audio: audioFile,\n * language: 'en',\n * });\n * ```\n */\nexport function createGroqTranscription<TModel extends GroqTranscriptionModel>(\n model: TModel,\n apiKey: string,\n config?: Omit<GroqTranscriptionConfig, 'apiKey'>,\n): GroqTranscriptionAdapter<TModel> {\n return new GroqTranscriptionAdapter({ apiKey, ...config }, model)\n}\n\n/**\n * Creates a Groq transcription adapter using the `GROQ_API_KEY` environment\n * variable. Type resolution happens here at the call site.\n *\n * Looks for `GROQ_API_KEY` in:\n * - `process.env` (Node.js)\n * - `window.env` (browser with injected env)\n *\n * @param model - The model name (e.g., 'whisper-large-v3-turbo')\n * @param config - Optional configuration (excluding apiKey which is auto-detected)\n * @returns Configured Groq transcription adapter instance\n * @throws Error if GROQ_API_KEY is not found in environment\n *\n * @example\n * ```typescript\n * const adapter = groqTranscription('whisper-large-v3-turbo');\n *\n * const result = await generateTranscription({\n * adapter,\n * audio: 'https://example.com/audio.mp3',\n * });\n *\n * console.log(result.text)\n * ```\n */\nexport function groqTranscription<TModel extends GroqTranscriptionModel>(\n model: TModel,\n config?: Omit<GroqTranscriptionConfig, 'apiKey'>,\n): GroqTranscriptionAdapter<TModel> {\n const apiKey = getGroqApiKeyFromEnv()\n return createGroqTranscription(model, apiKey, config)\n}\n"],"mappings":";;;;;;;;;;;;;AA0BA,SAAS,iBACP,SACwB;CACxB,MAAM,MAA8B,CAAC;CACrC,IAAI,CAAC,SAAS,OAAO;CACrB,MAAM,UAAU,KAAa,UAAmB;EAC9C,IAAI,SAAS,MAAM,IAAI,OAAO,OAAO,KAAK;CAC5C;CACA,IAAI,mBAAmB,SACrB,QAAQ,SAAS,OAAO,QAAQ,OAAO,KAAK,KAAK,CAAC;MAC7C,IAAI,MAAM,QAAQ,OAAO,GAC9B,KAAK,MAAM,CAAC,KAAK,UAAU,SAAS,OAAO,KAAK,KAAK;MAErD,KAAK,MAAM,CAAC,KAAK,UAAU,OAAO,QAAQ,OAAO,GAAG,OAAO,KAAK,KAAK;CAEvE,OAAO;AACT;;;;;;;;;;;;;;AA2CA,IAAa,2BAAb,cAEU,yBAAmE;CAC3E,OAAgB;CAEhB;CACA;CACA;CAEA,YAAY,QAAiC,OAAe;EAC1D,MAAM,OAAO,CAAC,CAAC;EACf,MAAM,WAAW,iBAAiB,MAAM;EACxC,KAAK,SAAS,SAAS;EACvB,KAAK,UAAU,SAAS,WAAW;EACnC,KAAK,iBAAiB,iBAAiB,SAAS,cAAc;CAChE;CAEA,MAAM,WACJ,SAC8B;EAC9B,MAAM,EAAE,OAAO,OAAO,UAAU,QAAQ,gBAAgB,iBACtD;EAKF,IAAI,mBAAmB,SAAS,mBAAmB,OACjD,MAAM,IAAI,MACR,uDAAuD,eAAe,qDAExE;EAKF,MAAM,kBAAkB,kBAAkB;EAC1C,MAAM,aAAa,oBAAoB;EAEvC,MAAM,OAAO,IAAI,SAAS;EAC1B,KAAK,OAAO,SAAS,KAAK;EAC1B,KAAK,OAAO,mBAAmB,eAAe;EAC9C,IAAI,aAAa,KAAA,GAAW,KAAK,OAAO,YAAY,QAAQ;EAC5D,IAAI,WAAW,KAAA,GAAW,KAAK,OAAO,UAAU,MAAM;EACtD,IAAI,cAAc,gBAAgB,KAAA,GAChC,KAAK,OAAO,eAAe,OAAO,aAAa,WAAW,CAAC;EAE7D,IAAI,cAAc,4BAA4B,KAAA,GAC5C,KAAK,MAAM,KAAK,aAAa,yBAC3B,KAAK,OAAO,6BAA6B,CAAC;EAO9C,IAAI,OAAO,UAAU,YAAY,eAAe,KAAK,KAAK,GACxD,KAAK,OAAO,OAAO,KAAK;OAExB,KAAK,OAAO,QAAQ,KAAK,iBAAiB,KAAK,CAAC;EAGlD,IAAI;GACF,QAAQ,OAAO,QACb,mCAAmC,KAAK,KAAK,SAAS,MAAM,WAAW,cACvE;IAAE,UAAU,KAAK;IAAM;GAAM,CAC/B;GAEA,MAAM,WAAW,MAAM,MAAM,GAAG,KAAK,QAAQ,wBAAwB;IACnE,QAAQ;IACR,SAAS;KACP,GAAG,KAAK;KACR,eAAe,UAAU,KAAK;IAChC;IACA,MAAM;GACR,CAAC;GAED,IAAI,CAAC,SAAS,IAAI;IAIhB,MAAM,YACH,MAJgB,SAChB,KAAK,CAAC,CACN,YAAY,IAAsC,EAAA,EAE5C,MAAA,EAA4C,WACnD,kBAAkB,SAAS;IAC7B,MAAM,IAAI,MAAM,OAAO;GACzB;GAEA,IAAI,YAAY;IACd,MAAM,OAAQ,MAAM,SAAS,KAAK;IAClC,MAAM,YAAY,KAAK,QAAQ,MAAM,WAAW,KAAK,IAAI;IAKzD,MAAM,WAAW,KAAK,UAAU,KAC7B,SAA+B;KAC9B,IAAI,IAAI;KACR,OAAO,IAAI;KACX,KAAK,IAAI;KACT,MAAM,IAAI;KACV,YAAY,KAAK,IAAI,IAAI,WAAW;IACtC,EACF;IACA,MAAM,QAAQ,KAAK,OAAO,KAAK,OAAO;KACpC,MAAM,EAAE;KACR,OAAO,EAAE;KACT,KAAK,EAAE;IACT,EAAE;IAEF,OAAO;KACL,IAAI;KACJ;KACA,MAAM,KAAK;KACX,GAAI,KAAK,aAAa,KAAA,KAAa,EAAE,UAAU,KAAK,SAAS;KAC7D,GAAI,KAAK,aAAa,KAAA,KAAa,EAAE,UAAU,KAAK,SAAS;KAC7D,GAAI,aAAa,KAAA,KAAa,EAAE,SAAS;KACzC,GAAI,UAAU,KAAA,KAAa,EAAE,MAAM;IACrC;GACF,OAAO,IAAI,oBAAoB,QAAQ;IACrC,MAAM,OAAO,MAAM,SAAS,KAAK;IACjC,OAAO;KACL,IAAI,WAAW,KAAK,IAAI;KACxB;KACA;KACA,GAAI,aAAa,KAAA,KAAa,EAAE,SAAS;IAC3C;GACF,OAAO;IACL,MAAM,OAAQ,MAAM,SAAS,KAAK;IAClC,OAAO;KACL,IAAI,KAAK,QAAQ,MAAM,WAAW,KAAK,IAAI;KAC3C;KACA,MAAM,KAAK;KACX,GAAI,aAAa,KAAA,KAAa,EAAE,SAAS;IAC3C;GACF;EACF,SAAS,OAAgB;GACvB,QAAQ,OAAO,OAAO,GAAG,KAAK,KAAK,oBAAoB;IACrD;IACA,QAAQ,GAAG,KAAK,KAAK;GACvB,CAAC;GACD,MAAM;EACR;CACF;CAEA,iBAAyB,OAAiD;EACxE,IAAI,OAAO,SAAS,eAAe,iBAAiB,MAClD,OAAO;EAET,IAAI,OAAO,SAAS,eAAe,iBAAiB,MAAM;GACxD,KAAK,kBAAkB;GACvB,OAAO,IAAI,KAAK,CAAC,KAAK,GAAG,aAAa,EACpC,MAAM,MAAM,QAAQ,aACtB,CAAC;EACH;EACA,IAAI,OAAO,gBAAgB,eAAe,iBAAiB,aAAa;GACtE,KAAK,kBAAkB;GACvB,OAAO,IAAI,KAAK,CAAC,KAAK,GAAG,aAAa,EAAE,MAAM,aAAa,CAAC;EAC9D;EACA,IAAI,OAAO,UAAU,UAAU;GAC7B,KAAK,kBAAkB;GAEvB,IAAI,MAAM,WAAW,OAAO,GAAG;IAC7B,MAAM,QAAQ,MAAM,MAAM,GAAG;IAC7B,MAAM,SAAS,MAAM;IACrB,MAAM,aAAa,MAAM,MAAM;IAE/B,MAAM,YADY,QAAQ,MAAM,cAAc,EAAA,GACjB,MAAM;IACnC,MAAM,QAAQ,oBAAoB,UAAU;IAC5C,MAAM,YAAY,SAAS,MAAM,GAAG,CAAC,CAAC,MAAM;IAC5C,OAAO,IAAI,KAAK,CAAC,KAAK,GAAG,SAAS,aAAa,EAAE,MAAM,SAAS,CAAC;GACnE;GAEA,MAAM,QAAQ,oBAAoB,KAAK;GACvC,OAAO,IAAI,KAAK,CAAC,KAAK,GAAG,aAAa,EAAE,MAAM,aAAa,CAAC;EAC9D;EAEA,MAAM,IAAI,MAAM,0BAA0B;CAC5C;CAGA,oBAAkC;EAChC,IAAI,OAAO,SAAS,aAClB,MAAM,IAAI,MACR,uGAEF;CAEJ;AACF;;;;;;;;;;;;;;;;;;;;;AAsBA,SAAgB,wBACd,OACA,QACA,QACkC;CAClC,OAAO,IAAI,yBAAyB;EAAE;EAAQ,GAAG;CAAO,GAAG,KAAK;AAClE;;;;;;;;;;;;;;;;;;;;;;;;;;AA2BA,SAAgB,kBACd,OACA,QACkC;CAElC,OAAO,wBAAwB,OADhB,qBACuB,GAAQ,MAAM;AACtD"}
|
package/dist/esm/adapters/tts.js
CHANGED
|
@@ -1,73 +1,135 @@
|
|
|
1
|
+
import { getGroqApiKeyFromEnv, withGroqDefaults } from "../utils/client.js";
|
|
2
|
+
import { validateAudioInput } from "../audio/audio-provider-options.js";
|
|
1
3
|
import OpenAI from "openai";
|
|
4
|
+
import { arrayBufferToBase64, generateId } from "@tanstack/ai-utils";
|
|
2
5
|
import { BaseTTSAdapter } from "@tanstack/ai/adapters";
|
|
3
6
|
import { toRunErrorPayload } from "@tanstack/ai/adapter-internals";
|
|
4
|
-
|
|
5
|
-
|
|
6
|
-
|
|
7
|
-
|
|
8
|
-
|
|
9
|
-
|
|
10
|
-
|
|
11
|
-
|
|
12
|
-
|
|
13
|
-
|
|
14
|
-
|
|
15
|
-
|
|
16
|
-
|
|
17
|
-
|
|
18
|
-
|
|
19
|
-
|
|
20
|
-
|
|
21
|
-
|
|
22
|
-
|
|
23
|
-
|
|
24
|
-
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
}
|
|
7
|
+
//#region src/adapters/tts.ts
|
|
8
|
+
/**
|
|
9
|
+
* Groq Text-to-Speech Adapter
|
|
10
|
+
*
|
|
11
|
+
* Tree-shakeable adapter for Groq TTS functionality. Groq exposes an
|
|
12
|
+
* OpenAI-compatible `/audio/speech` endpoint, so the adapter drives it with
|
|
13
|
+
* the OpenAI SDK via a `baseURL` override (the same pattern as the Groq text
|
|
14
|
+
* adapter).
|
|
15
|
+
*
|
|
16
|
+
* Supports `canopylabs/orpheus-v1-english` and
|
|
17
|
+
* `canopylabs/orpheus-arabic-saudi`.
|
|
18
|
+
*
|
|
19
|
+
* Features:
|
|
20
|
+
* - English voices: autumn(f), diana(f), hannah(f), austin(m), daniel(m), troy(m)
|
|
21
|
+
* - Arabic voices: fahad(m), sultan(m), lulwa(f), noura(f)
|
|
22
|
+
* - Output formats: flac, mp3, mulaw, ogg, wav (default wav)
|
|
23
|
+
* - Speed control
|
|
24
|
+
* - Configurable sample rate via `modelOptions`
|
|
25
|
+
*/
|
|
26
|
+
var GroqTTSAdapter = class extends BaseTTSAdapter {
|
|
27
|
+
name = "groq";
|
|
28
|
+
client;
|
|
29
|
+
constructor(config, model) {
|
|
30
|
+
super(model, {});
|
|
31
|
+
this.client = new OpenAI(withGroqDefaults(config));
|
|
32
|
+
}
|
|
33
|
+
async generateSpeech(options) {
|
|
34
|
+
const { model, text, voice, format, speed, modelOptions } = options;
|
|
35
|
+
validateAudioInput({
|
|
36
|
+
input: text,
|
|
37
|
+
model: this.model
|
|
38
|
+
});
|
|
39
|
+
const request = {
|
|
40
|
+
model,
|
|
41
|
+
input: text,
|
|
42
|
+
voice: voice ?? "autumn",
|
|
43
|
+
response_format: format ?? "wav",
|
|
44
|
+
...speed !== void 0 && { speed },
|
|
45
|
+
...modelOptions ?? {}
|
|
46
|
+
};
|
|
47
|
+
try {
|
|
48
|
+
options.logger.request(`activity=tts provider=${this.name} model=${model} format=${request.response_format ?? "default"} voice=${request.voice}`, {
|
|
49
|
+
provider: this.name,
|
|
50
|
+
model
|
|
51
|
+
});
|
|
52
|
+
const base64 = arrayBufferToBase64(await (await this.client.audio.speech.create(request)).arrayBuffer());
|
|
53
|
+
const outputFormat = request.response_format ?? "wav";
|
|
54
|
+
const contentType = this.getContentType(outputFormat);
|
|
55
|
+
return {
|
|
56
|
+
id: generateId(this.name),
|
|
57
|
+
model,
|
|
58
|
+
audio: base64,
|
|
59
|
+
format: outputFormat,
|
|
60
|
+
contentType
|
|
61
|
+
};
|
|
62
|
+
} catch (error) {
|
|
63
|
+
options.logger.errors(`${this.name}.generateSpeech fatal`, {
|
|
64
|
+
error: toRunErrorPayload(error, `${this.name}.generateSpeech failed`),
|
|
65
|
+
source: `${this.name}.generateSpeech`
|
|
66
|
+
});
|
|
67
|
+
throw error;
|
|
68
|
+
}
|
|
69
|
+
}
|
|
70
|
+
getContentType(format) {
|
|
71
|
+
return {
|
|
72
|
+
flac: "audio/flac",
|
|
73
|
+
mp3: "audio/mpeg",
|
|
74
|
+
mulaw: "audio/basic",
|
|
75
|
+
ogg: "audio/ogg",
|
|
76
|
+
wav: "audio/wav"
|
|
77
|
+
}[format] || "audio/wav";
|
|
78
|
+
}
|
|
79
|
+
};
|
|
80
|
+
/**
|
|
81
|
+
* Creates a Groq speech adapter with explicit API key.
|
|
82
|
+
* Type resolution happens here at the call site.
|
|
83
|
+
*
|
|
84
|
+
* @param model - The model name (e.g., 'canopylabs/orpheus-v1-english')
|
|
85
|
+
* @param apiKey - Your Groq API key
|
|
86
|
+
* @param config - Optional additional configuration
|
|
87
|
+
* @returns Configured Groq speech adapter instance with resolved types
|
|
88
|
+
*
|
|
89
|
+
* @example
|
|
90
|
+
* ```typescript
|
|
91
|
+
* const adapter = createGroqSpeech('canopylabs/orpheus-v1-english', 'gsk_...')
|
|
92
|
+
*
|
|
93
|
+
* const result = await generateSpeech({
|
|
94
|
+
* adapter,
|
|
95
|
+
* text: 'Hello, world!',
|
|
96
|
+
* voice: 'autumn',
|
|
97
|
+
* })
|
|
98
|
+
* ```
|
|
99
|
+
*/
|
|
61
100
|
function createGroqSpeech(model, apiKey, config) {
|
|
62
|
-
|
|
101
|
+
return new GroqTTSAdapter({
|
|
102
|
+
apiKey,
|
|
103
|
+
...config
|
|
104
|
+
}, model);
|
|
63
105
|
}
|
|
106
|
+
/**
|
|
107
|
+
* Creates a Groq speech adapter with automatic API key detection from
|
|
108
|
+
* environment variables.
|
|
109
|
+
*
|
|
110
|
+
* Looks for `GROQ_API_KEY` in the environment.
|
|
111
|
+
*
|
|
112
|
+
* @param model - The model name (e.g., 'canopylabs/orpheus-v1-english')
|
|
113
|
+
* @param config - Optional configuration (excluding apiKey which is auto-detected)
|
|
114
|
+
* @returns Configured Groq speech adapter instance with resolved types
|
|
115
|
+
* @throws Error if GROQ_API_KEY is not found in environment
|
|
116
|
+
*
|
|
117
|
+
* @example
|
|
118
|
+
* ```typescript
|
|
119
|
+
* const adapter = groqSpeech('canopylabs/orpheus-v1-english')
|
|
120
|
+
*
|
|
121
|
+
* const result = await generateSpeech({
|
|
122
|
+
* adapter,
|
|
123
|
+
* text: 'Welcome to TanStack AI!',
|
|
124
|
+
* voice: 'autumn',
|
|
125
|
+
* format: 'wav',
|
|
126
|
+
* })
|
|
127
|
+
* ```
|
|
128
|
+
*/
|
|
64
129
|
function groqSpeech(model, config) {
|
|
65
|
-
|
|
66
|
-
return createGroqSpeech(model, apiKey, config);
|
|
130
|
+
return createGroqSpeech(model, getGroqApiKeyFromEnv(), config);
|
|
67
131
|
}
|
|
68
|
-
|
|
69
|
-
|
|
70
|
-
|
|
71
|
-
|
|
72
|
-
};
|
|
73
|
-
//# sourceMappingURL=tts.js.map
|
|
132
|
+
//#endregion
|
|
133
|
+
export { GroqTTSAdapter, createGroqSpeech, groqSpeech };
|
|
134
|
+
|
|
135
|
+
//# sourceMappingURL=tts.js.map
|