@ai-sdk/cartesia 0.0.1 → 3.0.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,285 @@
1
+ import type { SpeechModelV4, SharedV4Warning } from '@ai-sdk/provider';
2
+ import {
3
+ combineHeaders,
4
+ createBinaryResponseHandler,
5
+ parseProviderOptions,
6
+ postJsonToApi,
7
+ serializeModelOptions,
8
+ WORKFLOW_SERIALIZE,
9
+ WORKFLOW_DESERIALIZE,
10
+ } from '@ai-sdk/provider-utils';
11
+ import type { CartesiaConfig } from './cartesia-config';
12
+ import { cartesiaFailedResponseHandler } from './cartesia-error';
13
+ import { cartesiaSpeechModelOptionsSchema } from './cartesia-speech-model-options';
14
+ import type {
15
+ CartesiaSpeechAPITypes,
16
+ CartesiaSpeechBitRate,
17
+ CartesiaSpeechEncoding,
18
+ CartesiaSpeechOutputFormat,
19
+ CartesiaSpeechSampleRate,
20
+ } from './cartesia-speech-api-types';
21
+ import type {
22
+ CartesiaSpeechModelId,
23
+ CartesiaSpeechVoiceId,
24
+ } from './cartesia-speech-options';
25
+
26
+ interface CartesiaSpeechModelConfig extends CartesiaConfig {
27
+ _internal?: {
28
+ currentDate?: () => Date;
29
+ };
30
+ }
31
+
32
+ // Default output format used when no outputFormat / provider options are set.
33
+ const DEFAULT_OUTPUT_FORMAT: CartesiaSpeechAPITypes['output_format'] = {
34
+ container: 'mp3',
35
+ sample_rate: 44100,
36
+ bit_rate: 128000,
37
+ };
38
+
39
+ const OUTPUT_FORMATS: Record<string, CartesiaSpeechOutputFormat> = {
40
+ alaw: { container: 'raw', encoding: 'pcm_alaw', sample_rate: 8000 },
41
+ mp3: DEFAULT_OUTPUT_FORMAT,
42
+ mulaw: { container: 'raw', encoding: 'pcm_mulaw', sample_rate: 8000 },
43
+ pcm: { container: 'raw', encoding: 'pcm_f32le', sample_rate: 44100 },
44
+ raw: { container: 'raw', encoding: 'pcm_f32le', sample_rate: 44100 },
45
+ wav: { container: 'wav', encoding: 'pcm_s16le', sample_rate: 44100 },
46
+ };
47
+
48
+ const SAMPLE_RATES = [8000, 16000, 22050, 24000, 44100, 48000] as const;
49
+
50
+ function isSampleRate(value: number): value is CartesiaSpeechSampleRate {
51
+ return SAMPLE_RATES.includes(value as CartesiaSpeechSampleRate);
52
+ }
53
+
54
+ function resolveOutputFormat({
55
+ outputFormat,
56
+ providerOptions,
57
+ warnings,
58
+ }: {
59
+ outputFormat: string;
60
+ providerOptions:
61
+ | {
62
+ container?: 'raw' | 'wav' | 'mp3' | null;
63
+ encoding?: CartesiaSpeechEncoding | null;
64
+ sampleRate?: CartesiaSpeechSampleRate | null;
65
+ bitRate?: CartesiaSpeechBitRate | null;
66
+ }
67
+ | undefined;
68
+ warnings: SharedV4Warning[];
69
+ }): CartesiaSpeechOutputFormat {
70
+ const [formatName, sampleRateText, ...extraParts] = outputFormat
71
+ .toLowerCase()
72
+ .split('_');
73
+ const mapped = OUTPUT_FORMATS[formatName];
74
+ let resolved = mapped ? { ...mapped } : { ...DEFAULT_OUTPUT_FORMAT };
75
+
76
+ if (!mapped) {
77
+ warnings.push({
78
+ type: 'unsupported',
79
+ feature: 'outputFormat',
80
+ details: `Unknown output format "${outputFormat}". Falling back to mp3. Use providerOptions.cartesia to configure container, encoding, and sampleRate directly.`,
81
+ });
82
+ } else if (sampleRateText != null) {
83
+ const parsedRate = Number(sampleRateText);
84
+ if (
85
+ extraParts.length === 0 &&
86
+ Number.isInteger(parsedRate) &&
87
+ isSampleRate(parsedRate)
88
+ ) {
89
+ resolved.sample_rate = parsedRate;
90
+ } else {
91
+ warnings.push({
92
+ type: 'unsupported',
93
+ feature: 'outputFormat',
94
+ details: `Unsupported Cartesia sample rate in output format "${outputFormat}". Using ${resolved.sample_rate} Hz instead.`,
95
+ });
96
+ }
97
+ }
98
+
99
+ const container = providerOptions?.container ?? resolved.container;
100
+ const sampleRate = providerOptions?.sampleRate ?? resolved.sample_rate;
101
+
102
+ if (container === 'mp3') {
103
+ if (providerOptions?.encoding != null) {
104
+ warnings.push({
105
+ type: 'unsupported',
106
+ feature: 'providerOptions.cartesia.encoding',
107
+ details:
108
+ 'Cartesia MP3 output does not accept an encoding. The encoding option was ignored.',
109
+ });
110
+ }
111
+
112
+ return {
113
+ container,
114
+ sample_rate: sampleRate,
115
+ bit_rate:
116
+ providerOptions?.bitRate ??
117
+ (resolved.container === 'mp3' ? resolved.bit_rate : 128000),
118
+ };
119
+ }
120
+
121
+ if (providerOptions?.bitRate != null) {
122
+ warnings.push({
123
+ type: 'unsupported',
124
+ feature: 'providerOptions.cartesia.bitRate',
125
+ details:
126
+ 'Cartesia raw and WAV output do not accept a bit rate. The bitRate option was ignored.',
127
+ });
128
+ }
129
+
130
+ return {
131
+ container,
132
+ encoding:
133
+ providerOptions?.encoding ??
134
+ (resolved.container === 'mp3'
135
+ ? container === 'wav'
136
+ ? 'pcm_s16le'
137
+ : 'pcm_f32le'
138
+ : resolved.encoding),
139
+ sample_rate: sampleRate,
140
+ };
141
+ }
142
+
143
+ export class CartesiaSpeechModel implements SpeechModelV4 {
144
+ readonly specificationVersion = 'v4';
145
+
146
+ get provider(): string {
147
+ return this.config.provider;
148
+ }
149
+
150
+ static [WORKFLOW_SERIALIZE](model: CartesiaSpeechModel) {
151
+ return serializeModelOptions({
152
+ modelId: model.modelId,
153
+ config: model.config,
154
+ });
155
+ }
156
+
157
+ static [WORKFLOW_DESERIALIZE](options: {
158
+ modelId: CartesiaSpeechModelId;
159
+ config: CartesiaSpeechModelConfig;
160
+ }) {
161
+ return new CartesiaSpeechModel(options.modelId, options.config);
162
+ }
163
+
164
+ constructor(
165
+ readonly modelId: CartesiaSpeechModelId,
166
+ private readonly config: CartesiaSpeechModelConfig,
167
+ ) {}
168
+
169
+ private async getArgs({
170
+ text,
171
+ voice,
172
+ outputFormat = 'mp3',
173
+ instructions,
174
+ language,
175
+ speed,
176
+ providerOptions,
177
+ }: Parameters<SpeechModelV4['doGenerate']>[0]) {
178
+ const warnings: SharedV4Warning[] = [];
179
+
180
+ // Parse provider options
181
+ const cartesiaOptions = await parseProviderOptions({
182
+ provider: 'cartesia',
183
+ providerOptions,
184
+ schema: cartesiaSpeechModelOptionsSchema,
185
+ });
186
+
187
+ if (!voice) {
188
+ throw new Error('Cartesia speech models require a `voice` to be set.');
189
+ }
190
+
191
+ const outputFormatObject = resolveOutputFormat({
192
+ outputFormat,
193
+ providerOptions: cartesiaOptions,
194
+ warnings,
195
+ });
196
+
197
+ // Create request body
198
+ const requestBody: CartesiaSpeechAPITypes = {
199
+ model_id: this.modelId,
200
+ transcript: text,
201
+ voice: {
202
+ mode: 'id',
203
+ id: voice as CartesiaSpeechVoiceId,
204
+ },
205
+ output_format: outputFormatObject,
206
+ };
207
+
208
+ // Map generic language
209
+ if (language) {
210
+ requestBody.language = language;
211
+ }
212
+
213
+ // Provider-specific options override the corresponding generic options.
214
+ if (cartesiaOptions) {
215
+ if (cartesiaOptions.language != null) {
216
+ requestBody.language = cartesiaOptions.language;
217
+ }
218
+ }
219
+
220
+ const resolvedSpeed = cartesiaOptions?.speed ?? speed;
221
+ if (resolvedSpeed != null) {
222
+ if (resolvedSpeed >= 0.6 && resolvedSpeed <= 1.5) {
223
+ requestBody.generation_config = { speed: resolvedSpeed };
224
+ } else {
225
+ warnings.push({
226
+ type: 'unsupported',
227
+ feature: 'speed',
228
+ details:
229
+ 'Cartesia speed must be between 0.6 and 1.5. The speed option was ignored.',
230
+ });
231
+ }
232
+ }
233
+
234
+ if (instructions) {
235
+ warnings.push({
236
+ type: 'unsupported',
237
+ feature: 'instructions',
238
+ details: `Cartesia speech models do not support instructions. Instructions parameter was ignored.`,
239
+ });
240
+ }
241
+
242
+ return {
243
+ requestBody,
244
+ warnings,
245
+ };
246
+ }
247
+
248
+ async doGenerate(
249
+ options: Parameters<SpeechModelV4['doGenerate']>[0],
250
+ ): Promise<Awaited<ReturnType<SpeechModelV4['doGenerate']>>> {
251
+ const currentDate = this.config._internal?.currentDate?.() ?? new Date();
252
+ const { requestBody, warnings } = await this.getArgs(options);
253
+
254
+ const {
255
+ value: audio,
256
+ responseHeaders,
257
+ rawValue: rawResponse,
258
+ } = await postJsonToApi({
259
+ url: this.config.url({
260
+ path: '/tts/bytes',
261
+ modelId: this.modelId,
262
+ }),
263
+ headers: combineHeaders(this.config.headers?.(), options.headers),
264
+ body: requestBody,
265
+ failedResponseHandler: cartesiaFailedResponseHandler,
266
+ successfulResponseHandler: createBinaryResponseHandler(),
267
+ abortSignal: options.abortSignal,
268
+ fetch: this.config.fetch,
269
+ });
270
+
271
+ return {
272
+ audio,
273
+ warnings,
274
+ request: {
275
+ body: JSON.stringify(requestBody),
276
+ },
277
+ response: {
278
+ timestamp: currentDate,
279
+ modelId: this.modelId,
280
+ headers: responseHeaders,
281
+ body: rawResponse,
282
+ },
283
+ };
284
+ }
285
+ }
@@ -0,0 +1,9 @@
1
+ export type CartesiaSpeechModelId =
2
+ | 'sonic-3.5'
3
+ | 'sonic-3'
4
+ | 'sonic-2'
5
+ | 'sonic-turbo'
6
+ | 'sonic-latest'
7
+ | (string & {});
8
+
9
+ export type CartesiaSpeechVoiceId = string;
@@ -0,0 +1,13 @@
1
+ import { z } from 'zod/v4';
2
+
3
+ // https://docs.cartesia.ai/api-reference/stt/transcribe
4
+ export const cartesiaTranscriptionModelOptionsSchema = z.object({
5
+ /** The language of the audio (ISO 639-1 code). Defaults to English. */
6
+ language: z.string().nullish(),
7
+ /** The timestamp granularities to populate. Currently only `word` is supported. */
8
+ timestampGranularities: z.array(z.enum(['word'])).nullish(),
9
+ });
10
+
11
+ export type CartesiaTranscriptionModelOptions = z.infer<
12
+ typeof cartesiaTranscriptionModelOptionsSchema
13
+ >;
@@ -0,0 +1,157 @@
1
+ import type { TranscriptionModelV4, SharedV4Warning } from '@ai-sdk/provider';
2
+ import {
3
+ combineHeaders,
4
+ convertBase64ToUint8Array,
5
+ createJsonResponseHandler,
6
+ mediaTypeToExtension,
7
+ parseProviderOptions,
8
+ postFormDataToApi,
9
+ serializeModelOptions,
10
+ WORKFLOW_SERIALIZE,
11
+ WORKFLOW_DESERIALIZE,
12
+ } from '@ai-sdk/provider-utils';
13
+ import { z } from 'zod/v4';
14
+ import type { CartesiaConfig } from './cartesia-config';
15
+ import { cartesiaFailedResponseHandler } from './cartesia-error';
16
+ import { cartesiaTranscriptionModelOptionsSchema } from './cartesia-transcription-model-options';
17
+ import type { CartesiaTranscriptionModelId } from './cartesia-transcription-options';
18
+
19
+ interface CartesiaTranscriptionModelConfig extends CartesiaConfig {
20
+ _internal?: {
21
+ currentDate?: () => Date;
22
+ };
23
+ }
24
+
25
+ export class CartesiaTranscriptionModel implements TranscriptionModelV4 {
26
+ readonly specificationVersion = 'v4';
27
+
28
+ get provider(): string {
29
+ return this.config.provider;
30
+ }
31
+
32
+ static [WORKFLOW_SERIALIZE](model: CartesiaTranscriptionModel) {
33
+ return serializeModelOptions({
34
+ modelId: model.modelId,
35
+ config: model.config,
36
+ });
37
+ }
38
+
39
+ static [WORKFLOW_DESERIALIZE](options: {
40
+ modelId: CartesiaTranscriptionModelId;
41
+ config: CartesiaTranscriptionModelConfig;
42
+ }) {
43
+ return new CartesiaTranscriptionModel(options.modelId, options.config);
44
+ }
45
+
46
+ constructor(
47
+ readonly modelId: CartesiaTranscriptionModelId,
48
+ private readonly config: CartesiaTranscriptionModelConfig,
49
+ ) {}
50
+
51
+ private async getArgs({
52
+ audio,
53
+ mediaType,
54
+ providerOptions,
55
+ }: Parameters<TranscriptionModelV4['doGenerate']>[0]) {
56
+ const warnings: SharedV4Warning[] = [];
57
+
58
+ // Parse provider options
59
+ const cartesiaOptions = await parseProviderOptions({
60
+ provider: 'cartesia',
61
+ providerOptions,
62
+ schema: cartesiaTranscriptionModelOptionsSchema,
63
+ });
64
+
65
+ // Create form data with base fields
66
+ const formData = new FormData();
67
+ const blob =
68
+ audio instanceof Uint8Array
69
+ ? new Blob([audio])
70
+ : new Blob([convertBase64ToUint8Array(audio)]);
71
+
72
+ const fileExtension = mediaTypeToExtension(mediaType);
73
+ formData.append('model', this.modelId);
74
+ formData.append(
75
+ 'file',
76
+ new File([blob], 'audio', { type: mediaType }),
77
+ `audio.${fileExtension}`,
78
+ );
79
+
80
+ // Add provider-specific options
81
+ if (cartesiaOptions) {
82
+ if (cartesiaOptions.language != null) {
83
+ formData.set('language', cartesiaOptions.language);
84
+ }
85
+ if (cartesiaOptions.timestampGranularities != null) {
86
+ for (const granularity of cartesiaOptions.timestampGranularities) {
87
+ formData.append('timestamp_granularities[]', granularity);
88
+ }
89
+ }
90
+ }
91
+
92
+ return {
93
+ formData,
94
+ warnings,
95
+ };
96
+ }
97
+
98
+ async doGenerate(
99
+ options: Parameters<TranscriptionModelV4['doGenerate']>[0],
100
+ ): Promise<Awaited<ReturnType<TranscriptionModelV4['doGenerate']>>> {
101
+ const currentDate = this.config._internal?.currentDate?.() ?? new Date();
102
+ const { formData, warnings } = await this.getArgs(options);
103
+
104
+ const {
105
+ value: response,
106
+ responseHeaders,
107
+ rawValue: rawResponse,
108
+ } = await postFormDataToApi({
109
+ url: this.config.url({
110
+ path: '/stt',
111
+ modelId: this.modelId,
112
+ }),
113
+ headers: combineHeaders(this.config.headers?.(), options.headers),
114
+ formData,
115
+ failedResponseHandler: cartesiaFailedResponseHandler,
116
+ successfulResponseHandler: createJsonResponseHandler(
117
+ cartesiaTranscriptionResponseSchema,
118
+ ),
119
+ abortSignal: options.abortSignal,
120
+ fetch: this.config.fetch,
121
+ });
122
+
123
+ return {
124
+ text: response.text,
125
+ segments:
126
+ response.words?.map(word => ({
127
+ text: word.word,
128
+ startSecond: word.start,
129
+ endSecond: word.end,
130
+ })) ?? [],
131
+ language: response.language ?? undefined,
132
+ durationInSeconds: response.duration ?? undefined,
133
+ warnings,
134
+ response: {
135
+ timestamp: currentDate,
136
+ modelId: this.modelId,
137
+ headers: responseHeaders,
138
+ body: rawResponse,
139
+ },
140
+ };
141
+ }
142
+ }
143
+
144
+ const cartesiaTranscriptionResponseSchema = z.object({
145
+ text: z.string(),
146
+ language: z.string().nullish(),
147
+ duration: z.number().nullish(),
148
+ words: z
149
+ .array(
150
+ z.object({
151
+ word: z.string(),
152
+ start: z.number(),
153
+ end: z.number(),
154
+ }),
155
+ )
156
+ .nullish(),
157
+ });
@@ -0,0 +1 @@
1
+ export type CartesiaTranscriptionModelId = 'ink-whisper' | (string & {});
package/src/index.ts ADDED
@@ -0,0 +1,20 @@
1
+ export { createCartesia, cartesia } from './cartesia-provider';
2
+ export type {
3
+ CartesiaProvider,
4
+ CartesiaProviderSettings,
5
+ } from './cartesia-provider';
6
+ export { CartesiaSpeechModel } from './cartesia-speech-model';
7
+ export { CartesiaTranscriptionModel } from './cartesia-transcription-model';
8
+ export { CartesiaRealtimeModel } from './cartesia-realtime-model';
9
+ export type {
10
+ CartesiaRealtimeModelId,
11
+ CartesiaRealtimeModelOptions,
12
+ } from './cartesia-realtime-model-options';
13
+ export type {
14
+ CartesiaSpeechModelId,
15
+ CartesiaSpeechVoiceId,
16
+ } from './cartesia-speech-options';
17
+ export type { CartesiaTranscriptionModelId } from './cartesia-transcription-options';
18
+ export type { CartesiaSpeechModelOptions } from './cartesia-speech-model-options';
19
+ export type { CartesiaTranscriptionModelOptions } from './cartesia-transcription-model-options';
20
+ export { VERSION } from './version';
Binary file
package/src/version.ts ADDED
@@ -0,0 +1,6 @@
1
+ // Version string of this package injected at build time.
2
+ declare const __PACKAGE_VERSION__: string | undefined;
3
+ export const VERSION: string =
4
+ typeof __PACKAGE_VERSION__ !== 'undefined'
5
+ ? __PACKAGE_VERSION__
6
+ : '0.0.0-test';