@tanstack/openai-base 0.2.1 → 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (79) hide show
  1. package/README.md +121 -0
  2. package/dist/esm/adapters/chat-completions-text.d.ts +49 -21
  3. package/dist/esm/adapters/chat-completions-text.js +476 -68
  4. package/dist/esm/adapters/chat-completions-text.js.map +1 -1
  5. package/dist/esm/adapters/chat-completions-tool-converter.d.ts +8 -4
  6. package/dist/esm/adapters/chat-completions-tool-converter.js.map +1 -1
  7. package/dist/esm/adapters/responses-text.d.ts +46 -33
  8. package/dist/esm/adapters/responses-text.js +657 -142
  9. package/dist/esm/adapters/responses-text.js.map +1 -1
  10. package/dist/esm/index.d.ts +2 -9
  11. package/dist/esm/index.js +4 -16
  12. package/dist/esm/index.js.map +1 -1
  13. package/dist/esm/tools/apply-patch-tool.d.ts +2 -2
  14. package/dist/esm/tools/apply-patch-tool.js.map +1 -1
  15. package/dist/esm/tools/code-interpreter-tool.d.ts +3 -2
  16. package/dist/esm/tools/code-interpreter-tool.js.map +1 -1
  17. package/dist/esm/tools/computer-use-tool.d.ts +2 -2
  18. package/dist/esm/tools/computer-use-tool.js.map +1 -1
  19. package/dist/esm/tools/custom-tool.d.ts +2 -2
  20. package/dist/esm/tools/custom-tool.js.map +1 -1
  21. package/dist/esm/tools/file-search-tool.d.ts +2 -2
  22. package/dist/esm/tools/file-search-tool.js.map +1 -1
  23. package/dist/esm/tools/function-tool.d.ts +2 -2
  24. package/dist/esm/tools/function-tool.js.map +1 -1
  25. package/dist/esm/tools/image-generation-tool.d.ts +3 -2
  26. package/dist/esm/tools/image-generation-tool.js.map +1 -1
  27. package/dist/esm/tools/local-shell-tool.d.ts +3 -2
  28. package/dist/esm/tools/local-shell-tool.js.map +1 -1
  29. package/dist/esm/tools/mcp-tool.d.ts +3 -2
  30. package/dist/esm/tools/mcp-tool.js.map +1 -1
  31. package/dist/esm/tools/shell-tool.d.ts +2 -2
  32. package/dist/esm/tools/shell-tool.js.map +1 -1
  33. package/dist/esm/tools/web-search-preview-tool.d.ts +2 -2
  34. package/dist/esm/tools/web-search-preview-tool.js.map +1 -1
  35. package/dist/esm/tools/web-search-tool.d.ts +2 -2
  36. package/dist/esm/tools/web-search-tool.js.map +1 -1
  37. package/package.json +6 -6
  38. package/src/adapters/chat-completions-text.ts +601 -117
  39. package/src/adapters/chat-completions-tool-converter.ts +9 -5
  40. package/src/adapters/responses-text.ts +865 -210
  41. package/src/index.ts +2 -12
  42. package/src/tools/apply-patch-tool.ts +2 -2
  43. package/src/tools/code-interpreter-tool.ts +4 -2
  44. package/src/tools/computer-use-tool.ts +2 -2
  45. package/src/tools/custom-tool.ts +2 -2
  46. package/src/tools/file-search-tool.ts +3 -3
  47. package/src/tools/function-tool.ts +2 -2
  48. package/src/tools/image-generation-tool.ts +4 -2
  49. package/src/tools/local-shell-tool.ts +4 -2
  50. package/src/tools/mcp-tool.ts +4 -2
  51. package/src/tools/shell-tool.ts +2 -2
  52. package/src/tools/web-search-preview-tool.ts +2 -2
  53. package/src/tools/web-search-tool.ts +2 -2
  54. package/dist/esm/adapters/image.d.ts +0 -32
  55. package/dist/esm/adapters/image.js +0 -89
  56. package/dist/esm/adapters/image.js.map +0 -1
  57. package/dist/esm/adapters/summarize.d.ts +0 -28
  58. package/dist/esm/adapters/summarize.js +0 -112
  59. package/dist/esm/adapters/summarize.js.map +0 -1
  60. package/dist/esm/adapters/transcription.d.ts +0 -34
  61. package/dist/esm/adapters/transcription.js +0 -131
  62. package/dist/esm/adapters/transcription.js.map +0 -1
  63. package/dist/esm/adapters/tts.d.ts +0 -26
  64. package/dist/esm/adapters/tts.js +0 -78
  65. package/dist/esm/adapters/tts.js.map +0 -1
  66. package/dist/esm/adapters/video.d.ts +0 -72
  67. package/dist/esm/adapters/video.js +0 -238
  68. package/dist/esm/adapters/video.js.map +0 -1
  69. package/dist/esm/types/config.d.ts +0 -4
  70. package/dist/esm/utils/client.d.ts +0 -3
  71. package/dist/esm/utils/client.js +0 -8
  72. package/dist/esm/utils/client.js.map +0 -1
  73. package/src/adapters/image.ts +0 -158
  74. package/src/adapters/summarize.ts +0 -174
  75. package/src/adapters/transcription.ts +0 -194
  76. package/src/adapters/tts.ts +0 -124
  77. package/src/adapters/video.ts +0 -385
  78. package/src/types/config.ts +0 -5
  79. package/src/utils/client.ts +0 -8
@@ -1,158 +0,0 @@
1
- import { BaseImageAdapter } from '@tanstack/ai/adapters'
2
- import { toRunErrorPayload } from '@tanstack/ai/adapter-internals'
3
- import { generateId } from '@tanstack/ai-utils'
4
- import { createOpenAICompatibleClient } from '../utils/client'
5
- import type {
6
- GeneratedImage,
7
- ImageGenerationOptions,
8
- ImageGenerationResult,
9
- } from '@tanstack/ai'
10
- import type OpenAI_SDK from 'openai'
11
- import type { OpenAICompatibleClientConfig } from '../types/config'
12
-
13
- /**
14
- * OpenAI-Compatible Image Generation Adapter
15
- *
16
- * A generalized base class for providers that implement OpenAI-compatible image
17
- * generation APIs. Providers like OpenAI, Grok, and others can extend this class
18
- * and only need to:
19
- * - Set `baseURL` in the config
20
- * - Lock the generic type parameters to provider-specific types
21
- * - Override validation or request building methods for provider-specific constraints
22
- *
23
- * All methods that validate inputs, build requests, or transform responses are
24
- * `protected` so subclasses can override them.
25
- */
26
- export class OpenAICompatibleImageAdapter<
27
- TModel extends string,
28
- TProviderOptions extends object = Record<string, any>,
29
- TModelProviderOptionsByName extends Record<string, any> = Record<string, any>,
30
- TModelSizeByName extends Record<string, string> = Record<string, string>,
31
- > extends BaseImageAdapter<
32
- TModel,
33
- TProviderOptions,
34
- TModelProviderOptionsByName,
35
- TModelSizeByName
36
- > {
37
- readonly kind = 'image' as const
38
- readonly name: string
39
-
40
- protected client: OpenAI_SDK
41
-
42
- constructor(
43
- config: OpenAICompatibleClientConfig,
44
- model: TModel,
45
- name: string = 'openai-compatible',
46
- ) {
47
- super(model, {})
48
- this.name = name
49
- this.client = createOpenAICompatibleClient(config)
50
- }
51
-
52
- async generateImages(
53
- options: ImageGenerationOptions<TProviderOptions>,
54
- ): Promise<ImageGenerationResult> {
55
- const { model, prompt, numberOfImages, size } = options
56
-
57
- // Validate inputs
58
- this.validatePrompt({ prompt, model })
59
- this.validateImageSize(model, size)
60
- this.validateNumberOfImages(model, numberOfImages)
61
-
62
- // Build request based on model type
63
- const request = this.buildRequest(options)
64
-
65
- try {
66
- options.logger.request(
67
- `activity=image provider=${this.name} model=${model} n=${request.n ?? 1} size=${request.size ?? 'default'}`,
68
- { provider: this.name, model },
69
- )
70
- const response = await this.client.images.generate({
71
- ...request,
72
- stream: false,
73
- })
74
-
75
- return this.transformResponse(model, response)
76
- } catch (error: unknown) {
77
- // Narrow before logging: raw SDK errors can carry request metadata
78
- // (including auth headers) which we must never surface to user loggers.
79
- options.logger.errors(`${this.name}.generateImages fatal`, {
80
- error: toRunErrorPayload(error, `${this.name}.generateImages failed`),
81
- source: `${this.name}.generateImages`,
82
- })
83
- throw error
84
- }
85
- }
86
-
87
- protected buildRequest(
88
- options: ImageGenerationOptions<TProviderOptions>,
89
- ): OpenAI_SDK.Images.ImageGenerateParams {
90
- const { model, prompt, numberOfImages, size, modelOptions } = options
91
-
92
- return {
93
- model,
94
- prompt,
95
- n: numberOfImages ?? 1,
96
- size: size as OpenAI_SDK.Images.ImageGenerateParams['size'],
97
- ...modelOptions,
98
- }
99
- }
100
-
101
- protected transformResponse(
102
- model: string,
103
- response: OpenAI_SDK.Images.ImagesResponse,
104
- ): ImageGenerationResult {
105
- const images: Array<GeneratedImage> = (response.data ?? []).flatMap(
106
- (item): Array<GeneratedImage> => {
107
- const revisedPrompt = item.revised_prompt
108
- if (item.b64_json) {
109
- return [{ b64Json: item.b64_json, revisedPrompt }]
110
- }
111
- if (item.url) {
112
- return [{ url: item.url, revisedPrompt }]
113
- }
114
- return []
115
- },
116
- )
117
-
118
- return {
119
- id: generateId(this.name),
120
- model,
121
- images,
122
- usage: response.usage
123
- ? {
124
- inputTokens: response.usage.input_tokens,
125
- outputTokens: response.usage.output_tokens,
126
- totalTokens: response.usage.total_tokens,
127
- }
128
- : undefined,
129
- }
130
- }
131
-
132
- protected validatePrompt(options: { prompt: string; model: string }): void {
133
- if (options.prompt.length === 0) {
134
- throw new Error('Prompt cannot be empty.')
135
- }
136
- }
137
-
138
- protected validateImageSize(_model: string, _size: string | undefined): void {
139
- // Default: no size validation — subclasses can override
140
- }
141
-
142
- protected validateNumberOfImages(
143
- _model: string,
144
- numberOfImages: number | undefined,
145
- ): void {
146
- if (numberOfImages === undefined) return
147
-
148
- // The base adapter only enforces "must be at least 1". Per-provider /
149
- // per-model upper bounds vary widely (some support 4, some 10, some
150
- // unlimited), so concrete adapter subclasses are expected to override
151
- // this method with a model-specific cap.
152
- if (numberOfImages < 1) {
153
- throw new Error(
154
- `Number of images must be at least 1. Requested: ${numberOfImages}`,
155
- )
156
- }
157
- }
158
- }
@@ -1,174 +0,0 @@
1
- import { BaseSummarizeAdapter } from '@tanstack/ai/adapters'
2
- import { toRunErrorPayload } from '@tanstack/ai/adapter-internals'
3
- import { generateId } from '@tanstack/ai-utils'
4
- import type {
5
- StreamChunk,
6
- SummarizationOptions,
7
- SummarizationResult,
8
- TextOptions,
9
- } from '@tanstack/ai'
10
-
11
- /**
12
- * Minimal interface for a text adapter that supports chatStream.
13
- * This allows the summarize adapter to work with any OpenAI-compatible
14
- * text adapter without tight coupling to a specific implementation.
15
- */
16
- export interface ChatStreamCapable<TProviderOptions extends object> {
17
- chatStream: (
18
- options: TextOptions<TProviderOptions>,
19
- ) => AsyncIterable<StreamChunk>
20
- }
21
-
22
- /**
23
- * OpenAI-Compatible Summarize Adapter
24
- *
25
- * A thin wrapper around a text adapter that adds summarization-specific prompting.
26
- * Delegates all API calls to the provided text adapter.
27
- *
28
- * Subclasses or instantiators provide a text adapter (or factory) at construction
29
- * time, allowing any OpenAI-compatible provider to get summarization for free by
30
- * reusing its text adapter.
31
- */
32
- export class OpenAICompatibleSummarizeAdapter<
33
- TModel extends string,
34
- TProviderOptions extends object = Record<string, any>,
35
- > extends BaseSummarizeAdapter<TModel, TProviderOptions> {
36
- readonly name: string
37
-
38
- private textAdapter: ChatStreamCapable<TProviderOptions>
39
-
40
- constructor(
41
- textAdapter: ChatStreamCapable<TProviderOptions>,
42
- model: TModel,
43
- name: string = 'openai-compatible',
44
- ) {
45
- super({}, model)
46
- this.name = name
47
- this.textAdapter = textAdapter
48
- }
49
-
50
- async summarize(options: SummarizationOptions): Promise<SummarizationResult> {
51
- const systemPrompt = this.buildSummarizationPrompt(options)
52
-
53
- let summary = ''
54
- const id = generateId(this.name)
55
- let model = options.model
56
- let usage = { promptTokens: 0, completionTokens: 0, totalTokens: 0 }
57
-
58
- options.logger.request(
59
- `activity=summarize provider=${this.name} model=${options.model} text-length=${options.text.length} maxLength=${options.maxLength ?? 'unset'}`,
60
- { provider: this.name, model: options.model },
61
- )
62
-
63
- try {
64
- for await (const chunk of this.textAdapter.chatStream({
65
- model: options.model,
66
- messages: [{ role: 'user', content: options.text }],
67
- systemPrompts: [systemPrompt],
68
- maxTokens: options.maxLength,
69
- temperature: 0.3,
70
- logger: options.logger,
71
- } satisfies TextOptions<TProviderOptions>)) {
72
- if (chunk.type === 'TEXT_MESSAGE_CONTENT') {
73
- if (chunk.content) {
74
- summary = chunk.content
75
- } else if (chunk.delta) {
76
- // Append delta only when present — a content-less chunk with no
77
- // delta would otherwise concat literal `'undefined'`.
78
- summary += chunk.delta
79
- }
80
- model = chunk.model || model
81
- }
82
- if (chunk.type === 'RUN_FINISHED') {
83
- if (chunk.usage) {
84
- usage = chunk.usage
85
- }
86
- }
87
- // Surface failures: the underlying chatStream emits RUN_ERROR instead
88
- // of throwing, so without this branch summarize() would return an
89
- // empty summary and pretend a failed run succeeded.
90
- if (chunk.type === 'RUN_ERROR') {
91
- const message =
92
- (chunk.error && typeof chunk.error.message === 'string'
93
- ? chunk.error.message
94
- : null) ?? 'Summarization failed'
95
- const code =
96
- chunk.error && typeof chunk.error.code === 'string'
97
- ? chunk.error.code
98
- : undefined
99
- const err = new Error(message)
100
- if (code) {
101
- ;(err as Error & { code?: string }).code = code
102
- }
103
- throw err
104
- }
105
- }
106
- } catch (error: unknown) {
107
- // Narrow before logging: raw SDK errors can carry request metadata
108
- // (including auth headers) which we must never surface to user loggers.
109
- options.logger.errors(`${this.name}.summarize fatal`, {
110
- error: toRunErrorPayload(error, `${this.name}.summarize failed`),
111
- source: `${this.name}.summarize`,
112
- })
113
- throw error
114
- }
115
-
116
- return { id, model, summary, usage }
117
- }
118
-
119
- async *summarizeStream(
120
- options: SummarizationOptions,
121
- ): AsyncIterable<StreamChunk> {
122
- const systemPrompt = this.buildSummarizationPrompt(options)
123
-
124
- options.logger.request(
125
- `activity=summarizeStream provider=${this.name} model=${options.model} text-length=${options.text.length} maxLength=${options.maxLength ?? 'unset'}`,
126
- { provider: this.name, model: options.model },
127
- )
128
-
129
- try {
130
- yield* this.textAdapter.chatStream({
131
- model: options.model,
132
- messages: [{ role: 'user', content: options.text }],
133
- systemPrompts: [systemPrompt],
134
- maxTokens: options.maxLength,
135
- temperature: 0.3,
136
- logger: options.logger,
137
- } satisfies TextOptions<TProviderOptions>)
138
- } catch (error: unknown) {
139
- options.logger.errors(`${this.name}.summarizeStream fatal`, {
140
- error: toRunErrorPayload(error, `${this.name}.summarizeStream failed`),
141
- source: `${this.name}.summarizeStream`,
142
- })
143
- throw error
144
- }
145
- }
146
-
147
- protected buildSummarizationPrompt(options: SummarizationOptions): string {
148
- let prompt = 'You are a professional summarizer. '
149
-
150
- switch (options.style) {
151
- case 'bullet-points':
152
- prompt += 'Provide a summary in bullet point format. '
153
- break
154
- case 'paragraph':
155
- prompt += 'Provide a summary in paragraph format. '
156
- break
157
- case 'concise':
158
- prompt += 'Provide a very concise summary in 1-2 sentences. '
159
- break
160
- default:
161
- prompt += 'Provide a clear and concise summary. '
162
- }
163
-
164
- if (options.focus && options.focus.length > 0) {
165
- prompt += `Focus on the following aspects: ${options.focus.join(', ')}. `
166
- }
167
-
168
- if (options.maxLength) {
169
- prompt += `Keep the summary under ${options.maxLength} tokens. `
170
- }
171
-
172
- return prompt
173
- }
174
- }
@@ -1,194 +0,0 @@
1
- import { BaseTranscriptionAdapter } from '@tanstack/ai/adapters'
2
- import { toRunErrorPayload } from '@tanstack/ai/adapter-internals'
3
- import { base64ToArrayBuffer, generateId } from '@tanstack/ai-utils'
4
- import { createOpenAICompatibleClient } from '../utils/client'
5
- import type {
6
- TranscriptionOptions,
7
- TranscriptionResult,
8
- TranscriptionSegment,
9
- } from '@tanstack/ai'
10
- import type OpenAI_SDK from 'openai'
11
- import type { OpenAICompatibleClientConfig } from '../types/config'
12
-
13
- /**
14
- * OpenAI-Compatible Transcription (Speech-to-Text) Adapter
15
- *
16
- * A generalized base class for providers that implement OpenAI-compatible audio
17
- * transcription APIs. Providers can extend this class and only need to:
18
- * - Set `baseURL` in the config
19
- * - Lock the generic type parameters to provider-specific types
20
- * - Override audio handling or response mapping methods as needed
21
- *
22
- * All methods that handle audio input or map response formats are `protected`
23
- * so subclasses can override them.
24
- */
25
- export class OpenAICompatibleTranscriptionAdapter<
26
- TModel extends string,
27
- TProviderOptions extends object = Record<string, any>,
28
- > extends BaseTranscriptionAdapter<TModel, TProviderOptions> {
29
- readonly name: string
30
-
31
- protected client: OpenAI_SDK
32
-
33
- constructor(
34
- config: OpenAICompatibleClientConfig,
35
- model: TModel,
36
- name: string = 'openai-compatible',
37
- ) {
38
- super(model, {})
39
- this.name = name
40
- this.client = createOpenAICompatibleClient(config)
41
- }
42
-
43
- async transcribe(
44
- options: TranscriptionOptions<TProviderOptions>,
45
- ): Promise<TranscriptionResult> {
46
- const { model, audio, language, prompt, responseFormat, modelOptions } =
47
- options
48
-
49
- // Convert audio input to File object
50
- const file = this.prepareAudioFile(audio)
51
-
52
- // Build request
53
- const request: OpenAI_SDK.Audio.TranscriptionCreateParams = {
54
- model,
55
- file,
56
- language,
57
- prompt,
58
- response_format: this.mapResponseFormat(responseFormat),
59
- ...modelOptions,
60
- }
61
-
62
- // Call API - use verbose_json to get timestamps when available
63
- const useVerbose =
64
- responseFormat === 'verbose_json' ||
65
- (!responseFormat && this.shouldDefaultToVerbose(model))
66
-
67
- try {
68
- options.logger.request(
69
- `activity=transcription provider=${this.name} model=${model} verbose=${useVerbose}`,
70
- { provider: this.name, model },
71
- )
72
- if (useVerbose) {
73
- const response = await this.client.audio.transcriptions.create({
74
- ...request,
75
- response_format: 'verbose_json',
76
- })
77
-
78
- return {
79
- id: generateId(this.name),
80
- model,
81
- text: response.text,
82
- language: response.language,
83
- duration: response.duration,
84
- segments: response.segments?.map(
85
- (seg): TranscriptionSegment => ({
86
- id: seg.id,
87
- start: seg.start,
88
- end: seg.end,
89
- text: seg.text,
90
- // The OpenAI SDK types `avg_logprob` as `number`, so call Math.exp
91
- // directly. Previously this was guarded with `seg.avg_logprob ?`
92
- // which treated `0` (perfect-confidence) as missing.
93
- confidence: Math.exp(seg.avg_logprob),
94
- }),
95
- ),
96
- words: response.words?.map((w) => ({
97
- word: w.word,
98
- start: w.start,
99
- end: w.end,
100
- })),
101
- }
102
- } else {
103
- const response = await this.client.audio.transcriptions.create(request)
104
-
105
- return {
106
- id: generateId(this.name),
107
- model,
108
- text: typeof response === 'string' ? response : response.text,
109
- language,
110
- }
111
- }
112
- } catch (error: unknown) {
113
- // Narrow before logging: raw SDK errors can carry request metadata
114
- // (including auth headers) which we must never surface to user loggers.
115
- options.logger.errors(`${this.name}.transcribe fatal`, {
116
- error: toRunErrorPayload(error, `${this.name}.transcribe failed`),
117
- source: `${this.name}.transcribe`,
118
- })
119
- throw error
120
- }
121
- }
122
-
123
- protected prepareAudioFile(audio: string | File | Blob | ArrayBuffer): File {
124
- // If already a File, return it
125
- if (typeof File !== 'undefined' && audio instanceof File) {
126
- return audio
127
- }
128
-
129
- // If Blob, convert to File
130
- if (typeof Blob !== 'undefined' && audio instanceof Blob) {
131
- this.ensureFileSupport()
132
- return new File([audio], 'audio.mp3', {
133
- type: audio.type || 'audio/mpeg',
134
- })
135
- }
136
-
137
- // If ArrayBuffer, convert to File
138
- if (typeof ArrayBuffer !== 'undefined' && audio instanceof ArrayBuffer) {
139
- this.ensureFileSupport()
140
- return new File([audio], 'audio.mp3', { type: 'audio/mpeg' })
141
- }
142
-
143
- // If base64 string, decode and convert to File
144
- if (typeof audio === 'string') {
145
- this.ensureFileSupport()
146
-
147
- // Check if it's a data URL
148
- if (audio.startsWith('data:')) {
149
- const parts = audio.split(',')
150
- const header = parts[0]
151
- const base64Data = parts[1] || ''
152
- const mimeMatch = header?.match(/data:([^;]+)/)
153
- const mimeType = mimeMatch?.[1] || 'audio/mpeg'
154
- const bytes = base64ToArrayBuffer(base64Data)
155
- const extension = mimeType.split('/')[1] || 'mp3'
156
- return new File([bytes], `audio.${extension}`, { type: mimeType })
157
- }
158
-
159
- // Assume raw base64
160
- const bytes = base64ToArrayBuffer(audio)
161
- return new File([bytes], 'audio.mp3', { type: 'audio/mpeg' })
162
- }
163
-
164
- throw new Error('Invalid audio input type')
165
- }
166
-
167
- /**
168
- * Checks that the global `File` constructor is available.
169
- * Throws a descriptive error in environments that lack it (e.g. Node < 20).
170
- */
171
- private ensureFileSupport(): void {
172
- if (typeof File === 'undefined') {
173
- throw new Error(
174
- '`File` is not available in this environment. ' +
175
- 'Use Node.js 20 or newer, or pass a File object directly.',
176
- )
177
- }
178
- }
179
-
180
- /**
181
- * Whether the adapter should default to verbose_json when no response format is specified.
182
- * Override in provider-specific subclasses for model-specific behavior.
183
- */
184
- protected shouldDefaultToVerbose(_model: string): boolean {
185
- return false
186
- }
187
-
188
- protected mapResponseFormat(
189
- format?: 'json' | 'text' | 'srt' | 'verbose_json' | 'vtt',
190
- ): OpenAI_SDK.Audio.TranscriptionCreateParams['response_format'] {
191
- if (!format) return 'json'
192
- return format as OpenAI_SDK.Audio.TranscriptionCreateParams['response_format']
193
- }
194
- }
@@ -1,124 +0,0 @@
1
- import { BaseTTSAdapter } from '@tanstack/ai/adapters'
2
- import { toRunErrorPayload } from '@tanstack/ai/adapter-internals'
3
- import { arrayBufferToBase64, generateId } from '@tanstack/ai-utils'
4
- import { createOpenAICompatibleClient } from '../utils/client'
5
- import type { TTSOptions, TTSResult } from '@tanstack/ai'
6
- import type OpenAI_SDK from 'openai'
7
- import type { OpenAICompatibleClientConfig } from '../types/config'
8
-
9
- /**
10
- * OpenAI-Compatible Text-to-Speech Adapter
11
- *
12
- * A generalized base class for providers that implement OpenAI-compatible TTS APIs.
13
- * Providers can extend this class and only need to:
14
- * - Set `baseURL` in the config
15
- * - Lock the generic type parameters to provider-specific types
16
- * - Override validation methods or request building for provider-specific constraints
17
- *
18
- * All methods that validate inputs or build requests are `protected` so subclasses
19
- * can override them.
20
- */
21
- export class OpenAICompatibleTTSAdapter<
22
- TModel extends string,
23
- TProviderOptions extends object = Record<string, any>,
24
- > extends BaseTTSAdapter<TModel, TProviderOptions> {
25
- readonly name: string
26
-
27
- protected client: OpenAI_SDK
28
-
29
- constructor(
30
- config: OpenAICompatibleClientConfig,
31
- model: TModel,
32
- name: string = 'openai-compatible',
33
- ) {
34
- super(model, {})
35
- this.name = name
36
- this.client = createOpenAICompatibleClient(config)
37
- }
38
-
39
- async generateSpeech(
40
- options: TTSOptions<TProviderOptions>,
41
- ): Promise<TTSResult> {
42
- const { model, text, voice, format, speed, modelOptions } = options
43
-
44
- // Validate inputs
45
- this.validateAudioInput(text)
46
- this.validateSpeed(speed)
47
- this.validateInstructions(model, modelOptions)
48
-
49
- // Build request
50
- const request: OpenAI_SDK.Audio.SpeechCreateParams = {
51
- model,
52
- input: text,
53
- voice: (voice || 'alloy') as OpenAI_SDK.Audio.SpeechCreateParams['voice'],
54
- response_format: format,
55
- speed,
56
- ...modelOptions,
57
- }
58
-
59
- try {
60
- options.logger.request(
61
- `activity=tts provider=${this.name} model=${model} format=${request.response_format ?? 'default'} voice=${request.voice}`,
62
- { provider: this.name, model },
63
- )
64
- const response = await this.client.audio.speech.create(request)
65
-
66
- // Convert response to base64. Buffer is Node-only; use atob fallback in
67
- // browser/edge runtimes where the SDK can run.
68
- const arrayBuffer = await response.arrayBuffer()
69
- const base64 = arrayBufferToBase64(arrayBuffer)
70
-
71
- const outputFormat = (request.response_format as string) || 'mp3'
72
- const contentType = this.getContentType(outputFormat)
73
-
74
- return {
75
- id: generateId(this.name),
76
- model,
77
- audio: base64,
78
- format: outputFormat,
79
- contentType,
80
- }
81
- } catch (error: unknown) {
82
- // Narrow before logging: raw SDK errors can carry request metadata
83
- // (including auth headers) which we must never surface to user loggers.
84
- options.logger.errors(`${this.name}.generateSpeech fatal`, {
85
- error: toRunErrorPayload(error, `${this.name}.generateSpeech failed`),
86
- source: `${this.name}.generateSpeech`,
87
- })
88
- throw error
89
- }
90
- }
91
-
92
- protected validateAudioInput(text: string): void {
93
- if (text.length > 4096) {
94
- throw new Error('Input text exceeds maximum length of 4096 characters.')
95
- }
96
- }
97
-
98
- protected validateSpeed(speed?: number): void {
99
- if (speed !== undefined) {
100
- if (speed < 0.25 || speed > 4.0) {
101
- throw new Error('Speed must be between 0.25 and 4.0.')
102
- }
103
- }
104
- }
105
-
106
- protected validateInstructions(
107
- _model: string,
108
- _modelOptions?: TProviderOptions,
109
- ): void {
110
- // Default: no instructions validation — subclasses can override
111
- }
112
-
113
- protected getContentType(format: string): string {
114
- const contentTypes: Record<string, string> = {
115
- mp3: 'audio/mpeg',
116
- opus: 'audio/opus',
117
- aac: 'audio/aac',
118
- flac: 'audio/flac',
119
- wav: 'audio/wav',
120
- pcm: 'audio/pcm',
121
- }
122
- return contentTypes[format] || 'audio/mpeg'
123
- }
124
- }