@tanstack/openai-base 0.2.0 → 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +121 -0
- package/dist/esm/adapters/chat-completions-text.d.ts +49 -21
- package/dist/esm/adapters/chat-completions-text.js +476 -68
- package/dist/esm/adapters/chat-completions-text.js.map +1 -1
- package/dist/esm/adapters/chat-completions-tool-converter.d.ts +8 -4
- package/dist/esm/adapters/chat-completions-tool-converter.js.map +1 -1
- package/dist/esm/adapters/responses-text.d.ts +46 -33
- package/dist/esm/adapters/responses-text.js +657 -142
- package/dist/esm/adapters/responses-text.js.map +1 -1
- package/dist/esm/index.d.ts +2 -9
- package/dist/esm/index.js +4 -16
- package/dist/esm/index.js.map +1 -1
- package/dist/esm/tools/apply-patch-tool.d.ts +2 -2
- package/dist/esm/tools/apply-patch-tool.js.map +1 -1
- package/dist/esm/tools/code-interpreter-tool.d.ts +3 -2
- package/dist/esm/tools/code-interpreter-tool.js.map +1 -1
- package/dist/esm/tools/computer-use-tool.d.ts +2 -2
- package/dist/esm/tools/computer-use-tool.js.map +1 -1
- package/dist/esm/tools/custom-tool.d.ts +2 -2
- package/dist/esm/tools/custom-tool.js.map +1 -1
- package/dist/esm/tools/file-search-tool.d.ts +2 -2
- package/dist/esm/tools/file-search-tool.js.map +1 -1
- package/dist/esm/tools/function-tool.d.ts +2 -2
- package/dist/esm/tools/function-tool.js.map +1 -1
- package/dist/esm/tools/image-generation-tool.d.ts +3 -2
- package/dist/esm/tools/image-generation-tool.js.map +1 -1
- package/dist/esm/tools/local-shell-tool.d.ts +3 -2
- package/dist/esm/tools/local-shell-tool.js.map +1 -1
- package/dist/esm/tools/mcp-tool.d.ts +3 -2
- package/dist/esm/tools/mcp-tool.js.map +1 -1
- package/dist/esm/tools/shell-tool.d.ts +2 -2
- package/dist/esm/tools/shell-tool.js.map +1 -1
- package/dist/esm/tools/web-search-preview-tool.d.ts +2 -2
- package/dist/esm/tools/web-search-preview-tool.js.map +1 -1
- package/dist/esm/tools/web-search-tool.d.ts +2 -2
- package/dist/esm/tools/web-search-tool.js.map +1 -1
- package/package.json +6 -6
- package/src/adapters/chat-completions-text.ts +601 -117
- package/src/adapters/chat-completions-tool-converter.ts +9 -5
- package/src/adapters/responses-text.ts +865 -210
- package/src/index.ts +2 -12
- package/src/tools/apply-patch-tool.ts +2 -2
- package/src/tools/code-interpreter-tool.ts +4 -2
- package/src/tools/computer-use-tool.ts +2 -2
- package/src/tools/custom-tool.ts +2 -2
- package/src/tools/file-search-tool.ts +3 -3
- package/src/tools/function-tool.ts +2 -2
- package/src/tools/image-generation-tool.ts +4 -2
- package/src/tools/local-shell-tool.ts +4 -2
- package/src/tools/mcp-tool.ts +4 -2
- package/src/tools/shell-tool.ts +2 -2
- package/src/tools/web-search-preview-tool.ts +2 -2
- package/src/tools/web-search-tool.ts +2 -2
- package/dist/esm/adapters/image.d.ts +0 -32
- package/dist/esm/adapters/image.js +0 -89
- package/dist/esm/adapters/image.js.map +0 -1
- package/dist/esm/adapters/summarize.d.ts +0 -28
- package/dist/esm/adapters/summarize.js +0 -112
- package/dist/esm/adapters/summarize.js.map +0 -1
- package/dist/esm/adapters/transcription.d.ts +0 -34
- package/dist/esm/adapters/transcription.js +0 -131
- package/dist/esm/adapters/transcription.js.map +0 -1
- package/dist/esm/adapters/tts.d.ts +0 -26
- package/dist/esm/adapters/tts.js +0 -78
- package/dist/esm/adapters/tts.js.map +0 -1
- package/dist/esm/adapters/video.d.ts +0 -72
- package/dist/esm/adapters/video.js +0 -238
- package/dist/esm/adapters/video.js.map +0 -1
- package/dist/esm/types/config.d.ts +0 -4
- package/dist/esm/utils/client.d.ts +0 -3
- package/dist/esm/utils/client.js +0 -8
- package/dist/esm/utils/client.js.map +0 -1
- package/src/adapters/image.ts +0 -158
- package/src/adapters/summarize.ts +0 -174
- package/src/adapters/transcription.ts +0 -194
- package/src/adapters/tts.ts +0 -124
- package/src/adapters/video.ts +0 -385
- package/src/types/config.ts +0 -5
- package/src/utils/client.ts +0 -8
package/src/adapters/image.ts
DELETED
|
@@ -1,158 +0,0 @@
|
|
|
1
|
-
import { BaseImageAdapter } from '@tanstack/ai/adapters'
|
|
2
|
-
import { toRunErrorPayload } from '@tanstack/ai/adapter-internals'
|
|
3
|
-
import { generateId } from '@tanstack/ai-utils'
|
|
4
|
-
import { createOpenAICompatibleClient } from '../utils/client'
|
|
5
|
-
import type {
|
|
6
|
-
GeneratedImage,
|
|
7
|
-
ImageGenerationOptions,
|
|
8
|
-
ImageGenerationResult,
|
|
9
|
-
} from '@tanstack/ai'
|
|
10
|
-
import type OpenAI_SDK from 'openai'
|
|
11
|
-
import type { OpenAICompatibleClientConfig } from '../types/config'
|
|
12
|
-
|
|
13
|
-
/**
|
|
14
|
-
* OpenAI-Compatible Image Generation Adapter
|
|
15
|
-
*
|
|
16
|
-
* A generalized base class for providers that implement OpenAI-compatible image
|
|
17
|
-
* generation APIs. Providers like OpenAI, Grok, and others can extend this class
|
|
18
|
-
* and only need to:
|
|
19
|
-
* - Set `baseURL` in the config
|
|
20
|
-
* - Lock the generic type parameters to provider-specific types
|
|
21
|
-
* - Override validation or request building methods for provider-specific constraints
|
|
22
|
-
*
|
|
23
|
-
* All methods that validate inputs, build requests, or transform responses are
|
|
24
|
-
* `protected` so subclasses can override them.
|
|
25
|
-
*/
|
|
26
|
-
export class OpenAICompatibleImageAdapter<
|
|
27
|
-
TModel extends string,
|
|
28
|
-
TProviderOptions extends object = Record<string, any>,
|
|
29
|
-
TModelProviderOptionsByName extends Record<string, any> = Record<string, any>,
|
|
30
|
-
TModelSizeByName extends Record<string, string> = Record<string, string>,
|
|
31
|
-
> extends BaseImageAdapter<
|
|
32
|
-
TModel,
|
|
33
|
-
TProviderOptions,
|
|
34
|
-
TModelProviderOptionsByName,
|
|
35
|
-
TModelSizeByName
|
|
36
|
-
> {
|
|
37
|
-
readonly kind = 'image' as const
|
|
38
|
-
readonly name: string
|
|
39
|
-
|
|
40
|
-
protected client: OpenAI_SDK
|
|
41
|
-
|
|
42
|
-
constructor(
|
|
43
|
-
config: OpenAICompatibleClientConfig,
|
|
44
|
-
model: TModel,
|
|
45
|
-
name: string = 'openai-compatible',
|
|
46
|
-
) {
|
|
47
|
-
super(model, {})
|
|
48
|
-
this.name = name
|
|
49
|
-
this.client = createOpenAICompatibleClient(config)
|
|
50
|
-
}
|
|
51
|
-
|
|
52
|
-
async generateImages(
|
|
53
|
-
options: ImageGenerationOptions<TProviderOptions>,
|
|
54
|
-
): Promise<ImageGenerationResult> {
|
|
55
|
-
const { model, prompt, numberOfImages, size } = options
|
|
56
|
-
|
|
57
|
-
// Validate inputs
|
|
58
|
-
this.validatePrompt({ prompt, model })
|
|
59
|
-
this.validateImageSize(model, size)
|
|
60
|
-
this.validateNumberOfImages(model, numberOfImages)
|
|
61
|
-
|
|
62
|
-
// Build request based on model type
|
|
63
|
-
const request = this.buildRequest(options)
|
|
64
|
-
|
|
65
|
-
try {
|
|
66
|
-
options.logger.request(
|
|
67
|
-
`activity=image provider=${this.name} model=${model} n=${request.n ?? 1} size=${request.size ?? 'default'}`,
|
|
68
|
-
{ provider: this.name, model },
|
|
69
|
-
)
|
|
70
|
-
const response = await this.client.images.generate({
|
|
71
|
-
...request,
|
|
72
|
-
stream: false,
|
|
73
|
-
})
|
|
74
|
-
|
|
75
|
-
return this.transformResponse(model, response)
|
|
76
|
-
} catch (error: unknown) {
|
|
77
|
-
// Narrow before logging: raw SDK errors can carry request metadata
|
|
78
|
-
// (including auth headers) which we must never surface to user loggers.
|
|
79
|
-
options.logger.errors(`${this.name}.generateImages fatal`, {
|
|
80
|
-
error: toRunErrorPayload(error, `${this.name}.generateImages failed`),
|
|
81
|
-
source: `${this.name}.generateImages`,
|
|
82
|
-
})
|
|
83
|
-
throw error
|
|
84
|
-
}
|
|
85
|
-
}
|
|
86
|
-
|
|
87
|
-
protected buildRequest(
|
|
88
|
-
options: ImageGenerationOptions<TProviderOptions>,
|
|
89
|
-
): OpenAI_SDK.Images.ImageGenerateParams {
|
|
90
|
-
const { model, prompt, numberOfImages, size, modelOptions } = options
|
|
91
|
-
|
|
92
|
-
return {
|
|
93
|
-
model,
|
|
94
|
-
prompt,
|
|
95
|
-
n: numberOfImages ?? 1,
|
|
96
|
-
size: size as OpenAI_SDK.Images.ImageGenerateParams['size'],
|
|
97
|
-
...modelOptions,
|
|
98
|
-
}
|
|
99
|
-
}
|
|
100
|
-
|
|
101
|
-
protected transformResponse(
|
|
102
|
-
model: string,
|
|
103
|
-
response: OpenAI_SDK.Images.ImagesResponse,
|
|
104
|
-
): ImageGenerationResult {
|
|
105
|
-
const images: Array<GeneratedImage> = (response.data ?? []).flatMap(
|
|
106
|
-
(item): Array<GeneratedImage> => {
|
|
107
|
-
const revisedPrompt = item.revised_prompt
|
|
108
|
-
if (item.b64_json) {
|
|
109
|
-
return [{ b64Json: item.b64_json, revisedPrompt }]
|
|
110
|
-
}
|
|
111
|
-
if (item.url) {
|
|
112
|
-
return [{ url: item.url, revisedPrompt }]
|
|
113
|
-
}
|
|
114
|
-
return []
|
|
115
|
-
},
|
|
116
|
-
)
|
|
117
|
-
|
|
118
|
-
return {
|
|
119
|
-
id: generateId(this.name),
|
|
120
|
-
model,
|
|
121
|
-
images,
|
|
122
|
-
usage: response.usage
|
|
123
|
-
? {
|
|
124
|
-
inputTokens: response.usage.input_tokens,
|
|
125
|
-
outputTokens: response.usage.output_tokens,
|
|
126
|
-
totalTokens: response.usage.total_tokens,
|
|
127
|
-
}
|
|
128
|
-
: undefined,
|
|
129
|
-
}
|
|
130
|
-
}
|
|
131
|
-
|
|
132
|
-
protected validatePrompt(options: { prompt: string; model: string }): void {
|
|
133
|
-
if (options.prompt.length === 0) {
|
|
134
|
-
throw new Error('Prompt cannot be empty.')
|
|
135
|
-
}
|
|
136
|
-
}
|
|
137
|
-
|
|
138
|
-
protected validateImageSize(_model: string, _size: string | undefined): void {
|
|
139
|
-
// Default: no size validation — subclasses can override
|
|
140
|
-
}
|
|
141
|
-
|
|
142
|
-
protected validateNumberOfImages(
|
|
143
|
-
_model: string,
|
|
144
|
-
numberOfImages: number | undefined,
|
|
145
|
-
): void {
|
|
146
|
-
if (numberOfImages === undefined) return
|
|
147
|
-
|
|
148
|
-
// The base adapter only enforces "must be at least 1". Per-provider /
|
|
149
|
-
// per-model upper bounds vary widely (some support 4, some 10, some
|
|
150
|
-
// unlimited), so concrete adapter subclasses are expected to override
|
|
151
|
-
// this method with a model-specific cap.
|
|
152
|
-
if (numberOfImages < 1) {
|
|
153
|
-
throw new Error(
|
|
154
|
-
`Number of images must be at least 1. Requested: ${numberOfImages}`,
|
|
155
|
-
)
|
|
156
|
-
}
|
|
157
|
-
}
|
|
158
|
-
}
|
|
@@ -1,174 +0,0 @@
|
|
|
1
|
-
import { BaseSummarizeAdapter } from '@tanstack/ai/adapters'
|
|
2
|
-
import { toRunErrorPayload } from '@tanstack/ai/adapter-internals'
|
|
3
|
-
import { generateId } from '@tanstack/ai-utils'
|
|
4
|
-
import type {
|
|
5
|
-
StreamChunk,
|
|
6
|
-
SummarizationOptions,
|
|
7
|
-
SummarizationResult,
|
|
8
|
-
TextOptions,
|
|
9
|
-
} from '@tanstack/ai'
|
|
10
|
-
|
|
11
|
-
/**
|
|
12
|
-
* Minimal interface for a text adapter that supports chatStream.
|
|
13
|
-
* This allows the summarize adapter to work with any OpenAI-compatible
|
|
14
|
-
* text adapter without tight coupling to a specific implementation.
|
|
15
|
-
*/
|
|
16
|
-
export interface ChatStreamCapable<TProviderOptions extends object> {
|
|
17
|
-
chatStream: (
|
|
18
|
-
options: TextOptions<TProviderOptions>,
|
|
19
|
-
) => AsyncIterable<StreamChunk>
|
|
20
|
-
}
|
|
21
|
-
|
|
22
|
-
/**
|
|
23
|
-
* OpenAI-Compatible Summarize Adapter
|
|
24
|
-
*
|
|
25
|
-
* A thin wrapper around a text adapter that adds summarization-specific prompting.
|
|
26
|
-
* Delegates all API calls to the provided text adapter.
|
|
27
|
-
*
|
|
28
|
-
* Subclasses or instantiators provide a text adapter (or factory) at construction
|
|
29
|
-
* time, allowing any OpenAI-compatible provider to get summarization for free by
|
|
30
|
-
* reusing its text adapter.
|
|
31
|
-
*/
|
|
32
|
-
export class OpenAICompatibleSummarizeAdapter<
|
|
33
|
-
TModel extends string,
|
|
34
|
-
TProviderOptions extends object = Record<string, any>,
|
|
35
|
-
> extends BaseSummarizeAdapter<TModel, TProviderOptions> {
|
|
36
|
-
readonly name: string
|
|
37
|
-
|
|
38
|
-
private textAdapter: ChatStreamCapable<TProviderOptions>
|
|
39
|
-
|
|
40
|
-
constructor(
|
|
41
|
-
textAdapter: ChatStreamCapable<TProviderOptions>,
|
|
42
|
-
model: TModel,
|
|
43
|
-
name: string = 'openai-compatible',
|
|
44
|
-
) {
|
|
45
|
-
super({}, model)
|
|
46
|
-
this.name = name
|
|
47
|
-
this.textAdapter = textAdapter
|
|
48
|
-
}
|
|
49
|
-
|
|
50
|
-
async summarize(options: SummarizationOptions): Promise<SummarizationResult> {
|
|
51
|
-
const systemPrompt = this.buildSummarizationPrompt(options)
|
|
52
|
-
|
|
53
|
-
let summary = ''
|
|
54
|
-
const id = generateId(this.name)
|
|
55
|
-
let model = options.model
|
|
56
|
-
let usage = { promptTokens: 0, completionTokens: 0, totalTokens: 0 }
|
|
57
|
-
|
|
58
|
-
options.logger.request(
|
|
59
|
-
`activity=summarize provider=${this.name} model=${options.model} text-length=${options.text.length} maxLength=${options.maxLength ?? 'unset'}`,
|
|
60
|
-
{ provider: this.name, model: options.model },
|
|
61
|
-
)
|
|
62
|
-
|
|
63
|
-
try {
|
|
64
|
-
for await (const chunk of this.textAdapter.chatStream({
|
|
65
|
-
model: options.model,
|
|
66
|
-
messages: [{ role: 'user', content: options.text }],
|
|
67
|
-
systemPrompts: [systemPrompt],
|
|
68
|
-
maxTokens: options.maxLength,
|
|
69
|
-
temperature: 0.3,
|
|
70
|
-
logger: options.logger,
|
|
71
|
-
} satisfies TextOptions<TProviderOptions>)) {
|
|
72
|
-
if (chunk.type === 'TEXT_MESSAGE_CONTENT') {
|
|
73
|
-
if (chunk.content) {
|
|
74
|
-
summary = chunk.content
|
|
75
|
-
} else if (chunk.delta) {
|
|
76
|
-
// Append delta only when present — a content-less chunk with no
|
|
77
|
-
// delta would otherwise concat literal `'undefined'`.
|
|
78
|
-
summary += chunk.delta
|
|
79
|
-
}
|
|
80
|
-
model = chunk.model || model
|
|
81
|
-
}
|
|
82
|
-
if (chunk.type === 'RUN_FINISHED') {
|
|
83
|
-
if (chunk.usage) {
|
|
84
|
-
usage = chunk.usage
|
|
85
|
-
}
|
|
86
|
-
}
|
|
87
|
-
// Surface failures: the underlying chatStream emits RUN_ERROR instead
|
|
88
|
-
// of throwing, so without this branch summarize() would return an
|
|
89
|
-
// empty summary and pretend a failed run succeeded.
|
|
90
|
-
if (chunk.type === 'RUN_ERROR') {
|
|
91
|
-
const message =
|
|
92
|
-
(chunk.error && typeof chunk.error.message === 'string'
|
|
93
|
-
? chunk.error.message
|
|
94
|
-
: null) ?? 'Summarization failed'
|
|
95
|
-
const code =
|
|
96
|
-
chunk.error && typeof chunk.error.code === 'string'
|
|
97
|
-
? chunk.error.code
|
|
98
|
-
: undefined
|
|
99
|
-
const err = new Error(message)
|
|
100
|
-
if (code) {
|
|
101
|
-
;(err as Error & { code?: string }).code = code
|
|
102
|
-
}
|
|
103
|
-
throw err
|
|
104
|
-
}
|
|
105
|
-
}
|
|
106
|
-
} catch (error: unknown) {
|
|
107
|
-
// Narrow before logging: raw SDK errors can carry request metadata
|
|
108
|
-
// (including auth headers) which we must never surface to user loggers.
|
|
109
|
-
options.logger.errors(`${this.name}.summarize fatal`, {
|
|
110
|
-
error: toRunErrorPayload(error, `${this.name}.summarize failed`),
|
|
111
|
-
source: `${this.name}.summarize`,
|
|
112
|
-
})
|
|
113
|
-
throw error
|
|
114
|
-
}
|
|
115
|
-
|
|
116
|
-
return { id, model, summary, usage }
|
|
117
|
-
}
|
|
118
|
-
|
|
119
|
-
async *summarizeStream(
|
|
120
|
-
options: SummarizationOptions,
|
|
121
|
-
): AsyncIterable<StreamChunk> {
|
|
122
|
-
const systemPrompt = this.buildSummarizationPrompt(options)
|
|
123
|
-
|
|
124
|
-
options.logger.request(
|
|
125
|
-
`activity=summarizeStream provider=${this.name} model=${options.model} text-length=${options.text.length} maxLength=${options.maxLength ?? 'unset'}`,
|
|
126
|
-
{ provider: this.name, model: options.model },
|
|
127
|
-
)
|
|
128
|
-
|
|
129
|
-
try {
|
|
130
|
-
yield* this.textAdapter.chatStream({
|
|
131
|
-
model: options.model,
|
|
132
|
-
messages: [{ role: 'user', content: options.text }],
|
|
133
|
-
systemPrompts: [systemPrompt],
|
|
134
|
-
maxTokens: options.maxLength,
|
|
135
|
-
temperature: 0.3,
|
|
136
|
-
logger: options.logger,
|
|
137
|
-
} satisfies TextOptions<TProviderOptions>)
|
|
138
|
-
} catch (error: unknown) {
|
|
139
|
-
options.logger.errors(`${this.name}.summarizeStream fatal`, {
|
|
140
|
-
error: toRunErrorPayload(error, `${this.name}.summarizeStream failed`),
|
|
141
|
-
source: `${this.name}.summarizeStream`,
|
|
142
|
-
})
|
|
143
|
-
throw error
|
|
144
|
-
}
|
|
145
|
-
}
|
|
146
|
-
|
|
147
|
-
protected buildSummarizationPrompt(options: SummarizationOptions): string {
|
|
148
|
-
let prompt = 'You are a professional summarizer. '
|
|
149
|
-
|
|
150
|
-
switch (options.style) {
|
|
151
|
-
case 'bullet-points':
|
|
152
|
-
prompt += 'Provide a summary in bullet point format. '
|
|
153
|
-
break
|
|
154
|
-
case 'paragraph':
|
|
155
|
-
prompt += 'Provide a summary in paragraph format. '
|
|
156
|
-
break
|
|
157
|
-
case 'concise':
|
|
158
|
-
prompt += 'Provide a very concise summary in 1-2 sentences. '
|
|
159
|
-
break
|
|
160
|
-
default:
|
|
161
|
-
prompt += 'Provide a clear and concise summary. '
|
|
162
|
-
}
|
|
163
|
-
|
|
164
|
-
if (options.focus && options.focus.length > 0) {
|
|
165
|
-
prompt += `Focus on the following aspects: ${options.focus.join(', ')}. `
|
|
166
|
-
}
|
|
167
|
-
|
|
168
|
-
if (options.maxLength) {
|
|
169
|
-
prompt += `Keep the summary under ${options.maxLength} tokens. `
|
|
170
|
-
}
|
|
171
|
-
|
|
172
|
-
return prompt
|
|
173
|
-
}
|
|
174
|
-
}
|
|
@@ -1,194 +0,0 @@
|
|
|
1
|
-
import { BaseTranscriptionAdapter } from '@tanstack/ai/adapters'
|
|
2
|
-
import { toRunErrorPayload } from '@tanstack/ai/adapter-internals'
|
|
3
|
-
import { base64ToArrayBuffer, generateId } from '@tanstack/ai-utils'
|
|
4
|
-
import { createOpenAICompatibleClient } from '../utils/client'
|
|
5
|
-
import type {
|
|
6
|
-
TranscriptionOptions,
|
|
7
|
-
TranscriptionResult,
|
|
8
|
-
TranscriptionSegment,
|
|
9
|
-
} from '@tanstack/ai'
|
|
10
|
-
import type OpenAI_SDK from 'openai'
|
|
11
|
-
import type { OpenAICompatibleClientConfig } from '../types/config'
|
|
12
|
-
|
|
13
|
-
/**
|
|
14
|
-
* OpenAI-Compatible Transcription (Speech-to-Text) Adapter
|
|
15
|
-
*
|
|
16
|
-
* A generalized base class for providers that implement OpenAI-compatible audio
|
|
17
|
-
* transcription APIs. Providers can extend this class and only need to:
|
|
18
|
-
* - Set `baseURL` in the config
|
|
19
|
-
* - Lock the generic type parameters to provider-specific types
|
|
20
|
-
* - Override audio handling or response mapping methods as needed
|
|
21
|
-
*
|
|
22
|
-
* All methods that handle audio input or map response formats are `protected`
|
|
23
|
-
* so subclasses can override them.
|
|
24
|
-
*/
|
|
25
|
-
export class OpenAICompatibleTranscriptionAdapter<
|
|
26
|
-
TModel extends string,
|
|
27
|
-
TProviderOptions extends object = Record<string, any>,
|
|
28
|
-
> extends BaseTranscriptionAdapter<TModel, TProviderOptions> {
|
|
29
|
-
readonly name: string
|
|
30
|
-
|
|
31
|
-
protected client: OpenAI_SDK
|
|
32
|
-
|
|
33
|
-
constructor(
|
|
34
|
-
config: OpenAICompatibleClientConfig,
|
|
35
|
-
model: TModel,
|
|
36
|
-
name: string = 'openai-compatible',
|
|
37
|
-
) {
|
|
38
|
-
super(model, {})
|
|
39
|
-
this.name = name
|
|
40
|
-
this.client = createOpenAICompatibleClient(config)
|
|
41
|
-
}
|
|
42
|
-
|
|
43
|
-
async transcribe(
|
|
44
|
-
options: TranscriptionOptions<TProviderOptions>,
|
|
45
|
-
): Promise<TranscriptionResult> {
|
|
46
|
-
const { model, audio, language, prompt, responseFormat, modelOptions } =
|
|
47
|
-
options
|
|
48
|
-
|
|
49
|
-
// Convert audio input to File object
|
|
50
|
-
const file = this.prepareAudioFile(audio)
|
|
51
|
-
|
|
52
|
-
// Build request
|
|
53
|
-
const request: OpenAI_SDK.Audio.TranscriptionCreateParams = {
|
|
54
|
-
model,
|
|
55
|
-
file,
|
|
56
|
-
language,
|
|
57
|
-
prompt,
|
|
58
|
-
response_format: this.mapResponseFormat(responseFormat),
|
|
59
|
-
...modelOptions,
|
|
60
|
-
}
|
|
61
|
-
|
|
62
|
-
// Call API - use verbose_json to get timestamps when available
|
|
63
|
-
const useVerbose =
|
|
64
|
-
responseFormat === 'verbose_json' ||
|
|
65
|
-
(!responseFormat && this.shouldDefaultToVerbose(model))
|
|
66
|
-
|
|
67
|
-
try {
|
|
68
|
-
options.logger.request(
|
|
69
|
-
`activity=transcription provider=${this.name} model=${model} verbose=${useVerbose}`,
|
|
70
|
-
{ provider: this.name, model },
|
|
71
|
-
)
|
|
72
|
-
if (useVerbose) {
|
|
73
|
-
const response = await this.client.audio.transcriptions.create({
|
|
74
|
-
...request,
|
|
75
|
-
response_format: 'verbose_json',
|
|
76
|
-
})
|
|
77
|
-
|
|
78
|
-
return {
|
|
79
|
-
id: generateId(this.name),
|
|
80
|
-
model,
|
|
81
|
-
text: response.text,
|
|
82
|
-
language: response.language,
|
|
83
|
-
duration: response.duration,
|
|
84
|
-
segments: response.segments?.map(
|
|
85
|
-
(seg): TranscriptionSegment => ({
|
|
86
|
-
id: seg.id,
|
|
87
|
-
start: seg.start,
|
|
88
|
-
end: seg.end,
|
|
89
|
-
text: seg.text,
|
|
90
|
-
// The OpenAI SDK types `avg_logprob` as `number`, so call Math.exp
|
|
91
|
-
// directly. Previously this was guarded with `seg.avg_logprob ?`
|
|
92
|
-
// which treated `0` (perfect-confidence) as missing.
|
|
93
|
-
confidence: Math.exp(seg.avg_logprob),
|
|
94
|
-
}),
|
|
95
|
-
),
|
|
96
|
-
words: response.words?.map((w) => ({
|
|
97
|
-
word: w.word,
|
|
98
|
-
start: w.start,
|
|
99
|
-
end: w.end,
|
|
100
|
-
})),
|
|
101
|
-
}
|
|
102
|
-
} else {
|
|
103
|
-
const response = await this.client.audio.transcriptions.create(request)
|
|
104
|
-
|
|
105
|
-
return {
|
|
106
|
-
id: generateId(this.name),
|
|
107
|
-
model,
|
|
108
|
-
text: typeof response === 'string' ? response : response.text,
|
|
109
|
-
language,
|
|
110
|
-
}
|
|
111
|
-
}
|
|
112
|
-
} catch (error: unknown) {
|
|
113
|
-
// Narrow before logging: raw SDK errors can carry request metadata
|
|
114
|
-
// (including auth headers) which we must never surface to user loggers.
|
|
115
|
-
options.logger.errors(`${this.name}.transcribe fatal`, {
|
|
116
|
-
error: toRunErrorPayload(error, `${this.name}.transcribe failed`),
|
|
117
|
-
source: `${this.name}.transcribe`,
|
|
118
|
-
})
|
|
119
|
-
throw error
|
|
120
|
-
}
|
|
121
|
-
}
|
|
122
|
-
|
|
123
|
-
protected prepareAudioFile(audio: string | File | Blob | ArrayBuffer): File {
|
|
124
|
-
// If already a File, return it
|
|
125
|
-
if (typeof File !== 'undefined' && audio instanceof File) {
|
|
126
|
-
return audio
|
|
127
|
-
}
|
|
128
|
-
|
|
129
|
-
// If Blob, convert to File
|
|
130
|
-
if (typeof Blob !== 'undefined' && audio instanceof Blob) {
|
|
131
|
-
this.ensureFileSupport()
|
|
132
|
-
return new File([audio], 'audio.mp3', {
|
|
133
|
-
type: audio.type || 'audio/mpeg',
|
|
134
|
-
})
|
|
135
|
-
}
|
|
136
|
-
|
|
137
|
-
// If ArrayBuffer, convert to File
|
|
138
|
-
if (typeof ArrayBuffer !== 'undefined' && audio instanceof ArrayBuffer) {
|
|
139
|
-
this.ensureFileSupport()
|
|
140
|
-
return new File([audio], 'audio.mp3', { type: 'audio/mpeg' })
|
|
141
|
-
}
|
|
142
|
-
|
|
143
|
-
// If base64 string, decode and convert to File
|
|
144
|
-
if (typeof audio === 'string') {
|
|
145
|
-
this.ensureFileSupport()
|
|
146
|
-
|
|
147
|
-
// Check if it's a data URL
|
|
148
|
-
if (audio.startsWith('data:')) {
|
|
149
|
-
const parts = audio.split(',')
|
|
150
|
-
const header = parts[0]
|
|
151
|
-
const base64Data = parts[1] || ''
|
|
152
|
-
const mimeMatch = header?.match(/data:([^;]+)/)
|
|
153
|
-
const mimeType = mimeMatch?.[1] || 'audio/mpeg'
|
|
154
|
-
const bytes = base64ToArrayBuffer(base64Data)
|
|
155
|
-
const extension = mimeType.split('/')[1] || 'mp3'
|
|
156
|
-
return new File([bytes], `audio.${extension}`, { type: mimeType })
|
|
157
|
-
}
|
|
158
|
-
|
|
159
|
-
// Assume raw base64
|
|
160
|
-
const bytes = base64ToArrayBuffer(audio)
|
|
161
|
-
return new File([bytes], 'audio.mp3', { type: 'audio/mpeg' })
|
|
162
|
-
}
|
|
163
|
-
|
|
164
|
-
throw new Error('Invalid audio input type')
|
|
165
|
-
}
|
|
166
|
-
|
|
167
|
-
/**
|
|
168
|
-
* Checks that the global `File` constructor is available.
|
|
169
|
-
* Throws a descriptive error in environments that lack it (e.g. Node < 20).
|
|
170
|
-
*/
|
|
171
|
-
private ensureFileSupport(): void {
|
|
172
|
-
if (typeof File === 'undefined') {
|
|
173
|
-
throw new Error(
|
|
174
|
-
'`File` is not available in this environment. ' +
|
|
175
|
-
'Use Node.js 20 or newer, or pass a File object directly.',
|
|
176
|
-
)
|
|
177
|
-
}
|
|
178
|
-
}
|
|
179
|
-
|
|
180
|
-
/**
|
|
181
|
-
* Whether the adapter should default to verbose_json when no response format is specified.
|
|
182
|
-
* Override in provider-specific subclasses for model-specific behavior.
|
|
183
|
-
*/
|
|
184
|
-
protected shouldDefaultToVerbose(_model: string): boolean {
|
|
185
|
-
return false
|
|
186
|
-
}
|
|
187
|
-
|
|
188
|
-
protected mapResponseFormat(
|
|
189
|
-
format?: 'json' | 'text' | 'srt' | 'verbose_json' | 'vtt',
|
|
190
|
-
): OpenAI_SDK.Audio.TranscriptionCreateParams['response_format'] {
|
|
191
|
-
if (!format) return 'json'
|
|
192
|
-
return format as OpenAI_SDK.Audio.TranscriptionCreateParams['response_format']
|
|
193
|
-
}
|
|
194
|
-
}
|
package/src/adapters/tts.ts
DELETED
|
@@ -1,124 +0,0 @@
|
|
|
1
|
-
import { BaseTTSAdapter } from '@tanstack/ai/adapters'
|
|
2
|
-
import { toRunErrorPayload } from '@tanstack/ai/adapter-internals'
|
|
3
|
-
import { arrayBufferToBase64, generateId } from '@tanstack/ai-utils'
|
|
4
|
-
import { createOpenAICompatibleClient } from '../utils/client'
|
|
5
|
-
import type { TTSOptions, TTSResult } from '@tanstack/ai'
|
|
6
|
-
import type OpenAI_SDK from 'openai'
|
|
7
|
-
import type { OpenAICompatibleClientConfig } from '../types/config'
|
|
8
|
-
|
|
9
|
-
/**
|
|
10
|
-
* OpenAI-Compatible Text-to-Speech Adapter
|
|
11
|
-
*
|
|
12
|
-
* A generalized base class for providers that implement OpenAI-compatible TTS APIs.
|
|
13
|
-
* Providers can extend this class and only need to:
|
|
14
|
-
* - Set `baseURL` in the config
|
|
15
|
-
* - Lock the generic type parameters to provider-specific types
|
|
16
|
-
* - Override validation methods or request building for provider-specific constraints
|
|
17
|
-
*
|
|
18
|
-
* All methods that validate inputs or build requests are `protected` so subclasses
|
|
19
|
-
* can override them.
|
|
20
|
-
*/
|
|
21
|
-
export class OpenAICompatibleTTSAdapter<
|
|
22
|
-
TModel extends string,
|
|
23
|
-
TProviderOptions extends object = Record<string, any>,
|
|
24
|
-
> extends BaseTTSAdapter<TModel, TProviderOptions> {
|
|
25
|
-
readonly name: string
|
|
26
|
-
|
|
27
|
-
protected client: OpenAI_SDK
|
|
28
|
-
|
|
29
|
-
constructor(
|
|
30
|
-
config: OpenAICompatibleClientConfig,
|
|
31
|
-
model: TModel,
|
|
32
|
-
name: string = 'openai-compatible',
|
|
33
|
-
) {
|
|
34
|
-
super(model, {})
|
|
35
|
-
this.name = name
|
|
36
|
-
this.client = createOpenAICompatibleClient(config)
|
|
37
|
-
}
|
|
38
|
-
|
|
39
|
-
async generateSpeech(
|
|
40
|
-
options: TTSOptions<TProviderOptions>,
|
|
41
|
-
): Promise<TTSResult> {
|
|
42
|
-
const { model, text, voice, format, speed, modelOptions } = options
|
|
43
|
-
|
|
44
|
-
// Validate inputs
|
|
45
|
-
this.validateAudioInput(text)
|
|
46
|
-
this.validateSpeed(speed)
|
|
47
|
-
this.validateInstructions(model, modelOptions)
|
|
48
|
-
|
|
49
|
-
// Build request
|
|
50
|
-
const request: OpenAI_SDK.Audio.SpeechCreateParams = {
|
|
51
|
-
model,
|
|
52
|
-
input: text,
|
|
53
|
-
voice: (voice || 'alloy') as OpenAI_SDK.Audio.SpeechCreateParams['voice'],
|
|
54
|
-
response_format: format,
|
|
55
|
-
speed,
|
|
56
|
-
...modelOptions,
|
|
57
|
-
}
|
|
58
|
-
|
|
59
|
-
try {
|
|
60
|
-
options.logger.request(
|
|
61
|
-
`activity=tts provider=${this.name} model=${model} format=${request.response_format ?? 'default'} voice=${request.voice}`,
|
|
62
|
-
{ provider: this.name, model },
|
|
63
|
-
)
|
|
64
|
-
const response = await this.client.audio.speech.create(request)
|
|
65
|
-
|
|
66
|
-
// Convert response to base64. Buffer is Node-only; use atob fallback in
|
|
67
|
-
// browser/edge runtimes where the SDK can run.
|
|
68
|
-
const arrayBuffer = await response.arrayBuffer()
|
|
69
|
-
const base64 = arrayBufferToBase64(arrayBuffer)
|
|
70
|
-
|
|
71
|
-
const outputFormat = (request.response_format as string) || 'mp3'
|
|
72
|
-
const contentType = this.getContentType(outputFormat)
|
|
73
|
-
|
|
74
|
-
return {
|
|
75
|
-
id: generateId(this.name),
|
|
76
|
-
model,
|
|
77
|
-
audio: base64,
|
|
78
|
-
format: outputFormat,
|
|
79
|
-
contentType,
|
|
80
|
-
}
|
|
81
|
-
} catch (error: unknown) {
|
|
82
|
-
// Narrow before logging: raw SDK errors can carry request metadata
|
|
83
|
-
// (including auth headers) which we must never surface to user loggers.
|
|
84
|
-
options.logger.errors(`${this.name}.generateSpeech fatal`, {
|
|
85
|
-
error: toRunErrorPayload(error, `${this.name}.generateSpeech failed`),
|
|
86
|
-
source: `${this.name}.generateSpeech`,
|
|
87
|
-
})
|
|
88
|
-
throw error
|
|
89
|
-
}
|
|
90
|
-
}
|
|
91
|
-
|
|
92
|
-
protected validateAudioInput(text: string): void {
|
|
93
|
-
if (text.length > 4096) {
|
|
94
|
-
throw new Error('Input text exceeds maximum length of 4096 characters.')
|
|
95
|
-
}
|
|
96
|
-
}
|
|
97
|
-
|
|
98
|
-
protected validateSpeed(speed?: number): void {
|
|
99
|
-
if (speed !== undefined) {
|
|
100
|
-
if (speed < 0.25 || speed > 4.0) {
|
|
101
|
-
throw new Error('Speed must be between 0.25 and 4.0.')
|
|
102
|
-
}
|
|
103
|
-
}
|
|
104
|
-
}
|
|
105
|
-
|
|
106
|
-
protected validateInstructions(
|
|
107
|
-
_model: string,
|
|
108
|
-
_modelOptions?: TProviderOptions,
|
|
109
|
-
): void {
|
|
110
|
-
// Default: no instructions validation — subclasses can override
|
|
111
|
-
}
|
|
112
|
-
|
|
113
|
-
protected getContentType(format: string): string {
|
|
114
|
-
const contentTypes: Record<string, string> = {
|
|
115
|
-
mp3: 'audio/mpeg',
|
|
116
|
-
opus: 'audio/opus',
|
|
117
|
-
aac: 'audio/aac',
|
|
118
|
-
flac: 'audio/flac',
|
|
119
|
-
wav: 'audio/wav',
|
|
120
|
-
pcm: 'audio/pcm',
|
|
121
|
-
}
|
|
122
|
-
return contentTypes[format] || 'audio/mpeg'
|
|
123
|
-
}
|
|
124
|
-
}
|